Skip to main content

rudb_native/
lib.rs

1//! Rudb's single-file columnar snapshot format.
2//!
3//! A committed directory names independently readable column pages. It has two levels: a catalog
4//! directory naming every table in the file, which is what a footer slot points at and what opening
5//! a database reads, and one directory per table under it holding that table's stripes, pages and
6//! statistics. One slot write publishes all of them, so a commit is atomic across tables.
7//!
8//! This version handles scalar columns; the file header has two generation slots so an unfinished
9//! replacement directory cannot hide the last complete one. See
10//! `spec/storage-v3/12-many-tables-in-one-file.md`.
11//!
12//! # Parts and stripes
13//!
14//! A part is one appended chunk, which is a thousand rows, and it is the unit a scan decodes and
15//! hands to the pipeline. A stripe is sixty four parts, and it is the unit the directory describes
16//! and the unit the file is laid out in: one page per column per stripe, holding that column's
17//! sixty four part payloads end to end.
18//!
19//! The two are separate because they are sized by different pressures. A part wants to be small
20//! because it is a vector and vectors live in cache. A stripe wants to be large because everything
21//! the directory holds is per stripe and the directory is one buffer that has to be read and
22//! decoded before a single row can be answered. A hundred million rows of the hundred and five
23//! column ClickBench table is ninety seven thousand parts, and a directory with a page entry and a
24//! pair of bounds per part per column is several hundred megabytes, which is what made that load
25//! fail before this split existed. Sixty four parts to a stripe divides that by sixty four.
26//!
27//! Where the parts of a page start is not in the directory either, for the same reason. Each
28//! stripe writes one index page holding a length and a checksum per part per column, and a reader
29//! preads the sixty four entries belonging to the column it wants. A scan reads the whole column
30//! page once and slices it; a sparse row fetch reads the index entries and then only the part it
31//! needs.
32
33#![forbid(unsafe_code)]
34
35use std::borrow::Cow;
36use std::cmp::{Ordering, Reverse};
37use std::collections::{BTreeMap, HashMap, VecDeque};
38use std::fs::File;
39use std::mem::{size_of, size_of_val};
40use std::path::Path;
41use std::slice;
42use std::sync::atomic::{AtomicBool, AtomicU64, AtomicUsize, Ordering as Atomic};
43use std::sync::{Arc, Condvar, Mutex, OnceLock, PoisonError, Weak};
44
45use rudb_common::bounds::{self, Bound, Op, scaled_as};
46use rudb_common::{Clustering, Error, Field, LogicalType, PhysicalType, Result, Value, Width};
47use rudb_encoding::sequence::Sequence;
48use rudb_encoding::{bitpack, chooser, integer, string};
49use rudb_io::{Filesystem, OpenMode, RealFilesystem};
50use rudb_metrics::{LoadProfile, Stage};
51use rudb_storage::sieve::Sieve;
52use rudb_storage::{Probe, Range, Zone};
53use rudb_vector::string::StringColumn;
54use rudb_vector::validity::Validity;
55use rudb_vector::{Buffer, Chunk, Data, Packed, TextSource, Vector, search_below};
56
57mod distinct;
58pub mod grams;
59pub mod graph;
60pub mod host;
61mod prepare;
62mod projection;
63mod run_projection;
64use prepare::Lent;
65pub mod section;
66pub mod stats;
67mod zones;
68
69pub use prepare::{Building, DICTIONARY_CAP_BYTES, Merged, Merger, Paged, Prepared, Preparer};
70pub use projection::build_sorted_projection;
71pub use run_projection::{RunProjectionPart, RunProjectionScan, build_run_projection};
72pub use section::Section;
73pub use zones::{Common, Stripes, ascending, distincts, widths};
74
75const MAGIC: &[u8; 8] = b"RUDBNV10";
76const DIRECTORY: &[u8; 8] = b"RUDBDI10";
77const CATALOG: &[u8; 8] = b"RUDBCA10";
78const NONZERO_COUNTS: &[u8; 8] = b"RUDBNZ10";
79const AGGREGATE_SUMS: &[u8; 8] = b"RUDBAG10";
80const DISTINCT_COUNTS: &[u8; 8] = b"RUDBDC10";
81const INTEGER_EXTREMES: &[u8; 8] = b"RUDBEX10";
82const COMPLETE_FREQUENCIES: &[u8; 8] = b"RUDBFQ10";
83const DEVICE_CARD: &[u8; 8] = b"RUDBDV10";
84const MAX_CATALOG_FREQUENCIES: usize = 64;
85const FORMAT: u32 = 30;
86
87/// Formats this build can open.
88///
89/// More than one, for the first time, and the reason is spec/graph/10-milestones.md's G1 exit
90/// criterion: a build with the section table in it has to open a file written before the section
91/// table existed, unchanged and without a rewrite. Formats 22 and 23 are those files, and both read
92/// as a table with an empty section table, which is exactly what section 3.1 says a table with no
93/// graph sections is.
94///
95/// All three of the older ones are readable for the same reason. What took the format from 22 to 23
96/// was tags for fourteen more column types, and a file written before that has none of them in it,
97/// so nothing in an older file is a tag this build cannot read. What took it from 23 to 24 is the
98/// section table, which a file written before it simply does not have. What takes it from 24 to 25
99/// is the view section on the end of the catalog, which an older file does not have either, and a
100/// catalog that ends where the tables end reads as a catalog with no views in it. What takes it
101/// from 25 to 26 is that a global dictionary's payload blocks now say where they are, and a file
102/// written before that has them behind one another, which [`open_global_dictionary`] reads by
103/// turning the ends it finds into the same places the newer files name outright. What takes it
104/// from 26 to 27 is that those blocks are written into the file as the load goes, between the
105/// stripes, rather than behind the dictionary's index at the end, so the dictionary's page is the
106/// index and the sorted order and nothing else. A format 26 file has its blocks inside the page,
107/// and the reader tells the two apart by whether the page has room left over for them.
108///
109/// Format 28 adds per-payload-block substring signatures to global string dictionaries. Older
110/// files have no signatures and use the ordinary exact string filter. Format 29 makes each
111/// signature four times as wide, which a dictionary says with [`DICTIONARY_WIDE_GRAMS`], and a
112/// format 28 file is read with the narrow ones it has.
113///
114/// Format 30 lets the catalog end with the device card of the device the file is on, which a
115/// format 29 catalog has no room for and a format 29 reader would call trailing bytes. A catalog
116/// that ends before it is a file with no card, which is every older file.
117///
118/// This is not a general compatibility promise. Seven formats are readable because there was a
119/// specific reason for each, and the list shrinks again the moment the older ones stop being worth
120/// carrying.
121const READABLE: &[u32] = &[22, 23, 24, 25, 26, 27, 28, 29, FORMAT];
122
123const HEADER: u64 = 80;
124const SLOT_BYTES: usize = 28;
125const MAX_PAGE: usize = 256 * 1024 * 1024;
126const MAX_DIRECTORY: usize = 128 * 1024 * 1024;
127const FREQUENCIES_V2: &[u8; 8] = b"RUDBFQ2\0";
128const FREQUENCIES: &[u8; 8] = b"RUDBFQ3\0";
129const FREQUENCIES_SPANS: &[u8; 8] = b"RUDBFQ4\0";
130/// Inline spellings for string entries in the bounded frequency synopsis.
131///
132/// A planner usually asks about one literal such as the empty string. Without this block it opens
133/// a multi-million-value global dictionary and visits the payload blocks of every retained entry
134/// merely to compare that literal with at most 512 heavy hitters. The spellings are already in
135/// memory while the writer sorts the dictionary, so storing this bounded copy makes planning a
136/// directory read and leaves the dictionary unopened.
137const FREQUENCY_TEXTS: &[u8; 8] = b"RUDBFT1\0";
138/// Certified host aggregate state for the version-one anchored replacement expression.
139const HOST_GROUPS: &[u8; 8] = b"RUDBHG1\0";
140/// Exact leading counts for a bounded pair of dictionary-backed grouping keys.
141///
142/// This is a separate optional directory block rather than another frequency format. Readers that
143/// predate it still understand every earlier directory, and a table without a pair worth keeping
144/// writes no block at all.
145const PAIR_FREQUENCIES: &[u8; 8] = b"RUDBPF1\0";
146/// The clustering declaration, written after the frequencies and only when there is one.
147///
148/// No format bump for this, which is the convention the frequency section set in #728: a new
149/// optional trailing section with its own magic leaves every file that does not use it byte for
150/// byte what it was, and the version is bumped for a change to a layout that already exists, as
151/// #1029 did. A file with no declaration is the same bytes this build wrote yesterday.
152///
153/// The width byte in this block gained a fifth value for #1285, for a declaration that leaves the
154/// bucket to the row count, and that did not bump the format either. It is the one case where the
155/// reasoning needs saying out loud, because it is a new value in a layout that already exists
156/// rather than a new section. A build without it reading one of these says `clustering width
157/// tag differs` and refuses the table, which is what that message was written for. Bumping the
158/// format instead would have made every file this build writes unreadable to an older one, whether
159/// it has a declaration in it or not, to warn about a case that only arises when it does.
160const CLUSTERING: &[u8; 8] = b"RUDBCL1\0";
161/// The string columns whose global dictionary stopped taking values partway through the load.
162///
163/// Section 5.5 of the encoding spec: a column whose stripes are nearly all new values, or the
164/// fastest growing one once the dictionaries together pass their cap, stops adding to its
165/// dictionary, and every stripe after that is written plainly. The stripes before keep their codes,
166/// so the dictionary is still written and still decodes them, but it no longer holds every value of
167/// the column, and nothing that reads it as if it did can be trusted: not the distinct count, not
168/// the frequencies, not the sorted order's first and last value, and not the codes as a group key
169/// or a membership index. A reader that finds a column named here decodes its coded pages to plain
170/// strings and answers everything else the way it answers a column with no dictionary.
171///
172/// Same convention as [`CLUSTERING`], written only when a column was demoted, so a file with none
173/// is the bytes it always was. A build that predates it refuses a file that has one with
174/// `directory extension magic differs`, which is the right answer, because that build would trust
175/// the dictionary.
176///
177/// A stripe written after the demotion has no membership index for the column. Its slot in the
178/// stripe is written as a page of no bytes, which no real membership index is, since the smallest
179/// one holds its code count.
180const DEMOTED: &[u8; 8] = b"RUDBDM1\0";
181/// The graph section table, written after the clustering declaration and written even when empty.
182///
183/// Same convention and the same reason as the block above it, with one difference: this one is
184/// always there, so a file written by this build says which sections it has rather than leaving a
185/// reader to infer it from where the bytes ran out. Section 3.1 of the graph spec is what makes
186/// that safe to add without a format bump, because a table with no sections answers every query
187/// the way it did before, only without the graph path.
188const SECTIONS: &[u8; 8] = b"RUDBSE1\0";
189/// The table's primary, unique and foreign keys, written only when it has any.
190///
191/// Same convention as [`CLUSTERING`]: a table with no constraint writes no block, so every file that
192/// has none is the bytes it always was, and a build that predates the block refuses a file with one
193/// with `directory extension magic differs`. That is the right answer, because a build that dropped
194/// the keys would take a row that repeats one.
195const KEYS: &[u8; 8] = b"RUDBKY1\0";
196/// How many bytes of each column's global dictionary live outside its page, written only when any do.
197///
198/// From format 27 a dictionary's payload blocks are written into the file while the load runs, so
199/// they sit between the stripes and the dictionary's page covers only its index and sorted order.
200/// Nothing needs the total to read the file, because the index names every block. It is here for
201/// what a file costs a column, which [`Reader::layout`] and the statistics budget both report, and
202/// which would otherwise lose most of the bytes of every large string column.
203const DICTIONARY_PAYLOADS: &[u8; 8] = b"RUDBDP1\0";
204
205/// The most sections one table's directory may name.
206///
207/// A relationship contributes at most three sections, so this bounds a table at a few thousand
208/// relationships, which is far past anything a schema has. The bound is here so that a torn
209/// directory naming four billion of them is refused at decode rather than turned into an
210/// allocation, the same reason the extent count has one.
211const MAX_SECTIONS: usize = 4096;
212const FREQUENCY_CANDIDATES: usize = 32_768;
213const FREQUENCY_ENTRIES: usize = 512;
214const FREQUENCY_BUILD_RANK: usize = 10;
215const FREQUENCY_ORDINALS: usize = 131_072;
216const MAX_PAIR_FREQUENCIES: usize = 1024;
217/// The most exact heavy-hitter text one column may copy into the directory.
218///
219/// A column with unusually large leading values keeps the old code-only synopsis instead. The
220/// optimization must never turn a valid load into a directory-size failure.
221const FREQUENCY_TEXT_BUDGET: usize = 1024 * 1024;
222/// The most threads the two per column passes at the end of a commit are spread over.
223///
224/// A table like `hits` has ninety numeric columns, so on a machine with more cores than this the
225/// cap is what decides how long the frequencies take rather than the columns are. It is here at all
226/// because each worker holds a candidate table and a decoded part, and a hundred of those at once
227/// on a narrow machine would be worse than waiting.
228const MAX_FREQUENCY_WORKERS: usize = 32;
229
230/// How many threads the passes at the end of a commit are spread over on this machine.
231fn close_workers() -> usize {
232    std::thread::available_parallelism().map_or(1, usize::from).min(MAX_FREQUENCY_WORKERS)
233}
234
235/// How many bytes the columns closing at the same time may hold between them.
236///
237/// Closing a global dictionary decodes every value it holds, sorts them and drops them, and #1356
238/// took the columns one at a time so that five of them decoded at once were not the peak of a load.
239/// A numeric column's frequencies hold a candidate table and, past it, an exact set of its distinct
240/// values that reaches 512 MiB. The two used to run side by side with only the dictionaries under a
241/// bound, and on the ClickBench `hits` 10M load the close took a load that had held 3.1 GB to 4.8
242/// GB. A column is taken while the ones already closing leave room for it under this, and always
243/// when nothing else is closing, so every dictionary of `hits` at 10M rows closes at once and `URL`
244/// at 100M, which is past this alone, still closes on its own.
245const CLOSE_BYTES: usize = 1 << 30;
246
247/// What a numeric column's frequencies hold before its exact distinct set, which is the candidate
248/// table, its recount and the page being read, with room to spare.
249const NUMERIC_CLOSE_BYTES: usize = 4 << 20;
250
251/// The most threads one stripe's encode is spread over.
252///
253/// Higher than the frequency cap because this is the load itself rather than a pass at the end of
254/// it, and the work is one column of sixty four parts, which is large enough that a thread that
255/// takes one is not a thread that was started for nothing. A machine with more cores than this has
256/// the rest of them on the Parquet read, which is still one thread and is the other half of #808.
257const MAX_ENCODE_WORKERS: usize = 32;
258
259/// How much a writer appends before it asks the kernel to start writing it to the device.
260///
261/// Without it every byte of a load waits in the page cache for the sync at the commit, and that
262/// sync was 1.3 to 1.7 s of a ClickBench `hits` 10M load of 8 to 9 s on the 32 core box. With it
263/// the device writes while the load is still encoding. Thirty two megabytes is a few stripes of
264/// `hits`, big enough that the call costs nothing next to the write, and small enough that what
265/// is left for the commit is one stretch.
266const WRITEBACK_STRETCH: u64 = 32 << 20;
267
268/// The most bytes one column of one part may spend on a membership sieve.
269///
270/// A part is a thousand rows, so a filter sized for every one of them being distinct is about
271/// thirteen hundred bytes and this never binds in practice. It is here so that a part that somehow
272/// arrives much wider than a vector cannot put an unbounded index in the file. What does bind is the
273/// rule in `Writer::encode_pages` that a sieve may not be as large as the part it indexes, which is a cap
274/// per column rather than one number for the whole file.
275const SIEVE_BUDGET: usize = 8 * 1024;
276
277/// The most bytes one end of a per part range may spend on a string.
278///
279/// A bound is allowed to be wider than the truth and never narrower, so a long string is cut down to
280/// this many bytes for the low end and cut down and then stepped up for the high end. The reason for
281/// a cap at all is that there are nine hundred and seventy four parts of a hundred and five columns
282/// in a million rows of ClickBench and `URL` runs to hundreds of bytes, so keeping every end whole
283/// would put more in the directory than the skipping is worth. Twenty four bytes is past the point
284/// where two URLs of the same site still look alike.
285const PART_BOUND_BYTES: usize = 24;
286
287fn io(error: std::io::Error) -> Error {
288    Error::io(error.to_string())
289}
290
291fn invalid(message: &str) -> Error {
292    Error::invalid_input(format!("invalid rudb native file: {message}"))
293}
294
295/// Adds a sequence of byte counts without an overflow the caller has to think about.
296fn sum(counts: impl Iterator<Item = u64>) -> u64 {
297    counts.fold(0, u64::saturating_add)
298}
299
300/// One column's span out of a per column list, or zero when the list is shorter than the column.
301fn span_bytes(spans: &[Span], at: usize) -> u64 {
302    spans.get(at).map_or(0, |span| u64::from(span.length))
303}
304
305/// One column's page out of a per column list, or zero when that column has no page at all.
306fn page_bytes(pages: &[Option<Page>], at: usize) -> u64 {
307    pages.get(at).and_then(Option::as_ref).map_or(0, Page::bytes)
308}
309
310/// Everything one column's global dictionary costs the file, its page and the blocks outside it.
311fn dictionary_bytes(table: &Table, at: usize) -> u64 {
312    page_bytes(&table.dictionaries, at)
313        .saturating_add(table.dictionary_payloads.get(at).copied().unwrap_or(0))
314}
315
316/// The xxHash64 of `bytes`, which is what every span this format stores is checked against.
317///
318/// It walks the input as chunks rather than as offsets into it, and that is the only thing about it
319/// worth a comment. The offset form reads `bytes[at..at + 8]`, and neither the slicing nor the
320/// `try_into` behind it can be proved in range by a compiler that does not know where `at` stopped,
321/// so each of the four lanes paid for a bounds check and a length check on every thirty two bytes.
322/// A chunk carries its own length, so both fold away and the loop is the multiplies and rotates it
323/// was meant to be. That loop runs over every byte of every span a query reads, which on ClickBench
324/// 8 is about five percent of the query.
325fn checksum(bytes: &[u8]) -> u64 {
326    seeded_checksum(bytes, 0)
327}
328
329/// A hundred and twenty eight bit name for `bytes`, as two xxHash64 walks under different seeds,
330/// with the format this build writes folded in so that a name made by one format is never taken
331/// for the name of a file in another.
332///
333/// For a caller outside this crate that has to name a file by what went into it, which is what a
334/// Parquet mirror's key is. See the global dictionary's use of the same pair for the arithmetic.
335#[must_use]
336pub fn content_name(bytes: &[u8]) -> u128 {
337    let seed = u64::from(FORMAT);
338    u128::from(seeded_checksum(bytes, seed)) << 64 | u128::from(seeded_checksum(bytes, !seed))
339}
340
341/// [`content_name`] of bytes that arrive in pieces, which gives the same name as the pieces joined.
342///
343/// A Parquet mirror is named for the file's footer, and that is 930 KB on the ten million row
344/// ClickBench file. Read whole to be hashed it is a freed megabyte in every process that opens the
345/// mirror, which the allocator keeps. Read a window at a time it is a window.
346#[derive(Debug, Clone)]
347pub struct ContentNamer {
348    seeds: [u64; 2],
349    lanes: [[u64; 4]; 2],
350    held: [u8; 32],
351    filled: usize,
352    length: u64,
353}
354
355impl Default for ContentNamer {
356    fn default() -> Self {
357        let seed = u64::from(FORMAT);
358        let seeds = [seed, !seed];
359        let lanes = seeds.map(|seed| {
360            [
361                seed.wrapping_add(XXH_P1).wrapping_add(XXH_P2),
362                seed.wrapping_add(XXH_P2),
363                seed,
364                seed.wrapping_sub(XXH_P1),
365            ]
366        });
367        Self { seeds, lanes, held: [0; 32], filled: 0, length: 0 }
368    }
369}
370
371impl ContentNamer {
372    /// Takes the next piece.
373    pub fn update(&mut self, mut bytes: &[u8]) {
374        self.length += bytes.len() as u64;
375        if self.filled > 0 {
376            let take = (32 - self.filled).min(bytes.len());
377            self.held[self.filled..self.filled + take].copy_from_slice(&bytes[..take]);
378            self.filled += take;
379            bytes = &bytes[take..];
380            if self.filled < 32 {
381                return;
382            }
383            let block = self.held;
384            self.lanes.iter_mut().for_each(|lanes| checksum_block(lanes, &block));
385            self.filled = 0;
386        }
387        let mut blocks = bytes.chunks_exact(32);
388        for block in blocks.by_ref() {
389            self.lanes.iter_mut().for_each(|lanes| checksum_block(lanes, block));
390        }
391        let rest = blocks.remainder();
392        self.held[..rest.len()].copy_from_slice(rest);
393        self.filled = rest.len();
394    }
395
396    /// The name of everything taken so far.
397    #[must_use]
398    pub fn finish(&self) -> u128 {
399        let rest = &self.held[..self.filled];
400        let [first, second] = [0, 1].map(|at| {
401            if self.length < 32 {
402                checksum_tail(self.seeds[at].wrapping_add(XXH_P5).wrapping_add(self.length), rest)
403            } else {
404                finish_checksum(self.lanes[at], rest, self.length)
405            }
406        });
407        u128::from(first) << 64 | u128::from(second)
408    }
409}
410
411/// The xxHash64 of `bytes` started from `seed`, which is the same walk with a different beginning.
412///
413/// A seed is here for one caller: a global dictionary decides whether two values are the same by
414/// their hashes rather than by their bytes, and one sixty four bit hash is not enough to do that
415/// with. Twenty million distinct values collide on sixty four bits about once in a hundred thousand
416/// loads, which for a wrong answer is far too often. Two hashes of the same value under different
417/// seeds are independent, so the pair is a hundred and twenty eight bits and the same arithmetic
418/// puts that at around one in 1e24.
419fn seeded_checksum(bytes: &[u8], seed: u64) -> u64 {
420    // Asked for before the loop rather than after it, because a `ChunksExact` settles what it
421    // cannot divide when it is built and hands back the same tail whether it has been walked or not.
422    let mut blocks = bytes.chunks_exact(32);
423    let rest = blocks.remainder();
424    if bytes.len() < 32 {
425        return checksum_tail(seed.wrapping_add(XXH_P5).wrapping_add(bytes.len() as u64), rest);
426    }
427    let mut lanes = [
428        seed.wrapping_add(XXH_P1).wrapping_add(XXH_P2),
429        seed.wrapping_add(XXH_P2),
430        seed,
431        seed.wrapping_sub(XXH_P1),
432    ];
433    for block in blocks.by_ref() {
434        checksum_block(&mut lanes, block);
435    }
436    finish_checksum(lanes, rest, bytes.len() as u64)
437}
438
439const XXH_P1: u64 = 11_400_714_785_074_694_791;
440const XXH_P2: u64 = 14_029_467_366_897_019_727;
441const XXH_P3: u64 = 1_609_587_929_392_839_161;
442const XXH_P4: u64 = 9_650_029_242_287_828_579;
443const XXH_P5: u64 = 2_870_177_450_012_600_261;
444
445fn checksum_round(state: u64, word: u64) -> u64 {
446    state.wrapping_add(word.wrapping_mul(XXH_P2)).rotate_left(31).wrapping_mul(XXH_P1)
447}
448
449fn checksum_word(chunk: &[u8]) -> u64 {
450    u64::from_le_bytes(chunk.try_into().expect("eight checksum bytes"))
451}
452
453/// One thirty two byte block into the four lanes.
454fn checksum_block(lanes: &mut [u64; 4], block: &[u8]) {
455    for (lane, chunk) in lanes.iter_mut().zip(block.chunks_exact(8)) {
456        *lane = checksum_round(*lane, checksum_word(chunk));
457    }
458}
459
460/// The lanes after every whole block, folded together with what was left over and the length.
461fn finish_checksum(lanes: [u64; 4], rest: &[u8], length: u64) -> u64 {
462    let merge = |state: u64, lane: u64| {
463        (state ^ checksum_round(0, lane)).wrapping_mul(XXH_P1).wrapping_add(XXH_P4)
464    };
465    let [one, two, three, four] = lanes;
466    let combined = one
467        .rotate_left(1)
468        .wrapping_add(two.rotate_left(7))
469        .wrapping_add(three.rotate_left(12))
470        .wrapping_add(four.rotate_left(18));
471    let hash = merge(merge(merge(merge(combined, one), two), three), four);
472    checksum_tail(hash.wrapping_add(length), rest)
473}
474
475/// The fewer than thirty two bytes after the last whole block, and the final mix.
476fn checksum_tail(mut hash: u64, mut rest: &[u8]) -> u64 {
477    let mut words = rest.chunks_exact(8);
478    for chunk in words.by_ref() {
479        hash ^= checksum_round(0, checksum_word(chunk));
480        hash = hash.rotate_left(27).wrapping_mul(XXH_P1).wrapping_add(XXH_P4);
481    }
482    rest = words.remainder();
483    if rest.len() >= 4 {
484        let (head, tail) = rest.split_at(4);
485        let quarter = u32::from_le_bytes(head.try_into().expect("four checksum bytes"));
486        hash ^= u64::from(quarter).wrapping_mul(XXH_P1);
487        hash = hash.rotate_left(23).wrapping_mul(XXH_P2).wrapping_add(XXH_P3);
488        rest = tail;
489    }
490    for &byte in rest {
491        hash ^= u64::from(byte).wrapping_mul(XXH_P5);
492        hash = hash.rotate_left(11).wrapping_mul(XXH_P1);
493    }
494    hash ^= hash >> 33;
495    hash = hash.wrapping_mul(XXH_P2);
496    hash ^= hash >> 29;
497    hash = hash.wrapping_mul(XXH_P3);
498    hash ^ (hash >> 32)
499}
500
501/// The checksum of `length` bytes of `file` from `offset`, read [`DIRECTORY_WINDOW`] at a time.
502///
503/// The same xxHash64 as [`checksum`], carried across reads rather than over one buffer, so that a
504/// directory can be checked without all of it being in memory at once. The four lanes take whole
505/// thirty two byte blocks, and a read that ends partway through one keeps the tail for the next.
506fn file_checksum(file: &File, offset: u64, length: usize) -> Result<u64> {
507    walk_checksummed(file, offset, length, DIRECTORY_WINDOW, |_| Ok(()))
508}
509
510/// Reads `length` bytes at `offset` a window at a time, hands each window to `each`, and answers
511/// the checksum of all of them.
512///
513/// `window` is a multiple of thirty two, so every window but the last is whole blocks of the hash
514/// and nothing has to be carried from one read to the next.
515fn walk_checksummed(
516    file: &File,
517    offset: u64,
518    length: usize,
519    window: usize,
520    mut each: impl FnMut(&[u8]) -> Result<()>,
521) -> Result<u64> {
522    debug_assert!(window.is_multiple_of(32) && window > 0, "a window is whole blocks of the hash");
523    if length < 32 {
524        let mut bytes = vec![0; length];
525        read_at(file, offset, &mut bytes)?;
526        each(&bytes)?;
527        return Ok(checksum(&bytes));
528    }
529    let mut lanes = [XXH_P1.wrapping_add(XXH_P2), XXH_P2, 0, 0_u64.wrapping_sub(XXH_P1)];
530    let mut buffer = vec![0; window.min(length)];
531    let mut read = 0;
532    let (mut whole, mut filled) = (0, 0);
533    while read < length {
534        filled = buffer.len().min(length - read);
535        read_at(file, offset + read as u64, &mut buffer[..filled])?;
536        read += filled;
537        each(&buffer[..filled])?;
538        whole = filled / 32 * 32;
539        for block in buffer[..whole].chunks_exact(32) {
540            checksum_block(&mut lanes, block);
541        }
542    }
543    Ok(finish_checksum(lanes, &buffer[whole..filled], length as u64))
544}
545
546#[derive(Debug, Clone, Copy)]
547struct Slot {
548    offset: u64,
549    length: u32,
550    generation: u64,
551    hash: u64,
552}
553
554impl Slot {
555    fn bytes(self) -> [u8; SLOT_BYTES] {
556        let mut result = [0; SLOT_BYTES];
557        result[..8].copy_from_slice(&self.offset.to_le_bytes());
558        result[8..12].copy_from_slice(&self.length.to_le_bytes());
559        result[12..20].copy_from_slice(&self.generation.to_le_bytes());
560        result[20..28].copy_from_slice(&self.hash.to_le_bytes());
561        result
562    }
563
564    fn read(bytes: &[u8]) -> Self {
565        Self {
566            offset: u64::from_le_bytes(bytes[..8].try_into().expect("eight bytes")),
567            length: u32::from_le_bytes(bytes[8..12].try_into().expect("four bytes")),
568            generation: u64::from_le_bytes(bytes[12..20].try_into().expect("eight bytes")),
569            hash: u64::from_le_bytes(bytes[20..28].try_into().expect("eight bytes")),
570        }
571    }
572}
573
574#[derive(Debug, Clone, Copy)]
575struct Page {
576    offset: u64,
577    length: u32,
578    hash: u64,
579}
580
581impl Page {
582    /// How much of the file this page takes, for [`Reader::layout`].
583    fn bytes(&self) -> u64 {
584        u64::from(self.length)
585    }
586}
587
588#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
589enum FrequencyValue {
590    Null,
591    Integer(i128),
592    Code(u32),
593}
594
595/// A table keyed by the sixty four bits of the values the numeric frequency pass counts.
596///
597/// Every integer of every numeric column goes through one of these at least once when a table
598/// closes, and with the standard hasher that was a fifth of the close on its own, all of it SipHash
599/// guarding against an attacker who would have to choose the rows of the file being written.
600type FrequencyMap<V> = HashMap<u64, V, Spread>;
601
602/// The first pass of [`Writer::numeric_frequency`]: a Misra-Gries candidate table keyed by a value's
603/// sixty four bits, with the null counted beside it.
604///
605/// The table is an open addressed one of its own rather than a `HashMap`. On a column that is near
606/// unique, which `hits` has a dozen of, nearly every row is a value the table has not seen, and a
607/// `HashMap` spent a lookup and then a second hash and probe to insert it, and a `retain` over every
608/// bucket each time the table filled. Those were 6 percent of the CPU of loading the 10m ClickBench
609/// file, and the slowest of those columns decided how long the whole frequency step took. Here a
610/// value is found or given the empty slot it stopped at in one probe, and a decrement rebuilds the
611/// table from the few candidates that outlive it.
612///
613/// What the table holds after a stream of rows is the same set of counts either way, since that is
614/// fixed by the algorithm and not by where the counts live.
615#[derive(Debug)]
616struct Candidates {
617    /// A power of two number of slots, at most half of them in use. A count of zero is an empty
618    /// slot, which no candidate ever is, because one whose count reaches zero is dropped.
619    slots: Vec<Candidate>,
620    held: usize,
621    nulls: u32,
622    decrements: u64,
623    /// The candidates that outlive a decrement, kept so that each decrement is not an allocation.
624    survivors: Vec<Candidate>,
625}
626
627/// One slot of [`Candidates`], the value's bits beside its count so a probe reads one line.
628#[derive(Debug, Default, Clone, Copy)]
629struct Candidate {
630    bits: u64,
631    count: u32,
632}
633
634/// The slots a candidate table starts with, grown by doubling as it fills.
635const FIRST_CANDIDATE_SLOTS: usize = 64;
636
637impl Default for Candidates {
638    fn default() -> Self {
639        Self {
640            slots: vec![Candidate::default(); FIRST_CANDIDATE_SLOTS],
641            held: 0,
642            nulls: 0,
643            decrements: 0,
644            survivors: Vec::new(),
645        }
646    }
647}
648
649impl Candidates {
650    /// Counts `times` rows of `bits` and ends in the state `times` rows counted one at a time would.
651    ///
652    /// A value already held, or one there is room to hold, takes the whole run at once, because
653    /// every row after the first would find it held. A value the full table turns away goes a row
654    /// at a time, because each of its rows decrements every candidate and one of those decrements
655    /// can free the place the next row takes.
656    fn add(&mut self, bits: Option<u64>, mut times: u32) {
657        while times > 0 {
658            let room = self.held + usize::from(self.nulls != 0) < FREQUENCY_CANDIDATES;
659            match bits {
660                Some(bits) => {
661                    let (at, found) = self.find(bits);
662                    if found {
663                        self.slots[at].count = self.slots[at].count.saturating_add(times);
664                        return;
665                    }
666                    if room {
667                        self.place(at, bits, times);
668                        return;
669                    }
670                }
671                None if self.nulls != 0 => {
672                    self.nulls = self.nulls.saturating_add(times);
673                    return;
674                }
675                None if room => {
676                    self.nulls = times;
677                    return;
678                }
679                None => {}
680            }
681            self.decrement();
682            times -= 1;
683        }
684    }
685
686    /// The slot holding `bits` and `true`, or the empty slot a search for it stopped at and `false`.
687    fn find(&self, bits: u64) -> (usize, bool) {
688        let mask = self.slots.len() - 1;
689        let mut at = home(bits, self.slots.len());
690        loop {
691            let slot = self.slots[at];
692            if slot.count == 0 {
693                return (at, false);
694            }
695            if slot.bits == bits {
696                return (at, true);
697            }
698            at = (at + 1) & mask;
699        }
700    }
701
702    /// Where `bits` is held, for the recount, which reads the table without changing it.
703    fn position(&self, bits: u64) -> Option<usize> {
704        match self.find(bits) {
705            (at, true) => Some(at),
706            (_, false) => None,
707        }
708    }
709
710    /// Puts a new candidate in the empty slot `at`, which a search for it just stopped at, doubling
711    /// the table first when that would fill more than half of it.
712    fn place(&mut self, at: usize, bits: u64, count: u32) {
713        let at = if (self.held + 1) * 2 > self.slots.len() {
714            let wider = self.slots.len() * 2;
715            let old = std::mem::replace(&mut self.slots, vec![Candidate::default(); wider]);
716            for slot in old.into_iter().filter(|slot| slot.count != 0) {
717                let (to, _) = self.find(slot.bits);
718                self.slots[to] = slot;
719            }
720            self.find(bits).0
721        } else {
722            at
723        };
724        self.slots[at] = Candidate { bits, count };
725        self.held += 1;
726    }
727
728    /// Takes one from every candidate and the null, dropping the ones that reach zero.
729    fn decrement(&mut self) {
730        let mut survivors = std::mem::take(&mut self.survivors);
731        survivors.clear();
732        survivors.extend(
733            self.slots
734                .iter()
735                .filter(|slot| slot.count > 1)
736                .map(|slot| Candidate { bits: slot.bits, count: slot.count - 1 }),
737        );
738        self.slots.fill(Candidate::default());
739        self.held = survivors.len();
740        for &slot in &survivors {
741            let (at, _) = self.find(slot.bits);
742            self.slots[at] = slot;
743        }
744        self.survivors = survivors;
745        self.nulls = self.nulls.saturating_sub(1);
746        self.decrements = self.decrements.saturating_add(1);
747    }
748
749    /// Every candidate's bits and count, in no particular order.
750    fn pairs(&self) -> impl Iterator<Item = (u64, u32)> + '_ {
751        self.slots.iter().filter(|slot| slot.count != 0).map(|slot| (slot.bits, slot.count))
752    }
753}
754
755/// The slot a search for `bits` starts at in a table of `slots`, a power of two.
756///
757/// The top bits of a multiply by the golden ratio, which every bit of the value reaches, so a
758/// timestamp column whose values are all multiples of a million still spreads over the table.
759fn home(bits: u64, slots: usize) -> usize {
760    (bits.wrapping_mul(0x9E37_79B9_7F4A_7C15) >> (64 - slots.trailing_zeros())) as usize
761}
762
763/// Equal rows in a row, gathered so they are counted once.
764#[derive(Debug, Default)]
765struct Run {
766    bits: Option<u64>,
767    times: u32,
768}
769
770impl Run {
771    /// Adds one row, and hands back the run it ended if it was not the same value.
772    fn push(&mut self, bits: Option<u64>) -> Option<(Option<u64>, u32)> {
773        if self.times != 0 && self.bits == bits && self.times < u32::MAX {
774            self.times += 1;
775            return None;
776        }
777        let ended = self.take();
778        self.bits = bits;
779        self.times = 1;
780        ended
781    }
782
783    /// The run being gathered, if there is one, leaving none.
784    fn take(&mut self) -> Option<(Option<u64>, u32)> {
785        let times = std::mem::take(&mut self.times);
786        (times != 0).then_some((self.bits, times))
787    }
788}
789
790/// Builds the hasher for [`FrequencyMap`].
791#[derive(Debug, Default, Clone, Copy)]
792struct Spread;
793
794impl std::hash::BuildHasher for Spread {
795    type Hasher = SpreadHasher;
796
797    fn build_hasher(&self) -> SpreadHasher {
798        SpreadHasher(0)
799    }
800}
801
802/// Folds each word in with a full width multiply whose two halves are xored together.
803///
804/// A plain multiply leaves the low bits of the hash as poor as the low bits of the key, and the
805/// table picks its bucket from the low bits, so a timestamp column, whose values are all multiples
806/// of a million microseconds, would pile into a sixty fourth of the buckets. Folding the high half
807/// of the product back in is what gives the low bits the whole word.
808#[derive(Debug)]
809struct SpreadHasher(u64);
810
811impl SpreadHasher {
812    fn mix(&mut self, word: u64) {
813        let product = u128::from(self.0 ^ word) * 0x9E37_79B9_7F4A_7C15_u128;
814        self.0 = (product as u64) ^ ((product >> 64) as u64);
815    }
816}
817
818impl std::hash::Hasher for SpreadHasher {
819    fn write(&mut self, bytes: &[u8]) {
820        for part in bytes.chunks(8) {
821            let mut word = [0; 8];
822            word[..part.len()].copy_from_slice(part);
823            self.mix(u64::from_le_bytes(word));
824        }
825    }
826
827    fn write_u32(&mut self, value: u32) {
828        self.mix(u64::from(value));
829    }
830
831    fn write_u64(&mut self, value: u64) {
832        self.mix(value);
833    }
834
835    fn write_i128(&mut self, value: i128) {
836        self.mix(value as u64);
837        self.mix((value >> 64) as u64);
838    }
839
840    fn write_isize(&mut self, value: isize) {
841        self.mix(value as u64);
842    }
843
844    fn finish(&self) -> u64 {
845        self.0
846    }
847}
848
849#[derive(Debug, Clone)]
850struct FrequencyEntry {
851    value: FrequencyValue,
852    count: u64,
853}
854
855/// Exact leading frequencies for one column.
856///
857/// Values outside `entries` occur at most `omitted_max` times. This lets a count-descending TopN
858/// use the synopsis only when its last winner is strictly above every omitted value.
859#[derive(Debug, Clone)]
860struct FrequencySummary {
861    entries: Vec<FrequencyEntry>,
862    omitted_max: u64,
863    ordinals: Vec<u64>,
864    ordinal_entries: Vec<u16>,
865}
866
867#[derive(Debug, Clone)]
868struct PairFrequencyEntry {
869    first_entry: u16,
870    second: Option<u32>,
871    count: u64,
872}
873
874/// Exact leading counts for one numeric frequency anchor and one stable string code space.
875///
876/// `omitted_max` covers both first-key values outside the numeric synopsis and pairs below the
877/// retained prefix. A TopN may therefore use the entries only when its boundary strictly exceeds
878/// this number.
879#[derive(Debug, Clone)]
880struct PairFrequencySummary {
881    first: u16,
882    second: u16,
883    entries: Vec<PairFrequencyEntry>,
884    omitted_max: u64,
885}
886
887/// One column's frequency synopsis, in memory or left where it is in the file.
888///
889/// A writer holds what it counted. A reader leaves every synopsis in the file and reads one back
890/// when a query asks about its column, because they are the largest thing in a directory once they
891/// are decoded, forty eight bytes an entry and nearly twenty thousand entries over `hits`, and
892/// most queries ask about none of them. New directories give each synopsis a checked span, so
893/// opening an unrelated projection need not parse its ordinals.
894#[derive(Debug, Clone)]
895enum Frequencies {
896    Held(FrequencySummary),
897    /// Where the synopsis sits, and whether it was written with the value of each ordinal, which
898    /// is what the directory's frequency magic says and the synopsis itself does not.
899    Stored {
900        span: Span,
901        values: bool,
902        entries: usize,
903    },
904}
905
906/// The values one column's frequency synopsis lists, with a bound on everything it left out.
907///
908/// What [`Reader::frequency_prefix`] answers. The counts are exact, and `omitted_max` is how many
909/// rows any value not in the list can hold, which is zero when nothing was left out at all.
910#[derive(Debug, Clone)]
911pub struct FrequencyPrefix {
912    /// Every value the synopsis lists, with the number of rows holding it, count descending.
913    pub entries: Vec<(Value, u64)>,
914    /// How many rows the most common value outside the list holds, and zero for a complete list.
915    pub omitted_max: u64,
916}
917
918/// Sparse row ordinals covered by a numeric frequency candidate set.
919#[derive(Debug, Clone, PartialEq)]
920pub struct FrequencyOccurrences {
921    /// Upper bound for the frequency of every value absent from the fetched rows.
922    pub omitted_max: u64,
923    /// Table-wide row ordinals in ascending order.
924    pub ordinals: Vec<u64>,
925    /// The retained heavy-hitter values named by `anchor_indices`.
926    pub anchors: Vec<Value>,
927    /// The index in `anchors` at each ordinal, or empty for a legacy FQ2 directory.
928    pub anchor_indices: Vec<u16>,
929}
930
931/// Exact grouped counts for a pair of values, in descending count order.
932pub type PairFrequencyCounts = Vec<(Vec<Value>, u64)>;
933
934/// Where one column's page for one stripe sits in the file.
935///
936/// A column page has no checksum of its own because every part inside it carries one, and the
937/// stripe's index page holds those. Checking a part on the way out of the page covers exactly the
938/// bytes a reader is about to decode, and covers them once whether the reader took the whole page
939/// or pulled one part out of the middle of it.
940#[derive(Debug, Clone, Copy, Default)]
941struct Span {
942    offset: u64,
943    length: u32,
944}
945
946/// One optional page for each column of a stripe, holding only the pages that are there.
947///
948/// A stripe has three of these, the membership, sieve and part range pages. As a
949/// `Vec<Option<Page>>` each was thirty two bytes a column whether the page was there or not, and
950/// over the ten million rows of `hits` that is half a megabyte at open for 7171 pages out of 16380
951/// slots. Kept sparse and packed, a page that is there is twenty four bytes and one that is not is
952/// nothing.
953#[derive(Debug, Clone, Default)]
954struct Pages {
955    columns: usize,
956    held: Box<[StripePage]>,
957}
958
959/// A page and the column it is for, packed so that the column sits where the padding was.
960#[derive(Debug, Clone, Copy)]
961struct StripePage {
962    offset: u64,
963    hash: u64,
964    length: u32,
965    column: u32,
966}
967
968impl Pages {
969    /// The pages of `columns` columns, one slot each in column order.
970    fn from_slots(slots: Vec<Option<Page>>) -> Result<Self> {
971        let mut held = Vec::with_capacity(slots.iter().flatten().count());
972        for (column, page) in slots.iter().enumerate() {
973            if let Some(page) = page {
974                let column =
975                    u32::try_from(column).map_err(|_| invalid("too many columns for a page"))?;
976                held.push(StripePage {
977                    offset: page.offset,
978                    hash: page.hash,
979                    length: page.length,
980                    column,
981                });
982            }
983        }
984        Ok(Self { columns: slots.len(), held: held.into_boxed_slice() })
985    }
986
987    /// The page of one column, if it has one.
988    fn get(&self, column: usize) -> Option<Page> {
989        let at = self.held.binary_search_by_key(&column, |placed| placed.column as usize).ok()?;
990        let placed = self.held[at];
991        Some(Page { offset: placed.offset, length: placed.length, hash: placed.hash })
992    }
993
994    /// One slot per column, in column order, the way the directory writes them.
995    fn slots(&self) -> impl Iterator<Item = Option<Page>> + '_ {
996        (0..self.columns).map(|column| self.get(column))
997    }
998
999    /// How much of the file one column's page takes, or zero when it has none.
1000    fn bytes(&self, column: usize) -> u64 {
1001        self.get(column).map_or(0, |page| page.bytes())
1002    }
1003}
1004
1005/// One independently readable stripe of a table.
1006#[derive(Debug, Clone)]
1007pub struct Stripe {
1008    rows: usize,
1009    /// Rows in each part, in source order. Kept in the directory so that mapping a row ordinal to a
1010    /// part, which every sparse fetch does, never reads the file.
1011    parts: Vec<u32>,
1012    /// The index page: one section per column, holding a length and a checksum for every part and
1013    /// then a checksum of the section itself, so that a reader can pread one column's section and
1014    /// still know it is intact.
1015    index: Span,
1016    pages: Vec<Span>,
1017    memberships: Pages,
1018    /// One page per column holding the membership sieve of every part of the stripe, for the
1019    /// columns that have one. A column whose parts all declined a sieve has no page at all.
1020    sieves: Pages,
1021    /// One page per column holding the two ends and the null count of every part of the stripe.
1022    ///
1023    /// The stripe's own `zone` below covers sixty four times as many rows, and on a column that is
1024    /// not the one the rows are ordered by that is the difference between skipping half the file and
1025    /// skipping all but three percent of it. On ClickBench 24 the cutoff the answer settles at
1026    /// leaves eight stripes of sixteen alive and thirty parts of nine hundred and seventy four.
1027    ///
1028    /// A page per column rather than one page for the stripe, so that a query that compares one
1029    /// column reads the ends of that column and not of the hundred and four beside it. Read lazily
1030    /// for the same reason, like the sieves.
1031    part_ranges: Pages,
1032    zone: Zone,
1033}
1034
1035impl Stripe {
1036    /// Number of rows in this stripe.
1037    #[must_use]
1038    pub fn rows(&self) -> usize {
1039        self.rows
1040    }
1041
1042    /// Number of parts in this stripe.
1043    #[must_use]
1044    pub fn parts(&self) -> usize {
1045        self.parts.len()
1046    }
1047
1048    /// The two ends and the null count of every column over the whole stripe.
1049    ///
1050    /// In the directory and so in memory, which is what makes it the one a planner can ask. The
1051    /// finer ones are a page per column per stripe in the file, read by [`Reader::skips`] when a
1052    /// scan wants to know which parts to open.
1053    #[must_use]
1054    pub fn zone(&self) -> &Zone {
1055        &self.zone
1056    }
1057}
1058
1059/// The committed table directory.
1060#[derive(Debug, Clone)]
1061pub struct Table {
1062    name: String,
1063    fields: Vec<Field>,
1064    stripes: Vec<Stripe>,
1065    rows: usize,
1066    dictionaries: Vec<Option<Page>>,
1067    /// Bytes of each column's dictionary payload that are outside its page, which is all of them
1068    /// from format 27 and none of them before. See [`DICTIONARY_PAYLOADS`].
1069    ///
1070    /// Empty rather than a row of zeros on a table that has none, and read with `get` for that
1071    /// reason, so that a table built by hand in a test does not have to know about it.
1072    dictionary_payloads: Vec<u64>,
1073    /// The columns whose dictionary stopped taking values partway through the load, see
1074    /// [`DEMOTED`].
1075    ///
1076    /// Empty rather than a row of `false` on a table that has none, and read with `get`, for the
1077    /// same reason `dictionary_payloads` is.
1078    demoted: Vec<bool>,
1079    frequencies: Vec<Option<Frequencies>>,
1080    pair_frequencies: Vec<PairFrequencySummary>,
1081    /// String spellings aligned with each column's frequency entries.
1082    ///
1083    /// Empty for files written before `RUDBFT1`. A `None` entry is the null frequency entry; every
1084    /// code entry in a column named by the block has its exact bytes here.
1085    frequency_texts: Vec<Vec<Option<Vec<u8>>>>,
1086    /// Exact candidate host aggregates and an upper bound for every omitted host.
1087    host_groups: Option<host::HostSummary>,
1088    /// How many distinct values each column holds, for the columns that know.
1089    ///
1090    /// A dictionary entry is made the first time a value is seen and nothing ever removes one, so
1091    /// the size of the dictionary is the number of distinct values in the column. That is the whole
1092    /// story for a column with no null in it, and the wrong number by one for a column with a null
1093    /// in it, because a null row is written as the code for the empty string and makes an entry the
1094    /// dictionary would not otherwise have. The writer knows which case it is, since it counts the
1095    /// non-null rows that use each code while it builds the frequency summary, and the reader cannot
1096    /// work it out from the dictionary alone. So the writer settles it here.
1097    distincts: Vec<Option<u64>>,
1098    /// The order the rows of this table are meant to be stored in, if anybody declared one.
1099    ///
1100    /// A declaration and not a measurement. Nothing here checks that the stripes actually arrived
1101    /// in this order, and the reason it is worth storing anyway is that the order is the only thing
1102    /// about a table that a rewrite destroys without anybody noticing. The fragment ranges prune on
1103    /// whatever order the rows came in, so a table loaded sorted prunes and the same table after a
1104    /// checkpoint that did not know to keep the order quietly stops pruning and nothing says why.
1105    clustering: Option<Clustering>,
1106    /// The file generation of the commit that last wrote this table's column pages.
1107    ///
1108    /// This is what spec/graph/03-the-file-format.md section 3.2 calls the table generation, and
1109    /// the definition is deliberately about the pages rather than about the directory. A graph
1110    /// section is a restatement of a column in terms of row ids, so what invalidates one is the
1111    /// rows being renumbered, and nothing else. Adding a second table to the file, or attaching a
1112    /// section to this one, commits a new file generation without touching a single row of this
1113    /// table, and a definition that moved with those would declare every section in the file stale
1114    /// for no reason.
1115    ///
1116    /// Zero on a table written before format 23, where nothing recorded it. Real generations start
1117    /// at one, so zero can never match a section's stamp, and a table from format 22 has no
1118    /// sections for it to match anyway.
1119    generation: u64,
1120    /// The graph sections this table carries, per spec/graph/03-the-file-format.md section 3.2.
1121    ///
1122    /// Empty for every table written before the section table existed, and empty is not a
1123    /// degraded state: section 3.1 says deleting every graph section from a file changes no answer,
1124    /// only the time, so a table with none here answers every query the same way and slower. That
1125    /// is what lets this field arrive without a migration.
1126    sections: Vec<Section>,
1127    /// The keys and foreign keys the table was created with, which the file keeps so that a
1128    /// reopened table refuses the rows it refused before.
1129    constraints: Constraints,
1130}
1131
1132/// A table's primary, unique and foreign keys, as the file stores them.
1133///
1134/// Columns are places in the table, and a foreign key names the table it points at by name alone,
1135/// because every table in one file is in one schema.
1136#[derive(Debug, Clone, Default, PartialEq, Eq)]
1137pub struct Constraints {
1138    /// Each key's columns, and whether it is the primary key rather than a unique one.
1139    pub keys: Vec<(Vec<u16>, bool)>,
1140    /// Each foreign key.
1141    pub foreign: Vec<StoredForeign>,
1142}
1143
1144impl Constraints {
1145    /// Whether there is nothing here, which is what writes no block.
1146    #[must_use]
1147    pub fn is_empty(&self) -> bool {
1148        self.keys.is_empty() && self.foreign.is_empty()
1149    }
1150}
1151
1152/// One `FOREIGN KEY`, as the file stores it.
1153#[derive(Debug, Clone, PartialEq, Eq)]
1154pub struct StoredForeign {
1155    /// The columns of this table.
1156    pub columns: Vec<u16>,
1157    /// The table it points at.
1158    pub table: String,
1159    /// The columns of that table, paired with `columns` one for one.
1160    pub referenced: Vec<u16>,
1161}
1162
1163impl Table {
1164    /// The SQL table name held by this snapshot.
1165    #[must_use]
1166    pub fn name(&self) -> &str {
1167        &self.name
1168    }
1169
1170    /// Columns in their SQL order.
1171    #[must_use]
1172    pub fn fields(&self) -> &[Field] {
1173        &self.fields
1174    }
1175
1176    /// Committed row count.
1177    #[must_use]
1178    pub fn rows(&self) -> usize {
1179        self.rows
1180    }
1181
1182    /// Independently readable stripes.
1183    #[must_use]
1184    pub fn stripes(&self) -> &[Stripe] {
1185        &self.stripes
1186    }
1187
1188    /// The order the rows are meant to be stored in, if this table was declared with one.
1189    #[must_use]
1190    pub fn clustering(&self) -> Option<&Clustering> {
1191        self.clustering.as_ref()
1192    }
1193
1194    /// The keys and foreign keys this table was created with.
1195    #[must_use]
1196    pub fn constraints(&self) -> &Constraints {
1197        &self.constraints
1198    }
1199
1200    /// The generation every section of this table is judged against.
1201    ///
1202    /// See the field. A caller deciding whether to read a section asks [`Section::usable`] with
1203    /// this.
1204    #[must_use]
1205    pub fn generation(&self) -> u64 {
1206        self.generation
1207    }
1208
1209    /// Every graph section this table names, including the kinds this build does not know.
1210    ///
1211    /// Including them is the point. A caller that wants only the ones it can use asks
1212    /// [`Section::usable`], and a caller rewriting the directory carries the rest through, so a
1213    /// file opened by an older build and written again does not silently lose a section that build
1214    /// had no name for.
1215    #[must_use]
1216    pub fn sections(&self) -> &[Section] {
1217        &self.sections
1218    }
1219}
1220
1221/// One table's line in the catalog directory.
1222///
1223/// The small level of the two. It holds what opening a database needs and nothing else: the name to
1224/// bind, the shape to plan against, the row count, and where the table's own directory sits. A file
1225/// of eight tables is eight of these, and reading them costs the same whether the tables hold a
1226/// thousand rows or a billion.
1227///
1228/// The name, the fields and the row count are repeated here rather than pointed at inside the table
1229/// directory, which is the entire point of having two levels. A catalog that pointed at them would
1230/// have to read every table directory at open to answer what tables there are, which is the cost
1231/// this level exists to avoid.
1232#[derive(Debug, Clone)]
1233struct Entry {
1234    name: String,
1235    fields: Vec<Field>,
1236    rows: usize,
1237    /// Where this table's own directory sits, with the checksum it was committed under.
1238    directory: Page,
1239    /// Legacy nonzero counts. New files leave these empty and derive filtered counts from
1240    /// reusable column frequencies when a query needs them.
1241    nonzero: Vec<Option<u64>>,
1242    /// Exact sum and non-null count for signed integer columns.
1243    aggregates: Vec<Option<(i128, u64)>>,
1244    /// Exact non-null distinct values when the writer finished counting the column.
1245    distincts: Vec<Option<u64>>,
1246    /// Exact integer or date bounds; the inner `None` means every row is null.
1247    extremes: Vec<StoredIntegerExtremes>,
1248    /// Complete bounded numeric frequencies, including NULL when present.
1249    frequencies: Vec<StoredNumericFrequencies>,
1250}
1251
1252type StoredIntegerExtremes = Option<Option<(i128, i128)>>;
1253type StoredNumericFrequencies = Option<NumericFrequencies>;
1254
1255/// One view's line in the catalog directory.
1256///
1257/// A view has no pages, so unlike a table it is entirely here and there is no second level under it.
1258/// What it is made of is text: the body the binder binds again at every reference, and the whole
1259/// statement written back out, which is what `duckdb_views()` reports and nothing else reads.
1260///
1261/// The columns are a cache and they are written down anyway, which is worth saying out loud because
1262/// a cache in a file looks like a mistake. It is what the pin does. Create a view on a file, open
1263/// the file again in another process, and `duckdb_views()` answers `column_count` and `is_bound`
1264/// true without anything having bound the body, so the list survived the write. Not writing it
1265/// would answer null and false there, and the only way back would be to bind every view at open,
1266/// which is the thing the cache exists to avoid.
1267#[derive(Debug, Clone, PartialEq, Eq)]
1268pub struct ViewEntry {
1269    /// The view's own name, without the schema, the way a table entry holds its name.
1270    pub name: String,
1271    /// The query the view stands for, as the text that was written.
1272    pub sql: String,
1273    /// The whole `CREATE VIEW` written back out.
1274    pub statement: String,
1275    /// The column names the statement gave, which rename a prefix of what the body produces.
1276    pub aliases: Vec<String>,
1277    /// The columns the last bind of the body produced.
1278    pub columns: Vec<Field>,
1279}
1280
1281/// Where one column's bytes went, taken from the directory rather than by reading pages.
1282#[derive(Debug, Clone)]
1283pub struct ColumnLayout {
1284    /// The column's name, so a report does not have to carry the field list beside this.
1285    pub name: String,
1286    /// The type, spelled the way the catalog spells it.
1287    pub kind: String,
1288    /// Every stripe's page of this column added up, which is the encoded data itself.
1289    pub pages: u64,
1290    /// Every stripe's exact code membership page for this column.
1291    pub memberships: u64,
1292    /// Every stripe's membership sieve page for this column.
1293    pub sieves: u64,
1294    /// Every stripe's per part range page for this column.
1295    pub part_ranges: u64,
1296    /// The table wide dictionary of this column, if it has one.
1297    pub dictionary: u64,
1298}
1299
1300impl ColumnLayout {
1301    /// Everything this column costs, which is what the file would lose if the column went.
1302    #[must_use]
1303    pub fn total(&self) -> u64 {
1304        self.pages
1305            .saturating_add(self.memberships)
1306            .saturating_add(self.sieves)
1307            .saturating_add(self.part_ranges)
1308            .saturating_add(self.dictionary)
1309    }
1310}
1311
1312/// Where a whole file's bytes went.
1313///
1314/// Every number here comes out of the committed directory, so taking it costs one directory read
1315/// however large the file is. That is the point: a 45 GB table has to be able to say where it went
1316/// without being read, or nobody will ask.
1317///
1318/// The parts that are not a column are kept apart rather than shared out over the columns. The
1319/// stripe index page holds a section per column and could be split, and the directory and the
1320/// header cannot be, so splitting one of the three and not the others would read as if the columns
1321/// accounted for everything. They do not, and the gap is the thing worth looking at.
1322#[derive(Debug, Clone)]
1323pub struct Layout {
1324    /// The size of the file on disk.
1325    pub file: u64,
1326    /// Committed rows.
1327    pub rows: usize,
1328    /// Committed stripes.
1329    pub stripes: usize,
1330    /// Committed parts, which is how many chunks a scan reads.
1331    pub parts: usize,
1332    /// One entry per column, in the table's column order.
1333    pub columns: Vec<ColumnLayout>,
1334    /// Every stripe's index page, which carries a length and a checksum for every part of every
1335    /// column and is charged per stripe rather than per column.
1336    pub indexes: u64,
1337    /// The committed directory itself, the one that was read to build this.
1338    pub directory: u64,
1339    /// The fixed header, which holds the magic, the format and the two directory slots.
1340    pub header: u64,
1341}
1342
1343impl Layout {
1344    /// Everything the columns cost together.
1345    #[must_use]
1346    pub fn columns_total(&self) -> u64 {
1347        self.columns.iter().map(ColumnLayout::total).fold(0, u64::saturating_add)
1348    }
1349
1350    /// What the file holds that this does not account for.
1351    ///
1352    /// A committed file is written once and never rewritten in place, so an earlier directory and
1353    /// the pages of an earlier snapshot are still in it. That is the honest place for them: they
1354    /// are bytes on disk that no column owns.
1355    #[must_use]
1356    pub fn unaccounted(&self) -> u64 {
1357        self.file
1358            .saturating_sub(self.columns_total())
1359            .saturating_sub(self.indexes)
1360            .saturating_sub(self.directory)
1361            .saturating_sub(self.header)
1362    }
1363}
1364
1365/// How one part of one column is stored, which is one row of `pragma_storage_info`.
1366///
1367/// Everything here is read off the file rather than worked out from the schema, because the whole
1368/// question this answers is what the encoder chose, and the encoder chooses per part. Two files
1369/// holding the same rows in a different order give different answers and that difference is the
1370/// reason to ask.
1371///
1372/// The encoding costs a read of the column's page, so this is not free the way [`Layout`] is. It is
1373/// one read per column per stripe rather than one per part, because a part is a few kilobytes out
1374/// of a page that is a quarter of a megabyte.
1375#[derive(Debug, Clone)]
1376pub struct StoredPart {
1377    /// Which stripe the part belongs to.
1378    pub stripe: usize,
1379    /// Which part of that stripe it is, counting from zero inside the stripe.
1380    pub part: usize,
1381    /// The table wide row number the part starts at.
1382    pub row: usize,
1383    /// How many rows it holds.
1384    pub rows: usize,
1385    /// What the encoder made of it, as a line of text like `DICT(PACKED, PACKED)`.
1386    pub encoding: String,
1387    /// The stored bytes of the part, which is what it costs in the file.
1388    pub bytes: u64,
1389    /// Where in the file the column page holding this part starts.
1390    pub page: u64,
1391    /// Where in that page the part starts.
1392    pub offset: u64,
1393    /// The smallest value the part holds, when the stored ranges say.
1394    pub low: Option<Value>,
1395    /// The largest, same.
1396    pub high: Option<Value>,
1397    /// How many of its rows are null, when the stored ranges say.
1398    pub nulls: Option<usize>,
1399}
1400
1401/// Seeds the second hash a global dictionary tells its values apart by.
1402///
1403/// Any value that is not zero does, since zero is the seed [`checksum`] already uses and the point
1404/// is only that the two hashes of one value are not the same number. This one is the fractional part
1405/// of the golden ratio in sixty four bits, which is the constant everything else here is built out
1406/// of and is as good a nothing-up-my-sleeve number as any.
1407const DICTIONARY_CHECK_SEED: u64 = 11_400_714_819_323_198_485;
1408
1409/// One column's table wide dictionary while the load is running.
1410///
1411/// The thing to understand about this is what it does not hold. A dictionary of `URL` at a hundred
1412/// million ClickBench rows has about eighteen million distinct values and 1.3 GB of bytes in them,
1413/// and five columns like it are twelve of the seventeen gigabytes a load of `hits` peaks at. So the
1414/// bytes are not kept. A value's bytes go into [`GlobalDictionary::filling`], and when that reaches
1415/// [`TEXT_PAYLOAD_VALUES`] values the block is sealed, handed out at the end of the merge that
1416/// sealed it to be encoded with the stripe's pages, and never seen in that form again. What is left
1417/// is the encoded block, which is two to three times smaller, and that is the same bytes the file
1418/// is going to hold anyway.
1419///
1420/// Two things needed the raw bytes and neither needs them now. Deciding whether a value has been
1421/// seen before was a hash lookup and then a comparison of the bytes, and is now a hash lookup and a
1422/// comparison of a second hash under a different seed, which is [`DICTIONARY_CHECK_SEED`] and the
1423/// argument for why that is sound. Sorting the values at the end needed all of them at once, and
1424/// now reads the blocks back through [`GlobalDictionary::decoded`] one column at a time, which is
1425/// one column's bytes rather than every column's.
1426///
1427/// The offsets going block relative comes free with it, and takes the four gigabyte wall with it.
1428/// They were `u32` into a per column payload, so a column could not hold more than four gigabytes of
1429/// values however much memory the machine had, and `URL` and `Referer` are within a small factor of
1430/// that at a hundred million rows. A `u32` into a block of 1,024 values is not a bound anything real
1431/// reaches. The stored form is unchanged, because [`encode_offsets`] was already subtracting a per
1432/// block base before writing.
1433#[derive(Debug)]
1434struct GlobalDictionary {
1435    /// Keyed by the value's hash, which is already well spread, so the maps hash it once more
1436    /// with a multiply rather than with SipHash. SipHash here was one percent of a ClickBench load,
1437    /// and every stripe's merge of a column waits on the one before it.
1438    primary: HashMap<u64, u32, Spread>,
1439    collisions: HashMap<u64, Vec<u32>, Spread>,
1440    /// Every value's hash under [`DICTIONARY_CHECK_SEED`], in code order.
1441    checks: Vec<u64>,
1442    /// Where every value ends inside the payload block it is in, in code order.
1443    ends: Vec<u32>,
1444    counts: Vec<u64>,
1445    nulls: u64,
1446    /// The values of the block being filled, back to back.
1447    filling: Vec<u8>,
1448    /// One conservative four-byte substring signature per encoded payload block, in block order.
1449    ///
1450    /// Made where the block is encoded rather than where it is sealed, because sealing is under the
1451    /// writer's lock and every byte of every value going through [`gram_bits`] was 2.9 of the 14
1452    /// seconds the 10m ClickBench load spent on the 32 core box.
1453    grams: Vec<[u8; TEXT_GRAM_BYTES]>,
1454    /// Blocks that have filled and not been handed out to be encoded yet, each with its block number.
1455    ///
1456    /// Empty except inside the merge that filled them, and while the column is still too small to
1457    /// settle a shape on.
1458    waiting: Vec<(usize, Vec<u8>)>,
1459    /// Blocks kept raw to settle a shape on, spread across the column, each with its number.
1460    ///
1461    /// At most [`PAYLOAD_SAMPLE_BLOCKS`] of them and so at most a few megabytes. Spread rather than
1462    /// taken off the front for the reason [`settle_shape`] gives, and kept rather than read back
1463    /// because reading back is a decode and this is a sample of a column that is still growing.
1464    sample: Vec<(usize, Vec<u8>)>,
1465    /// How far apart the blocks in `sample` are, which doubles every time there are too many.
1466    stride: usize,
1467    /// What the blocks encoded so far were encoded with, once the column is big enough to settle it.
1468    shape: Option<chooser::Settled>,
1469    /// How many blocks had filled when that shape was settled.
1470    settled: usize,
1471    /// The blocks that are encoded and not yet in the file, in block order, following `placed`.
1472    ///
1473    /// Empty between stripes, because [`Writer::place_blocks`] writes them the moment they come
1474    /// back. Only a dictionary that never meets a writer, which is a test's, keeps them here.
1475    blocks: Vec<Vec<u8>>,
1476    /// Blocks that came back encoded ahead of a block before them, by block number.
1477    ///
1478    /// Two stripes merged one after the other can have their pages built in the other order, and a
1479    /// block cannot go into `blocks` until every block before it is there. They wait here until the
1480    /// gap closes, which is at most until the stripe merged just before this one is written.
1481    early: BTreeMap<usize, EncodedBlock>,
1482    /// Where every block already written to the file is, in block order.
1483    placed: Vec<Placed>,
1484    /// What the dictionary held the last time it was asked, see [`Self::recharge`], which is also
1485    /// what the load profile was told when there is one.
1486    charged: u64,
1487    /// Whether the dictionary stopped taking values, see [`Self::demote`].
1488    demoted: bool,
1489}
1490
1491/// Where one payload block of a global dictionary is in the file, and its checksum.
1492#[derive(Debug, Clone, Copy)]
1493struct Placed {
1494    start: u64,
1495    length: u64,
1496    hash: u64,
1497}
1498
1499/// Sorted `(head, code)` entries and the decoded bytes and block bases they were sorted over.
1500type RankedDictionary = (Vec<(u64, u32)>, Vec<u8>, Vec<u64>);
1501
1502impl GlobalDictionary {
1503    fn new() -> Self {
1504        Self {
1505            primary: HashMap::default(),
1506            collisions: HashMap::default(),
1507            checks: Vec::new(),
1508            ends: Vec::new(),
1509            counts: Vec::new(),
1510            nulls: 0,
1511            filling: Vec::new(),
1512            grams: Vec::new(),
1513            waiting: Vec::new(),
1514            sample: Vec::new(),
1515            stride: 1,
1516            shape: None,
1517            settled: 0,
1518            blocks: Vec::new(),
1519            early: BTreeMap::new(),
1520            placed: Vec::new(),
1521            charged: 0,
1522            demoted: false,
1523        }
1524    }
1525
1526    /// How many distinct values this dictionary holds, which is one past its largest code.
1527    fn values(&self) -> usize {
1528        self.ends.len()
1529    }
1530
1531    /// About how many bytes closing this dictionary holds at once: every value decoded, and a
1532    /// sort entry and a code for each.
1533    fn closing_bytes(&self) -> usize {
1534        let values = self.values();
1535        let decoded = (0..values.div_ceil(TEXT_PAYLOAD_VALUES))
1536            .map(|block| self.ends[((block + 1) * TEXT_PAYLOAD_VALUES).min(values) - 1] as usize)
1537            .sum::<usize>();
1538        decoded.saturating_add(values.saturating_mul(size_of::<(u64, u32)>() + size_of::<u32>()))
1539    }
1540
1541    /// About what the dictionary holds in memory, by capacity rather than by length.
1542    ///
1543    /// A hash table is charged its buckets, which is a power of two over eight sevenths of what it
1544    /// says it can hold, and a byte of control per bucket. The blocks waiting to be encoded and the
1545    /// ones kept to settle a shape on are counted one by one, and there are only ever a few.
1546    fn held_bytes(&self) -> u64 {
1547        fn table<K, V, S>(map: &HashMap<K, V, S>) -> usize {
1548            (map.capacity() * 8 / 7).next_power_of_two() * (size_of::<(K, V)>() + 1)
1549        }
1550        fn spilled<T>(values: &Vec<T>) -> usize {
1551            values.capacity() * size_of::<T>()
1552        }
1553        let raw = |blocks: &Vec<(usize, Vec<u8>)>| {
1554            spilled(blocks) + blocks.iter().map(|(_, block)| block.capacity()).sum::<usize>()
1555        };
1556        let bytes = table(&self.primary)
1557            + table(&self.collisions)
1558            + self.collisions.values().map(spilled).sum::<usize>()
1559            + spilled(&self.checks)
1560            + spilled(&self.ends)
1561            + spilled(&self.counts)
1562            + self.filling.capacity()
1563            + spilled(&self.grams)
1564            + raw(&self.waiting)
1565            + raw(&self.sample)
1566            + self.blocks.iter().map(Vec::capacity).sum::<usize>()
1567            + spilled(&self.placed);
1568        bytes as u64
1569    }
1570
1571    /// Tells `profile` what the dictionary has grown or shrunk by since the last time, and hands
1572    /// back what it held then and what it holds now.
1573    fn recharge(&mut self, profile: Option<&LoadProfile>) -> (u64, u64) {
1574        let before = self.charged;
1575        let now = self.held_bytes();
1576        if let Some(profile) = profile {
1577            if now >= before {
1578                profile.hold(now - before);
1579            } else {
1580                profile.release(before - now);
1581            }
1582        }
1583        self.charged = now;
1584        (before, now)
1585    }
1586
1587    /// Stops the dictionary taking values, for good.
1588    ///
1589    /// The block being filled is sealed so that it goes out with the others, and what the
1590    /// dictionary keeps for looking values up is let go of, which on a column of mostly new values
1591    /// is most of what it holds. What stays is what the close needs to write the dictionary's page:
1592    /// where every value ends, how often each was seen and where its blocks went. The stripes that
1593    /// were coded against it still need that page to be read. See [`DEMOTED`].
1594    fn demote(&mut self) {
1595        if self.demoted {
1596            return;
1597        }
1598        self.seal_rest();
1599        self.release_lookup();
1600        self.demoted = true;
1601    }
1602
1603    /// Frees what the dictionary keeps for coding new values, once none are coming.
1604    ///
1605    /// The hash tables, the check hash of every value and the blocks kept to settle a shape on are
1606    /// what a merge looks values up in. The close reads the counts, the ends and the written blocks
1607    /// and none of these, which are most of what the dictionary holds per value, so they go before
1608    /// the close takes memory of its own rather than after.
1609    fn release_lookup(&mut self) {
1610        self.primary = HashMap::default();
1611        self.collisions = HashMap::default();
1612        self.checks = Vec::new();
1613        self.sample = Vec::new();
1614        self.filling = Vec::new();
1615    }
1616
1617    /// How many blocks are encoded, written or not, which is the number the next one has to have.
1618    fn encoded(&self) -> usize {
1619        self.placed.len() + self.blocks.len()
1620    }
1621
1622    #[cfg(test)]
1623    fn code(&mut self, text: &str) -> Result<u32> {
1624        let bytes = text.as_bytes();
1625        self.code_hashed(bytes, checksum(bytes), seeded_checksum(bytes, DICTIONARY_CHECK_SEED))
1626    }
1627
1628    /// The code for a value whose two hashes the caller already has.
1629    ///
1630    /// A stripe prepared outside the writer's lock hashed every value it holds while it was coding
1631    /// them, and merging it into this dictionary is one of these a distinct value rather than two
1632    /// hashes of every row. See [`prepare`].
1633    fn code_hashed(&mut self, text: &[u8], hash: u64, check: u64) -> Result<u32> {
1634        if let Some(&code) = self.primary.get(&hash) {
1635            if self.checks.get(code as usize) == Some(&check) {
1636                return Ok(code);
1637            }
1638            if let Some(codes) = self.collisions.get(&hash)
1639                && let Some(code) =
1640                    codes.iter().copied().find(|&code| self.checks[code as usize] == check)
1641            {
1642                return Ok(code);
1643            }
1644            let code = self.insert(text, check)?;
1645            self.collisions.entry(hash).or_default().push(code);
1646            return Ok(code);
1647        }
1648        let code = self.insert(text, check)?;
1649        self.primary.insert(hash, code);
1650        Ok(code)
1651    }
1652
1653    fn insert(&mut self, text: &[u8], check: u64) -> Result<u32> {
1654        if self.demoted {
1655            return Err(Error::internal("a value was coded against a demoted dictionary"));
1656        }
1657        let code = u32::try_from(self.ends.len())
1658            .map_err(|_| invalid("global dictionary has too many values"))?;
1659        self.filling.extend_from_slice(text);
1660        self.ends.push(
1661            u32::try_from(self.filling.len())
1662                .map_err(|_| invalid("a global dictionary value exceeds 4 GiB"))?,
1663        );
1664        self.checks.push(check);
1665        self.counts.push(0);
1666        if self.ends.len().is_multiple_of(TEXT_PAYLOAD_VALUES) {
1667            self.seal();
1668        }
1669        Ok(code)
1670    }
1671
1672    /// Closes the block being filled and puts it in the queue to be encoded.
1673    ///
1674    /// Also keeps a copy of it if it lands on the sample's stride, and halves the sample when that
1675    /// has left too many, which is what keeps the kept blocks spread evenly over however much of the
1676    /// column exists rather than bunched at whichever end was cheap to remember.
1677    fn seal(&mut self) {
1678        let at = self.ends.len().div_ceil(TEXT_PAYLOAD_VALUES) - 1;
1679        let bytes = std::mem::take(&mut self.filling);
1680        if at.is_multiple_of(self.stride) {
1681            self.sample.push((at, bytes.clone()));
1682            if self.sample.len() > PAYLOAD_SAMPLE_BLOCKS {
1683                self.stride *= 2;
1684                let stride = self.stride;
1685                self.sample.retain(|(at, _)| at % stride == 0);
1686            }
1687        }
1688        self.waiting.push((at, bytes));
1689    }
1690
1691    /// The values of one block, as slices into the bytes the block was filled with.
1692    fn slices<'a>(&self, at: usize, bytes: &'a [u8]) -> Vec<&'a [u8]> {
1693        block_values(self.block_ends(at), bytes)
1694    }
1695
1696    /// Where every value of one block ends, relative to the block.
1697    fn block_ends(&self, at: usize) -> &[u32] {
1698        let first = (at * TEXT_PAYLOAD_VALUES).min(self.ends.len());
1699        let last = (first + TEXT_PAYLOAD_VALUES).min(self.ends.len());
1700        &self.ends[first..last]
1701    }
1702
1703    /// Takes every waiting block out to be encoded somewhere else, if the column has a shape to
1704    /// encode them with.
1705    ///
1706    /// This is what keeps the encoding out of the writer's lock. A block needs its bytes, where its
1707    /// values end and the shape, and nothing else of the dictionary, so it goes out with a copy of
1708    /// the four kilobytes of ends it has and comes back through [`GlobalDictionary::take_back`].
1709    fn hand_out(&mut self, column: usize) -> Vec<Unencoded> {
1710        let Some(shape) = &self.shape else { return Vec::new() };
1711        let waiting = std::mem::take(&mut self.waiting);
1712        waiting
1713            .into_iter()
1714            .map(|(at, bytes)| Unencoded {
1715                column,
1716                at,
1717                ends: self.block_ends(at).to_vec(),
1718                bytes,
1719                shape: shape.clone(),
1720            })
1721            .collect()
1722    }
1723
1724    /// Takes back one block that was handed out, and moves every block that is now next in line
1725    /// into `blocks`.
1726    fn take_back(&mut self, at: usize, block: EncodedBlock) -> Result<()> {
1727        if at < self.encoded() || self.early.insert(at, block).is_some() {
1728            return Err(Error::internal("a dictionary block came back twice"));
1729        }
1730        while let Some(block) = self.early.remove(&self.encoded()) {
1731            self.push_block(block);
1732        }
1733        Ok(())
1734    }
1735
1736    /// Appends the next encoded block and its signature.
1737    fn push_block(&mut self, (bytes, grams): EncodedBlock) {
1738        self.blocks.push(bytes);
1739        self.grams.push(*grams);
1740    }
1741
1742    /// Settles the shape the waiting blocks are about to be encoded with, if there is enough column
1743    /// to settle one on.
1744    ///
1745    /// Settled again once the column has grown fourfold, because the sample it was settled on then
1746    /// covered a quarter of what exists now and a dictionary in first seen order does not look the
1747    /// same at both ends. Blocks already encoded keep the shape they were encoded with. They can,
1748    /// because a block says what it is: nothing reading one asks the column what shape to expect.
1749    fn settle(&mut self) -> Result<()> {
1750        if self.sample.len() < PAYLOAD_SAMPLE_BLOCKS {
1751            return Ok(());
1752        }
1753        self.settle_on_sample()
1754    }
1755
1756    /// Settles a shape on whatever sample there is, for a column the load ended before it had
1757    /// enough of to settle one the usual way.
1758    ///
1759    /// Such a column has fewer than [`PAYLOAD_SAMPLE_BLOCKS`] blocks, so the sample is every block
1760    /// it has. Trying every candidate on each of them instead runs at two to six megabytes a second,
1761    /// and once `hits` stored its string columns with a dictionary, the forty or so small ones were
1762    /// more than half the CPU of a million row load, all of it in the close.
1763    fn settle_rest(&mut self) -> Result<()> {
1764        if self.shape.is_some() || self.sample.is_empty() {
1765            return Ok(());
1766        }
1767        self.settle_on_sample()
1768    }
1769
1770    fn settle_on_sample(&mut self) -> Result<()> {
1771        let complete = self.ends.len() / TEXT_PAYLOAD_VALUES;
1772        if self.shape.is_some() && complete < self.settled.saturating_mul(4) {
1773            return Ok(());
1774        }
1775        let sample =
1776            self.sample.iter().map(|(at, bytes)| self.slices(*at, bytes)).collect::<Vec<_>>();
1777        self.shape = Some(string::with_symbols(settle_shape(&sample)?, &sample));
1778        self.settled = complete;
1779        Ok(())
1780    }
1781
1782    /// Seals the part block at the end of the load, if there is one.
1783    fn seal_rest(&mut self) {
1784        // Asked of the values rather than of the bytes, because a block of empty strings has values
1785        // in it and no bytes, and a column of nulls is exactly that. A demoted dictionary sealed its
1786        // part block when it was demoted and has taken nothing since.
1787        if !self.demoted && !self.ends.len().is_multiple_of(TEXT_PAYLOAD_VALUES) {
1788            self.seal();
1789        }
1790    }
1791
1792    /// Encodes the waiting block at `at`, with the settled shape when there is one and by trying
1793    /// everything when the column was too small to settle one.
1794    fn encode_waiting(&self, at: usize) -> Result<EncodedBlock> {
1795        let (block, bytes) = &self.waiting[at];
1796        let values = self.slices(*block, bytes);
1797        let encoded = match &self.shape {
1798            Some(shape) => string::encode_with(&values, shape)?,
1799            None => string::encode(&values)?,
1800        };
1801        Ok((encoded, block_grams(&values)))
1802    }
1803
1804    /// [`finish_dictionaries`] for one dictionary on this thread, for the tests that hold one.
1805    #[cfg(test)]
1806    fn finish_blocks(&mut self) -> Result<()> {
1807        self.seal_rest();
1808        let made = (0..self.waiting.len())
1809            .map(|at| self.encode_waiting(at))
1810            .collect::<Result<Vec<_>>>()?;
1811        for ((at, _), block) in std::mem::take(&mut self.waiting).into_iter().zip(made) {
1812            if self.encoded() != at {
1813                return Err(Error::internal("a dictionary block was encoded out of order"));
1814            }
1815            self.push_block(block);
1816        }
1817        Ok(())
1818    }
1819
1820    /// Every value of this dictionary read back out of its encoded blocks, as the bytes back to back
1821    /// and where each block starts in them.
1822    ///
1823    /// This is the one place the whole column is in memory at once and the reason [`Writer::close`]
1824    /// takes the columns one at a time rather than across threads. One column's values is 1.3 GB on
1825    /// the worst ClickBench column, and five columns of that at once is the peak this was all meant
1826    /// to remove.
1827    ///
1828    /// The blocks are spread over threads instead. Each block's decoded length is already known from
1829    /// the ends of its values, so the answer is laid out before anything is decoded and every thread
1830    /// decodes its own run of blocks straight into its own part of it. On the 10m ClickBench sample
1831    /// this was a second of the close for `URL` alone, on one core of thirty two, and the close is
1832    /// what a load waits on once its stripes are written.
1833    ///
1834    /// The blocks already written are read back out of `file`, so what a thread holds beyond the
1835    /// answer is one encoded block. They were written moments or minutes ago and are almost always
1836    /// still in the page cache, so this is a copy rather than a read of the disk.
1837    fn decoded(&self, file: Option<&dyn rudb_io::File>) -> Result<(Vec<u8>, Vec<u64>)> {
1838        let count = self.placed.len() + self.blocks.len();
1839        if count != self.values().div_ceil(TEXT_PAYLOAD_VALUES) {
1840            return Err(invalid("global dictionary blocks do not cover its values"));
1841        }
1842        let mut bases = Vec::with_capacity(count);
1843        let mut total = 0_usize;
1844        for block in 0..count {
1845            bases.push(total as u64);
1846            let last = ((block + 1) * TEXT_PAYLOAD_VALUES).min(self.values()) - 1;
1847            total = total
1848                .checked_add(self.ends[last] as usize)
1849                .ok_or_else(|| invalid("global dictionary does not fit in memory"))?;
1850        }
1851        let mut flat = vec![0_u8; total];
1852        let mut outs = Vec::with_capacity(count);
1853        let mut rest = flat.as_mut_slice();
1854        for block in 0..count {
1855            let end = bases.get(block + 1).map_or(total, |&base| base as usize);
1856            let (out, after) = rest.split_at_mut(end - bases[block] as usize);
1857            outs.push((block, out));
1858            rest = after;
1859        }
1860        let one = |run: &mut [(usize, &mut [u8])]| -> Result<()> {
1861            let mut stored = Vec::new();
1862            for (block, out) in run {
1863                let encoded = match self.placed.get(*block) {
1864                    Some(place) => {
1865                        let file = file.ok_or_else(|| {
1866                            Error::internal("a written dictionary block has no file")
1867                        })?;
1868                        let length = usize::try_from(place.length).map_err(|_| {
1869                            invalid("global dictionary block does not fit in memory")
1870                        })?;
1871                        stored.resize(length, 0);
1872                        read_at(file, place.start, &mut stored)?;
1873                        if checksum(&stored) != place.hash {
1874                            return Err(invalid(
1875                                "a global dictionary block did not read back as written",
1876                            ));
1877                        }
1878                        stored.as_slice()
1879                    }
1880                    None => &self.blocks[*block - self.placed.len()],
1881                };
1882                let decoded = string::decode_flat(encoded)?;
1883                if decoded.bytes().len() != out.len() {
1884                    return Err(invalid(
1885                        "a global dictionary block is not the length its ends say",
1886                    ));
1887                }
1888                out.copy_from_slice(decoded.bytes());
1889            }
1890            Ok(())
1891        };
1892        // Sixteen blocks a thread at the least, because a thread costs about what decoding a few
1893        // blocks does and most columns have one or two.
1894        let workers = close_workers().min(count / 16).max(1);
1895        if workers <= 1 {
1896            one(&mut outs)?;
1897        } else {
1898            let per = count.div_ceil(workers);
1899            std::thread::scope(|scope| {
1900                outs.chunks_mut(per)
1901                    .map(|run| scope.spawn(|| one(run)))
1902                    .collect::<Vec<_>>()
1903                    .into_iter()
1904                    .try_for_each(|handle| {
1905                        handle.join().map_err(|_| {
1906                            Error::internal("a global dictionary decode worker panicked")
1907                        })?
1908                    })
1909            })?;
1910        }
1911        drop(outs);
1912        Ok((flat, bases))
1913    }
1914
1915    /// Where the value at `code` sits in the bytes [`GlobalDictionary::decoded`] handed back.
1916    ///
1917    /// A block's first value starts at the block, and every other value starts where the one before
1918    /// it ended, which is what makes 1,024 values 1,024 numbers rather than 1,025.
1919    fn value_span(ends: &[u32], bases: &[u64], code: usize) -> (usize, usize) {
1920        let Some(&base) = bases.get(code / TEXT_PAYLOAD_VALUES) else { return (0, 0) };
1921        let Some(&end) = ends.get(code) else { return (0, 0) };
1922        let base = base as usize;
1923        let from =
1924            if code.is_multiple_of(TEXT_PAYLOAD_VALUES) { 0 } else { ends[code - 1] as usize };
1925        (base + from, base + end as usize)
1926    }
1927
1928    /// This dictionary's values in sorted order, each as the first eight bytes of the value and the
1929    /// code that holds it, so entry `rank` describes the value that sits at `rank` when the values
1930    /// are sorted by their bytes.
1931    ///
1932    /// Codes themselves stay in first appearance order, which is what lets the writer hand one out
1933    /// the moment it sees a value rather than waiting for the last stripe, and which also keeps a
1934    /// stripe's codes close together because the data is clustered. This is what puts the values
1935    /// back in order for anything that needs it, and it is separate from the codes so that getting
1936    /// it costs a sort of the distinct values at the end rather than a rewrite of every code page.
1937    ///
1938    /// The order is the byte order of the values and nothing else. The heads are attached after the
1939    /// sort rather than sorted on, because padding with zero on the right is order preserving for
1940    /// byte strings and so sorting by head and then by bytes lands in the same place as sorting by
1941    /// bytes: a shorter value differs from a longer one that starts the same way at a position
1942    /// where the shorter one has run out, and zero is below every byte that could be there.
1943    ///
1944    /// The heads are kept because a reader searching this order wants a comparison it can make out
1945    /// of the index alone. What they buy there depends entirely on the column and is much less than
1946    /// it looks on the columns that cost the most, which [`sort_by_value`] measures.
1947    fn ranked_with_values(&self, file: Option<&dyn rudb_io::File>) -> Result<RankedDictionary> {
1948        let (flat, bases) = self.decoded(file)?;
1949        let value = |code: u32| {
1950            let (from, to) = Self::value_span(&self.ends, &bases, code as usize);
1951            flat.get(from..to).unwrap_or_default()
1952        };
1953        let mut codes = (0..self.values() as u32).collect::<Vec<_>>();
1954        sort_by_value_across(&mut codes, value, close_workers());
1955        let order = codes.into_iter().map(|code| (head(value(code)), code)).collect();
1956        Ok((order, flat, bases))
1957    }
1958
1959    #[cfg(test)]
1960    fn ranked(&self, file: Option<&dyn rudb_io::File>) -> Result<Vec<(u64, u32)>> {
1961        self.ranked_with_values(file).map(|(order, _, _)| order)
1962    }
1963}
1964
1965/// Appends pages and commits a new directory.
1966///
1967/// One writer covers a whole file rather than one table. [`Writer::next`] closes the table it is on
1968/// and opens another over the same file, and [`Writer::finish`] commits every table it has closed in
1969/// one generation. That is what makes a checkpoint atomic across tables: there is one slot write at
1970/// the end of it and a reader sees every table at the generation before it or every table at the
1971/// generation after it.
1972#[derive(Debug)]
1973pub struct Writer {
1974    /// The file, through `rudb-io` rather than `std::fs`, so that a test can hand the writer a
1975    /// simulated filesystem and crash a load at every call it makes.
1976    file: Box<dyn rudb_io::File>,
1977    /// Where the next write goes, counted here rather than asked of the file.
1978    ///
1979    /// The file's own cursor is not ours. Building the numeric frequencies reads pages back through
1980    /// [`read_at`], and a positional read is only positional about where it reads from: `pread`
1981    /// leaves the cursor alone, and the call Windows has for it moves the cursor to the end of what
1982    /// it read. A writer that asked the file where it was would then write the directory over a
1983    /// page it had already written, which is what it did.
1984    at: u64,
1985    /// How far into the file the kernel has been asked to start writing, see [`WRITEBACK_STRETCH`].
1986    written_back: u64,
1987    table: Table,
1988    generation: u64,
1989    /// The first and the last source position in every stripe, in the order the stripes were
1990    /// written.
1991    order: Vec<((u64, u64), (u64, u64))>,
1992    next_order: u64,
1993    dictionaries: Vec<Option<GlobalDictionary>>,
1994    /// Which columns still have a global dictionary, shared with every [`Preparer`] this writer
1995    /// hands out so that a stripe prepared after a column lost its dictionary is not coded for it.
1996    coded: Arc<prepare::Coding>,
1997    /// One per column, folding the rows into a summary and a sketch as they go past.
1998    ///
1999    /// `None` for a column with no hash rule, which is the interval and the nested types. See
2000    /// [`stats::Gather`] for why the statistics are built here rather than by reading the file back
2001    /// once it is committed.
2002    gathers: Vec<Option<stats::Gather>>,
2003    /// The dictionaries and the statistics while a [`Merger`] has them, which is from
2004    /// [`Writer::merger`] until the table is closed. `dictionaries` and `gathers` are empty then.
2005    lent: Option<Arc<Lent>>,
2006    pending: Vec<PendingChunk>,
2007    /// The tables already closed in this generation, in the order they were written.
2008    closed: Vec<Entry>,
2009    /// The views the next commit writes down, which [`Writer::with_views`] sets.
2010    ///
2011    /// Carried forward from the committed generation by [`Writer::open`], so a writer that was only
2012    /// opened to append a table does not have to know about views to avoid dropping them.
2013    views: Vec<ViewEntry>,
2014    /// The device card the next commit writes down, which is [`card_for`] the file.
2015    card: Option<KeptCard>,
2016    /// Where the stages this writer runs are charged, which [`Writer::with_profile`] sets.
2017    ///
2018    /// The writer runs the page builder, the dictionary blocks, the writes and the publish, and it
2019    /// charges them once per stripe and once per worker, never per chunk. See
2020    /// `rudb_metrics::LoadProfile` for why that is the grain.
2021    profile: Option<Arc<LoadProfile>>,
2022}
2023
2024/// A chunk that has arrived and is waiting for the rest of its stripe.
2025///
2026/// The rows are kept rather than the pages they encode to, which is the whole of #808's first half.
2027/// Encoding on arrival put every column of every part on the thread that called `append_at`, and
2028/// that thread is the only one the load has. Encoding at the flush instead means a stripe's worth
2029/// of work is on the table at once, and a stripe splits by column into a hundred and five pieces
2030/// that share nothing.
2031#[derive(Debug)]
2032struct PendingChunk {
2033    order: (u64, u64),
2034    chunk: Chunk,
2035}
2036
2037/// What the writer still needs of a part once its columns are encoded: where in the source it came
2038/// from, how many rows it has and how large those rows were.
2039///
2040/// A stripe waiting for the writer's lock carries these rather than its chunks, so its rows are
2041/// freed as soon as they are encoded and not after the stripe is written. See [`prepare`].
2042#[derive(Debug, Clone, Copy)]
2043struct Part {
2044    order: (u64, u64),
2045    rows: usize,
2046    footprint: usize,
2047}
2048
2049impl Part {
2050    fn of(pending: &PendingChunk) -> Self {
2051        Self {
2052            order: pending.order,
2053            rows: pending.chunk.len(),
2054            footprint: pending.chunk.footprint(),
2055        }
2056    }
2057}
2058
2059/// One column's share of a stripe, which is what one encode worker produces.
2060///
2061/// Indexed by part, so a stripe is a column of these and the write loop reads down one of them.
2062/// That is also the order the loop wanted: `flush_pending` walks a column at a time and lays its
2063/// parts next to each other, and it used to reach across a row of parts to do it.
2064#[derive(Debug, Default)]
2065struct ColumnStripe {
2066    pages: Vec<Vec<u8>>,
2067    /// Each page's checksum, taken where the page is built so that the writer, which holds its
2068    /// lock while it writes, does not walk every byte of the stripe a second time.
2069    sums: Vec<u64>,
2070    codes: Vec<Option<Vec<u32>>>,
2071    sieves: Vec<Option<Sieve>>,
2072    ranges: Vec<Range>,
2073}
2074
2075/// Whether a column of this type is coded against a global dictionary.
2076///
2077/// A dictionary, its codes and the membership index beside them are about bytes and not about
2078/// text, so a blob gets one the same as a varchar does. ClickBench's `hits.parquet` stores every
2079/// string column as a plain byte array, which reads back as a blob, and those columns were being
2080/// written as a length and the bytes for every row: 533 MB for the first million rows where DuckDB
2081/// writes 142.
2082fn coded_type(ty: &LogicalType) -> bool {
2083    matches!(ty, LogicalType::Varchar | LogicalType::Blob)
2084}
2085
2086/// The tag a directory gives a column's global dictionary.
2087///
2088/// A varchar's is 1, as it always was. A blob's is 2, so that a reader from before blobs had
2089/// dictionaries meets a tag it does not know and refuses the file, rather than laying the rest of
2090/// the directory out as if the blob columns had no dictionary and reading everything after the
2091/// first one from the wrong place.
2092fn dictionary_tag(ty: &LogicalType) -> u8 {
2093    if ty == &LogicalType::Blob { 2 } else { 1 }
2094}
2095
2096/// Roughly what encoding a column of this type costs, for ordering the encode queue.
2097///
2098/// Only the order matters and only roughly. A string column hashes and copies every value into a
2099/// dictionary and is in a different class from everything else, and among the fixed widths the wide
2100/// ones carry more bytes through the cascade than the narrow ones. Anything finer than that would
2101/// be a cost model, and the queue already absorbs a wrong guess: it only has to avoid finishing on
2102/// a column nobody else can help with.
2103fn weight(ty: &LogicalType) -> usize {
2104    match ty {
2105        LogicalType::Varchar | LogicalType::Blob | LogicalType::Bit => 64,
2106        LogicalType::HugeInt
2107        | LogicalType::UHugeInt
2108        | LogicalType::Uuid
2109        | LogicalType::Interval => 16,
2110        LogicalType::BigInt
2111        | LogicalType::UBigInt
2112        | LogicalType::Timestamp
2113        | LogicalType::Time
2114        | LogicalType::TimeTz
2115        | LogicalType::TimestampTz
2116        | LogicalType::TimestampS
2117        | LogicalType::TimestampMs
2118        | LogicalType::TimestampNs
2119        | LogicalType::Double
2120        | LogicalType::Decimal { .. } => 8,
2121        LogicalType::Integer | LogicalType::UInteger | LogicalType::Date | LogicalType::Float => 4,
2122        LogicalType::SmallInt | LogicalType::USmallInt => 2,
2123        _ => 1,
2124    }
2125}
2126
2127/// Parts in one stripe.
2128///
2129/// Sixty four thousand rows is the smallest stripe that keeps the ClickBench directory in single
2130/// digit megabytes at a hundred million rows, and it puts a four byte column's page at a quarter of
2131/// a megabyte, which is the size a sequential read wants. Larger stripes buy a smaller directory
2132/// and cost a sparse fetch, which has to read a page index before it can reach one part.
2133pub const STRIPE_PARTS: usize = 64;
2134
2135/// How many rows the writer wants to see before it decides whether a varchar column gets to keep
2136/// its global dictionary.
2137///
2138/// See [`prepare::drops_dictionary`]. A stripe is up to [`STRIPE_PARTS`] parts, so most tables give it
2139/// far more than this and it binds only on a table that is smaller than one stripe. A handful of
2140/// rows says nothing about whether a column repeats itself, and the answer that costs nothing when
2141/// the sample is that small is the one the writer has always given, which is to keep the dictionary.
2142const DICTIONARY_DECIDE_ROWS: usize = 4_096;
2143
2144/// Out of ten. A varchar column loses its dictionary when more than this many rows in ten of the
2145/// first stripe held a value that stripe had not seen before.
2146///
2147/// See [`prepare::drops_dictionary`]. Nine and not five, because the properties a dictionary buys are
2148/// worth keeping everywhere they are real. On ClickBench the widest string column is `Referer` at
2149/// 0.131 of its first stripe and every other one is below that, so nothing there is near this and
2150/// every one of them keeps its dictionary, which is what a group by on codes wants. On TPC-H
2151/// `o_comment` and `c_comment` are at 0.97 and are what this catches.
2152///
2153/// `l_comment` sits at 0.883 and so keeps its dictionary. Eight was built and measured rather than
2154/// argued about, and it is not a clear win: it takes `select l_comment from lineitem` from 4.335 G
2155/// instructions to 3.473 G and the file from 280.2 MB to 260.4 MB, and it takes a `like` over the
2156/// same column from 3.29 G to 4.27 G, because a dictionary runs the predicate once a distinct value
2157/// and there are 3.6 M of those to 6.0 M rows. The 22 query suite came out 6.91 s against 7.07 s in
2158/// favour of nine. So nine stays until there is a reason to prefer one of those shapes. See #1137.
2159const DICTIONARY_DISTINCT_IN_TEN: usize = 9;
2160
2161/// Bytes one part takes in a stripe's index page: four for the length, eight for the checksum.
2162const INDEX_ENTRY: usize = size_of::<u32>() + size_of::<u64>();
2163
2164/// Bytes one column's section of a stripe's index page takes, including its own trailing checksum.
2165fn index_section(parts: usize) -> Result<usize> {
2166    parts
2167        .checked_mul(INDEX_ENTRY)
2168        .and_then(|bytes| bytes.checked_add(size_of::<u64>()))
2169        .ok_or_else(|| invalid("index page length overflow"))
2170}
2171
2172impl Writer {
2173    /// Opens a committed file and starts a table in the generation after the one it holds.
2174    ///
2175    /// The tables already in the file are carried forward by name and by directory pointer, and
2176    /// their pages are not read. Nothing in the file is overwritten: the new table's pages and the
2177    /// new catalog go on the end, past the catalog the committed generation points at, and the one
2178    /// write that is not an append is the slot in the header that [`Writer::finish`] does last.
2179    ///
2180    /// That slot is the other one. A file committed at generation 1 is named by the slot at 16 and
2181    /// generation 2 writes the one at 44, so until the last four bytes of the commit land the file
2182    /// still reads as the generation before it, and a slot torn across a write fails its checksum
2183    /// and the reader falls back to the one beside it. This is what the second slot has always been
2184    /// for.
2185    ///
2186    /// # Errors
2187    ///
2188    /// If the file has no valid committed directory, is not this build's format, repeats the name
2189    /// of a table already in it that holds rows, has a field with no scalar encoding, or cannot be
2190    /// written.
2191    pub fn open(
2192        path: impl AsRef<Path>,
2193        name: impl Into<String>,
2194        fields: Vec<Field>,
2195    ) -> Result<Self> {
2196        Self::open_in(&RealFilesystem::new(), path, name, fields)
2197    }
2198
2199    /// [`Writer::open`] on a file in `fs`, which is how a crash test runs an append against the
2200    /// simulated filesystem.
2201    ///
2202    /// # Errors
2203    ///
2204    /// The same as [`Writer::open`].
2205    pub fn open_in(
2206        fs: &dyn Filesystem,
2207        path: impl AsRef<Path>,
2208        name: impl Into<String>,
2209        fields: Vec<Field>,
2210    ) -> Result<Self> {
2211        for field in &fields {
2212            type_tag(&field.ty)?;
2213        }
2214        let name = name.into();
2215        let file = fs.open(path.as_ref(), OpenMode::ReadWrite)?;
2216        let size = file.len()?;
2217        let (slot, bytes, _) = committed_slot(&*file, size)?;
2218        let (mut closed, views, card) = decode_catalog(&bytes, size)?;
2219        let card = card_for(path.as_ref(), card);
2220        // A table already in the file under this name is only in the way if it holds rows. One that
2221        // holds none has no pages for this generation to carry and no reader that could lose
2222        // anything, so the table being started here takes its place in the catalog rather than
2223        // colliding with it, and `finish` writes the new entry where the old one was.
2224        //
2225        // That is not a corner. It is the shape every loading script writes: the schema goes in one
2226        // statement and the rows go in the next, and a checkpoint between them commits the empty
2227        // table. Before this, the second statement had to build the whole table in memory because
2228        // the first had already put the name in the file, which is how a load of a table larger
2229        // than memory became a load that needed memory the size of the table.
2230        if let Some(at) = closed.iter().position(|held| held.name == name) {
2231            if closed[at].rows > 0 {
2232                return Err(invalid("two tables in one native file have the same name"));
2233            }
2234            closed.remove(at);
2235        }
2236        // The generation of the slot whose bytes checksummed, and not the highest number in the
2237        // header. A slot torn across a write can hold any number at all, and taking that one would
2238        // be choosing which slot to overwrite from a value nothing has vouched for, which is how a
2239        // half written commit gets to destroy the one good copy beside it.
2240        let generation = slot
2241            .generation
2242            .checked_add(1)
2243            .ok_or_else(|| invalid("native file generation overflow"))?;
2244        Ok(Self {
2245            file,
2246            // The end of the file, so that the committed generation's catalog stays where its slot
2247            // says it is and keeps naming a file a reader can still open.
2248            at: size,
2249            written_back: size,
2250            dictionaries: fields
2251                .iter()
2252                .map(|field| coded_type(&field.ty).then(GlobalDictionary::new))
2253                .collect(),
2254            coded: Arc::new(prepare::Coding::new(fields.iter().map(|field| coded_type(&field.ty)))),
2255            gathers: fields.iter().map(|field| stats::Gather::new(&field.ty, generation)).collect(),
2256            lent: None,
2257            table: Table {
2258                name,
2259                dictionaries: vec![None; fields.len()],
2260                dictionary_payloads: Vec::new(),
2261                demoted: Vec::new(),
2262                distincts: vec![None; fields.len()],
2263                fields,
2264                stripes: Vec::new(),
2265                rows: 0,
2266                frequencies: Vec::new(),
2267                pair_frequencies: Vec::new(),
2268                frequency_texts: Vec::new(),
2269                host_groups: None,
2270                clustering: None,
2271                constraints: Constraints::default(),
2272                generation,
2273                sections: Vec::new(),
2274            },
2275            generation,
2276            order: Vec::new(),
2277            next_order: 0,
2278            pending: Vec::with_capacity(STRIPE_PARTS),
2279            closed,
2280            views,
2281            card,
2282            profile: None,
2283        })
2284    }
2285
2286    /// Creates a new v10 file and its first table.
2287    ///
2288    /// # Errors
2289    ///
2290    /// If the file exists, a field has no scalar encoding, or the path cannot be written.
2291    pub fn create(
2292        path: impl AsRef<Path>,
2293        name: impl Into<String>,
2294        fields: Vec<Field>,
2295    ) -> Result<Self> {
2296        Self::create_in(&RealFilesystem::new(), path, name, fields)
2297    }
2298
2299    /// [`Writer::create`] with the file made in `fs` rather than on the real filesystem.
2300    ///
2301    /// Every call the writer makes on the file from here to [`Writer::finish`] goes to that
2302    /// filesystem, which is what lets a test built on `rudb_io::SimFilesystem` stop a load at any
2303    /// one of them and look at what a crash there would leave on the disk.
2304    ///
2305    /// # Errors
2306    ///
2307    /// The same as [`Writer::create`].
2308    pub fn create_in(
2309        fs: &dyn Filesystem,
2310        path: impl AsRef<Path>,
2311        name: impl Into<String>,
2312        fields: Vec<Field>,
2313    ) -> Result<Self> {
2314        for field in &fields {
2315            type_tag(&field.ty)?;
2316        }
2317        let file = fs.open(path.as_ref(), OpenMode::CreateNew)?;
2318        let mut header = [0; HEADER as usize];
2319        header[..8].copy_from_slice(MAGIC);
2320        header[8..12].copy_from_slice(&FORMAT.to_le_bytes());
2321        file.write_at(0, &header)?;
2322        Ok(Self {
2323            file,
2324            at: HEADER,
2325            written_back: HEADER,
2326            dictionaries: fields
2327                .iter()
2328                .map(|field| coded_type(&field.ty).then(GlobalDictionary::new))
2329                .collect(),
2330            coded: Arc::new(prepare::Coding::new(fields.iter().map(|field| coded_type(&field.ty)))),
2331            gathers: fields.iter().map(|field| stats::Gather::new(&field.ty, 1)).collect(),
2332            lent: None,
2333            table: Table {
2334                name: name.into(),
2335                dictionaries: vec![None; fields.len()],
2336                dictionary_payloads: Vec::new(),
2337                demoted: Vec::new(),
2338                distincts: vec![None; fields.len()],
2339                fields,
2340                stripes: Vec::new(),
2341                rows: 0,
2342                frequencies: Vec::new(),
2343                pair_frequencies: Vec::new(),
2344                frequency_texts: Vec::new(),
2345                host_groups: None,
2346                clustering: None,
2347                constraints: Constraints::default(),
2348                generation: 1,
2349                sections: Vec::new(),
2350            },
2351            generation: 1,
2352            order: Vec::new(),
2353            next_order: 0,
2354            pending: Vec::with_capacity(STRIPE_PARTS),
2355            closed: Vec::new(),
2356            views: Vec::new(),
2357            card: card_for(path.as_ref(), None),
2358            profile: None,
2359        })
2360    }
2361
2362    /// Creates a new file that holds no table at all, committed and ready to open.
2363    ///
2364    /// A database somebody dropped the last table out of is still a database, and until this there
2365    /// was no way to write one down. Every other way into this file goes through a table, because
2366    /// [`Writer::create`] takes the first one and [`Writer::finish`] commits the one it is on, so a
2367    /// catalog with nothing in it could be read and not written. The format already allowed it: the
2368    /// catalog is a count and that many entries, and a count of nought encodes and decodes the same
2369    /// way every other count does, which is why nothing here is a version change.
2370    ///
2371    /// It hands back nothing rather than a writer, because a writer with no table is a writer with
2372    /// nothing to append to. A file that is going to hold a table is [`Writer::create`], and one
2373    /// that is going to have a table added to it later is [`Writer::open`], which reads what this
2374    /// wrote the same way it reads any other generation.
2375    ///
2376    /// It takes the views anyway, because a database with no table can still have views in it. A
2377    /// view over `range` or over another view names no table, so dropping the last table out of a
2378    /// database does not have to leave the catalog with nothing worth writing down.
2379    ///
2380    /// # Errors
2381    ///
2382    /// If the file exists or the path cannot be written.
2383    pub fn empty(path: impl AsRef<Path>, views: &[ViewEntry]) -> Result<()> {
2384        let file = RealFilesystem::new().open(path.as_ref(), OpenMode::CreateNew)?;
2385        let mut header = [0; HEADER as usize];
2386        header[..8].copy_from_slice(MAGIC);
2387        header[8..12].copy_from_slice(&FORMAT.to_le_bytes());
2388        file.write_at(0, &header)?;
2389        let catalog = encode_catalog(&[], views, card_for(path.as_ref(), None).as_ref())?;
2390        file.write_at(HEADER, &catalog)?;
2391        // The same two syncs in the same order as [`Writer::finish`], and for the same reason. The
2392        // catalog is on the disk before the slot names it, so a file this is interrupted in the
2393        // middle of is a header with no valid slot rather than a slot pointing at nothing.
2394        file.sync()?;
2395        let slot = Slot {
2396            offset: HEADER,
2397            length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
2398            generation: 1,
2399            hash: checksum(&catalog),
2400        };
2401        file.write_at(slot_offset(1), &slot.bytes())?;
2402        file.sync()?;
2403        Ok(())
2404    }
2405
2406    /// Closes the table this writer is on and starts another one in the same file.
2407    ///
2408    /// Nothing is published here. The closed table's directory is written so that the bytes are on
2409    /// disk and its span is known, and the catalog that names it is only written by
2410    /// [`Writer::finish`], so a crash between two tables leaves the previous generation intact.
2411    ///
2412    /// # Errors
2413    ///
2414    /// If the name repeats a table already closed, a field has no scalar encoding, or the table
2415    /// being closed cannot be written.
2416    pub fn next(mut self, name: impl Into<String>, fields: Vec<Field>) -> Result<Self> {
2417        for field in &fields {
2418            type_tag(&field.ty)?;
2419        }
2420        let name = name.into();
2421        let entry = self.close()?;
2422        if entry.name == name {
2423            return Err(invalid("two tables in one native file have the same name"));
2424        }
2425        // An empty table the committed generation holds under this name steps aside for this one,
2426        // the same as it does for the first table in [`Writer::open`], and for the same reason: it
2427        // has no pages to carry and the load writing it now is the one that fills it.
2428        if let Some(at) = self.closed.iter().position(|held| held.name == name) {
2429            if self.closed[at].rows > 0 {
2430                return Err(invalid("two tables in one native file have the same name"));
2431            }
2432            self.closed.remove(at);
2433        }
2434        let Self { file, at, generation, mut closed, views, card, .. } = self;
2435        closed.push(entry);
2436        Ok(Self {
2437            file,
2438            written_back: at,
2439            at,
2440            generation,
2441            closed,
2442            views,
2443            card,
2444            profile: None,
2445            dictionaries: fields
2446                .iter()
2447                .map(|field| coded_type(&field.ty).then(GlobalDictionary::new))
2448                .collect(),
2449            coded: Arc::new(prepare::Coding::new(fields.iter().map(|field| coded_type(&field.ty)))),
2450            gathers: fields.iter().map(|field| stats::Gather::new(&field.ty, generation)).collect(),
2451            lent: None,
2452            table: Table {
2453                name,
2454                dictionaries: vec![None; fields.len()],
2455                dictionary_payloads: Vec::new(),
2456                demoted: Vec::new(),
2457                distincts: vec![None; fields.len()],
2458                fields,
2459                stripes: Vec::new(),
2460                rows: 0,
2461                frequencies: Vec::new(),
2462                pair_frequencies: Vec::new(),
2463                frequency_texts: Vec::new(),
2464                host_groups: None,
2465                clustering: None,
2466                constraints: Constraints::default(),
2467                generation,
2468                sections: Vec::new(),
2469            },
2470            order: Vec::new(),
2471            next_order: 0,
2472            pending: Vec::with_capacity(STRIPE_PARTS),
2473        })
2474    }
2475
2476    /// Sets the views the next commit writes down, replacing whatever was carried forward.
2477    ///
2478    /// It replaces rather than adds because the caller has the whole catalog in front of it and the
2479    /// writer does not. A view that was dropped is a view that is not in the list any more, and
2480    /// there is no other way for the writer to hear about that, since nothing else it is told about
2481    /// mentions views at all.
2482    ///
2483    /// A writer that is never told anything writes back the views it read at [`Writer::open`], so a
2484    /// checkpoint that only had a table to append does not quietly drop them.
2485    #[must_use]
2486    pub fn with_views(mut self, views: Vec<ViewEntry>) -> Self {
2487        self.views = views;
2488        self
2489    }
2490
2491    /// Charges the stages this writer runs to `profile`.
2492    ///
2493    /// For the table being written now. [`Writer::next`] starts the next table without one,
2494    /// because a second table's stripes charged to the first table's load would be a profile of
2495    /// neither.
2496    #[must_use]
2497    pub fn with_profile(mut self, profile: Arc<LoadProfile>) -> Self {
2498        self.profile = Some(profile);
2499        self
2500    }
2501
2502    /// Sets what the table's global dictionaries may hold between them before the one growing
2503    /// fastest stops taking values, which is [`DICTIONARY_CAP_BYTES`] unless this says
2504    /// otherwise. It applies to every [`Preparer`] and [`Merger`] this writer has handed out too.
2505    #[must_use]
2506    pub fn with_dictionary_cap(self, bytes: u64) -> Self {
2507        self.coded.cap(bytes);
2508        self
2509    }
2510
2511    /// Records the order this table's rows are meant to be stored in.
2512    ///
2513    /// The declaration goes in the table directory and comes back out of
2514    /// [`Table::clustering`]. Nothing here sorts anything, and nothing here checks that the rows
2515    /// handed to [`Writer::append`] arrive in the order this claims. That is deliberate for now:
2516    /// the thing that was missing was a place to write the order down, and a loader that honours
2517    /// the declaration is the next piece rather than this one.
2518    ///
2519    /// The declaration applies to the table the writer is currently on, so it is set after
2520    /// [`Writer::next`] rather than once for the file.
2521    ///
2522    /// # Errors
2523    ///
2524    /// If the declaration names a column this table does not have.
2525    pub fn declare(mut self, clustering: Clustering) -> Result<Self> {
2526        // Rebuilt against this table's own column count rather than trusted, because the caller
2527        // built it against a catalog entry and the two could have drifted.
2528        self.table.clustering = Some(Clustering::new(
2529            clustering.columns().to_vec(),
2530            clustering.width(),
2531            &self.table.fields,
2532        )?);
2533        Ok(self)
2534    }
2535
2536    /// Records the keys and foreign keys of the table the writer is on, which come back out of
2537    /// [`Table::constraints`]. Nothing here checks the rows against them, since the catalog already
2538    /// did before it let the rows in.
2539    ///
2540    /// # Errors
2541    ///
2542    /// If a key or a foreign key names a column this table does not have, or has no columns.
2543    pub fn constrain(mut self, constraints: Constraints) -> Result<Self> {
2544        let width = self.table.fields.len();
2545        let fits = |columns: &[u16]| {
2546            !columns.is_empty() && columns.iter().all(|&column| usize::from(column) < width)
2547        };
2548        if !constraints.keys.iter().all(|(columns, _)| fits(columns))
2549            || !constraints.foreign.iter().all(|foreign| {
2550                fits(&foreign.columns) && foreign.referenced.len() == foreign.columns.len()
2551            })
2552        {
2553            return Err(invalid("a constraint names a column the table does not have"));
2554        }
2555        self.table.constraints = constraints;
2556        Ok(self)
2557    }
2558
2559    /// Appends bytes at the end of the file and moves the writer's own offset past them.
2560    ///
2561    /// Every write in here goes through this, so that [`Writer::at`] is the only answer to where
2562    /// anything is and the file's cursor is never consulted for it.
2563    fn put(&mut self, bytes: &[u8]) -> Result<()> {
2564        self.file.write_at(self.at, bytes)?;
2565        self.at = self
2566            .at
2567            .checked_add(bytes.len() as u64)
2568            .ok_or_else(|| invalid("native file length overflow"))?;
2569        if self.at - self.written_back >= WRITEBACK_STRETCH {
2570            self.file.start_writeback(self.written_back, self.at - self.written_back);
2571            self.written_back = self.at;
2572        }
2573        Ok(())
2574    }
2575
2576    /// Writes one chunk as independently readable column pages.
2577    ///
2578    /// # Errors
2579    ///
2580    /// If its width or types differ from the declared table, or a page exceeds its bound.
2581    pub fn append(&mut self, chunk: &Chunk) -> Result<()> {
2582        let order = (self.next_order, 0);
2583        self.next_order = self.next_order.saturating_add(1);
2584        self.append_at(order, chunk)
2585    }
2586
2587    /// Writes one chunk and records its source position for directory ordering.
2588    ///
2589    /// Pages may be encoded by parallel pipeline instances and reach the file in completion order.
2590    /// The stripe they land in is sorted by this key at commit, and [`Self::finish`] rejects a
2591    /// sequence whose parts do not come out in source order once the stripes are sorted, because a
2592    /// stripe groups whatever arrived together and cannot put a late part back where it belongs.
2593    ///
2594    /// # Errors
2595    ///
2596    /// The same as [`Self::append`].
2597    pub fn append_at(&mut self, order: (u64, u64), chunk: &Chunk) -> Result<()> {
2598        if chunk.is_empty() {
2599            return Ok(());
2600        }
2601        self.admit(chunk)?;
2602        if self.pending.last().is_some_and(|last| last.order > order) {
2603            self.flush_pending()?;
2604        }
2605        // Cloned rather than encoded, and a clone of a chunk that owns its buffers is a copy of
2606        // them. Sixty four parts of a hundred and five columns is tens of megabytes held for the
2607        // length of a stripe and a few seconds of memory traffic over a whole ClickBench load,
2608        // against the hundreds of seconds of encode this is what lets off one thread.
2609        self.pending.push(PendingChunk { order, chunk: chunk.clone() });
2610        if self.pending.len() == STRIPE_PARTS {
2611            self.flush_pending()?;
2612        }
2613        Ok(())
2614    }
2615
2616    /// Writes a run of chunks as one stripe of its own.
2617    ///
2618    /// [`Self::append_at`] decides where a stripe ends by watching the orders go past, which works
2619    /// when one caller hands over every chunk in source order and does not when several do. A
2620    /// writer being fed by more than one pipeline instance sees the orders interleave, and a stripe
2621    /// that ends every time two of them cross is a stripe of one or two parts.
2622    ///
2623    /// So the grouping moves to the caller. Whoever is buffering hands over a run it already knows
2624    /// is contiguous and in order, and gets a stripe holding exactly that run. The orders still
2625    /// have to come out in source order once the stripes are sorted, which [`Self::finish`] checks,
2626    /// so the runs from different callers may interleave with each other but may not overlap.
2627    ///
2628    /// # Errors
2629    ///
2630    /// The same as [`Self::append`], and if the run is longer than [`STRIPE_PARTS`].
2631    pub fn append_stripe(&mut self, parts: Vec<((u64, u64), Chunk)>) -> Result<()> {
2632        if parts.len() > STRIPE_PARTS {
2633            return Err(invalid("a stripe was handed more parts than it holds"));
2634        }
2635        // Whatever an earlier caller left behind is its own stripe rather than the front of this
2636        // one, because the two runs are from different places in the source and a stripe is a run.
2637        self.flush_pending()?;
2638        for (order, chunk) in parts {
2639            if chunk.is_empty() {
2640                continue;
2641            }
2642            self.admit(&chunk)?;
2643            self.pending.push(PendingChunk { order, chunk });
2644        }
2645        self.flush_pending()
2646    }
2647
2648    /// Checks a chunk against the declared table and counts its rows in.
2649    fn admit(&mut self, chunk: &Chunk) -> Result<()> {
2650        if chunk.width() != self.table.fields.len() {
2651            return Err(invalid("chunk width differs from table schema"));
2652        }
2653        for (index, field) in self.table.fields.iter().enumerate() {
2654            if chunk.column(index)?.logical_type() != &field.ty {
2655                return Err(invalid("chunk type differs from table schema"));
2656            }
2657        }
2658        self.table.rows = self
2659            .table
2660            .rows
2661            .checked_add(chunk.len())
2662            .ok_or_else(|| invalid("row count overflow"))?;
2663        Ok(())
2664    }
2665
2666    /// One column's parts of a stripe as pages, for a column with no global dictionary.
2667    fn encode_pages(columns: &[&Vector]) -> Result<ColumnStripe> {
2668        let mut stripe = ColumnStripe {
2669            pages: Vec::with_capacity(columns.len()),
2670            sums: Vec::with_capacity(columns.len()),
2671            codes: Vec::with_capacity(columns.len()),
2672            sieves: Vec::with_capacity(columns.len()),
2673            ranges: Vec::with_capacity(columns.len()),
2674        };
2675        let mut settling = Settling::default();
2676        for &column in columns {
2677            Self::encode_page(&mut stripe, &mut settling, column)?;
2678        }
2679        Ok(stripe)
2680    }
2681
2682    /// One more part of a column with no global dictionary as a page, after the ones already in
2683    /// `stripe`. The parts have to come in order, since `settling` carries from one to the next.
2684    fn encode_page(
2685        stripe: &mut ColumnStripe,
2686        settling: &mut Settling,
2687        column: &Vector,
2688    ) -> Result<()> {
2689        let bytes = encode(column, settling)?;
2690        if bytes.len() > MAX_PAGE {
2691            return Err(invalid("column page exceeds the configured bound"));
2692        }
2693        // The range is built first because the sieve reads it rather than walking the column a
2694        // second time to find out how wide it is.
2695        let range = Range::of(column);
2696        // A sieve at least as large as the part it indexes is not written. A reader reads the
2697        // sieve to decide whether to read the part, so when the sieve is the larger of the two
2698        // it has already spent more than the read it is trying to avoid, and that holds even if
2699        // it rejects every time. It is a necessary condition rather than the whole rule, which
2700        // is that a sieve pays when its bytes are under the rejection rate times the part's,
2701        // but the rejection rate depends on what a query probes for and the writer does not
2702        // know that. The necessary half needs two numbers that are both in hand here.
2703        //
2704        // A column with a global dictionary gets none, because it already has an exact
2705        // membership index per stripe. Those do not come through here. See [`prepare`].
2706        let sieve =
2707            Sieve::of(column, &range, SIEVE_BUDGET).filter(|sieve| sieve.len() < bytes.len());
2708        stripe.sums.push(checksum(&bytes));
2709        stripe.pages.push(bytes);
2710        stripe.codes.push(None);
2711        stripe.sieves.push(sieve);
2712        stripe.ranges.push(range);
2713        Ok(())
2714    }
2715
2716    /// Writes every encoded dictionary block that is not in the file yet and forgets its bytes.
2717    ///
2718    /// This is what keeps a load from holding its dictionaries' payload. The blocks land between
2719    /// stripes wherever the writer is, which is fine because the index says where each one is.
2720    fn place_blocks(&mut self) -> Result<()> {
2721        if let Some(lent) = self.lent.clone() {
2722            return self.place_lent_blocks(&lent);
2723        }
2724        let mut dictionaries = std::mem::take(&mut self.dictionaries);
2725        let placed = dictionaries.iter_mut().flatten().try_for_each(|dictionary| {
2726            for block in std::mem::take(&mut dictionary.blocks) {
2727                let start = self.at;
2728                self.put(&block)?;
2729                dictionary.placed.push(Placed {
2730                    start,
2731                    length: block.len() as u64,
2732                    hash: checksum(&block),
2733                });
2734            }
2735            Ok(())
2736        });
2737        self.dictionaries = dictionaries;
2738        placed
2739    }
2740
2741    /// [`Writer::place_blocks`] while a [`Merger`] has the dictionaries.
2742    ///
2743    /// A column whose merge is running is passed over rather than waited for, because the writer's
2744    /// lock is held here and a merge of `URL` can take tens of milliseconds. Its blocks go out with
2745    /// a later stripe, or at the close.
2746    fn place_lent_blocks(&mut self, lent: &Lent) -> Result<()> {
2747        for column in lent.columns() {
2748            let Ok(mut held) = column.try_lock() else { continue };
2749            let Some(dictionary) = held.dictionary.as_mut() else { continue };
2750            for block in std::mem::take(&mut dictionary.blocks) {
2751                let start = self.at;
2752                self.put(&block)?;
2753                dictionary.placed.push(Placed {
2754                    start,
2755                    length: block.len() as u64,
2756                    hash: checksum(&block),
2757                });
2758            }
2759        }
2760        Ok(())
2761    }
2762
2763    /// Takes the dictionaries and the statistics back from the [`Merger`] that has them.
2764    ///
2765    /// A merge that starts after this is refused, since whatever it merged would be lost.
2766    fn reclaim(&mut self) -> Result<()> {
2767        let Some(lent) = self.lent.take() else { return Ok(()) };
2768        let (dictionaries, gathers) = lent.reclaim()?;
2769        self.dictionaries = dictionaries;
2770        self.gathers = gathers;
2771        Ok(())
2772    }
2773
2774    /// Writes the buffered parts as one stripe, each column's parts contiguous on disk.
2775    ///
2776    /// The same four steps a caller holding this writer behind a lock takes, with nobody else
2777    /// waiting between them. See [`prepare`].
2778    fn flush_pending(&mut self) -> Result<()> {
2779        if self.pending.is_empty() {
2780            return Ok(());
2781        }
2782        let held = std::mem::take(&mut self.pending);
2783        let prepared = self.preparer().prepare_held(held)?;
2784        let merged = self.merge_held(prepared)?;
2785        let paged = merged.pages()?;
2786        self.write_paged(paged)
2787    }
2788
2789    /// Writes one stripe whose pages are built, each column's parts contiguous on disk.
2790    fn write_stripe(&mut self, held: &[Part], encoded: Vec<ColumnStripe>) -> Result<()> {
2791        let width = self.table.fields.len();
2792        let parts = held.len();
2793        if encoded.len() != width {
2794            return Err(Error::internal("a stripe came to the writer with the wrong columns"));
2795        }
2796        let profile = self.profile.clone();
2797        if let Some(profile) = &profile {
2798            let rows = held.iter().map(|part| part.rows as u64).sum();
2799            let raw = held.iter().map(|part| part.footprint as u64).sum();
2800            let pages =
2801                encoded.iter().flat_map(|stripe| &stripe.pages).map(|page| page.len() as u64).sum();
2802            profile.moved(Stage::Pages, raw, pages, rows);
2803        }
2804        // Before a byte of the stripe is written, so that the blocks the stripe's pages were built
2805        // with, and any that were waiting on them, are let go of now rather than a stripe later.
2806        let timing = profile.as_deref().map(|profile| profile.span(Stage::Dictionary));
2807        let before = self.at;
2808        self.place_blocks()?;
2809        drop(timing);
2810        if let Some(profile) = &profile {
2811            profile.moved(Stage::Dictionary, 0, self.at - before, 0);
2812        }
2813        let timing = profile.as_deref().map(|profile| profile.span(Stage::Write));
2814        let before = self.at;
2815        let mut pages = Vec::with_capacity(width);
2816        let mut memberships = vec![None; width];
2817        let mut ranges = Vec::with_capacity(width);
2818        let mut index = Vec::with_capacity(width.saturating_mul(index_section(parts)?));
2819        // Every page of the stripe goes to the file in one call after the loop, since they sit
2820        // back to back from where the stripe starts and a page is often a few kilobytes.
2821        let start = self.at;
2822        let mut out = Vec::with_capacity(width.saturating_mul(parts));
2823        for stripe in &encoded {
2824            let offset = self.at;
2825            let section = index.len();
2826            let mut length = 0_usize;
2827            if stripe.sums.len() != stripe.pages.len() {
2828                return Err(Error::internal("a stripe's pages came without their checksums"));
2829            }
2830            for (bytes, &sum) in stripe.pages.iter().zip(&stripe.sums) {
2831                put_u32(
2832                    &mut index,
2833                    u32::try_from(bytes.len()).map_err(|_| invalid("part length overflow"))?,
2834                );
2835                put_u64(&mut index, sum);
2836                out.push(bytes.as_slice());
2837                length = length
2838                    .checked_add(bytes.len())
2839                    .ok_or_else(|| invalid("column page length overflow"))?;
2840            }
2841            let hash = checksum(&index[section..]);
2842            put_u64(&mut index, hash);
2843            if length > MAX_PAGE {
2844                return Err(invalid("column page exceeds the configured bound"));
2845            }
2846            self.at = self
2847                .at
2848                .checked_add(length as u64)
2849                .ok_or_else(|| invalid("native file length overflow"))?;
2850            pages.push(Span {
2851                offset,
2852                length: u32::try_from(length).map_err(|_| invalid("page length overflow"))?,
2853            });
2854            ranges.push(merged_range(stripe.ranges.iter().cloned()));
2855        }
2856        self.file.write_parts_at(start, &out)?;
2857        drop(out);
2858        for (membership, stripe) in memberships.iter_mut().zip(&encoded) {
2859            if stripe.codes.iter().all(Option::is_none) {
2860                continue;
2861            }
2862            let lists = stripe
2863                .codes
2864                .iter()
2865                .map(|codes| codes.clone().unwrap_or_default())
2866                .collect::<Vec<_>>();
2867            let bytes = encode_membership(&merged_codes(lists));
2868            let offset = self.at;
2869            self.put(&bytes)?;
2870            *membership = Some(Page {
2871                offset,
2872                length: u32::try_from(bytes.len())
2873                    .map_err(|_| invalid("membership page length overflow"))?,
2874                hash: checksum(&bytes),
2875            });
2876        }
2877        let mut sieves = vec![None; width];
2878        for (page, stripe) in sieves.iter_mut().zip(&encoded) {
2879            if stripe.sieves.iter().all(Option::is_none) {
2880                continue;
2881            }
2882            let bytes = encode_sieves(stripe.sieves.iter())?;
2883            let offset = self.at;
2884            self.put(&bytes)?;
2885            *page = Some(Page {
2886                offset,
2887                length: u32::try_from(bytes.len())
2888                    .map_err(|_| invalid("sieve page length overflow"))?,
2889                hash: checksum(&bytes),
2890            });
2891        }
2892        // A stripe of one part has the same rows in it as that part, so its own bounds are already
2893        // the part's and a page here would say what the directory says. Everywhere else the page is
2894        // written unless it comes to more than the column it indexes, which is the rule the sieves
2895        // go by and for the same reason: a reader reads this to decide whether to read the column,
2896        // so a page larger than the column has spent more than the read it is avoiding.
2897        let mut part_ranges = vec![None; width];
2898        if parts > 1 {
2899            for ((page, stripe), span) in part_ranges.iter_mut().zip(&encoded).zip(&pages) {
2900                let bytes = encode_part_ranges(&stripe.ranges)?;
2901                if bytes.len() >= span.length as usize {
2902                    continue;
2903                }
2904                let offset = self.at;
2905                self.put(&bytes)?;
2906                *page = Some(Page {
2907                    offset,
2908                    length: u32::try_from(bytes.len())
2909                        .map_err(|_| invalid("part range page length overflow"))?,
2910                    hash: checksum(&bytes),
2911                });
2912            }
2913        }
2914        let offset = self.at;
2915        self.put(&index)?;
2916        let index = Span {
2917            offset,
2918            length: u32::try_from(index.len())
2919                .map_err(|_| invalid("index page length overflow"))?,
2920        };
2921        let mut rows = 0_usize;
2922        let mut lengths = Vec::with_capacity(parts);
2923        let mut span = None;
2924        for part in held {
2925            rows = rows.checked_add(part.rows).ok_or_else(|| invalid("row count overflow"))?;
2926            lengths.push(u32::try_from(part.rows).map_err(|_| invalid("part row count overflow"))?);
2927            span = Some(span.map_or((part.order, part.order), |(first, _)| (first, part.order)));
2928        }
2929        self.order.push(span.ok_or_else(|| invalid("a stripe was flushed with no parts"))?);
2930        self.table.stripes.push(Stripe {
2931            rows,
2932            parts: lengths,
2933            index,
2934            pages,
2935            memberships: Pages::from_slots(memberships)?,
2936            sieves: Pages::from_slots(sieves)?,
2937            part_ranges: Pages::from_slots(part_ranges)?,
2938            zone: Zone::from_ranges(ranges),
2939        });
2940        drop(timing);
2941        if let Some(profile) = &profile {
2942            profile.moved(Stage::Write, 0, self.at - before, rows as u64);
2943        }
2944        Ok(())
2945    }
2946
2947    /// Finds exact heavy hitters without keeping a hash table for every numeric column while the
2948    /// load is live. The pages are already in the target file, so one column at a time uses a
2949    /// bounded Misra-Gries candidate table and then recounts only those candidates.
2950    ///
2951    /// The first of those passes also counts the column's distinct values exactly, up to the cap in
2952    /// [`distinct`], which is the number a string column gets from its dictionary. It comes back
2953    /// beside the summary because a column whose heavy hitters cannot be proved can still have been
2954    /// counted.
2955    ///
2956    /// The tables are keyed by a value's sixty four bits rather than by [`FrequencyValue`], and a
2957    /// null is counted beside them. Every integer type the format stores fits in those bits, so
2958    /// within one column two values share bits only if they are the same value, and a sixteen byte
2959    /// entry keeps the whole candidate table in the second level cache where the forty eight byte
2960    /// one did not. The null takes part in the candidate table exactly as a key would: it holds a
2961    /// place while its count is above zero, and it is decremented with the rest.
2962    ///
2963    /// `counted` is false for a column whose sketch says its distinct values are far past what the
2964    /// exact set holds. It still gets its frequencies, and a count only if it turns out to have
2965    /// fewer values than the candidate table, which is the count that costs nothing.
2966    fn numeric_frequency(
2967        &self,
2968        column: usize,
2969        counted: bool,
2970        dense: Option<(u64, usize)>,
2971    ) -> Result<(Option<FrequencySummary>, Option<u64>)> {
2972        let signed = match self.table.fields[column].ty {
2973            LogicalType::TinyInt
2974            | LogicalType::SmallInt
2975            | LogicalType::Integer
2976            | LogicalType::BigInt
2977            | LogicalType::Date
2978            | LogicalType::Timestamp => true,
2979            LogicalType::UTinyInt
2980            | LogicalType::USmallInt
2981            | LogicalType::UInteger
2982            | LogicalType::UBigInt => false,
2983            _ => return Ok((None, None)),
2984        };
2985        let value_of = |bits: Option<u64>| match bits {
2986            None => FrequencyValue::Null,
2987            Some(bits) => integer_value(bits, signed),
2988        };
2989        // A column the writer's tally held whole has its exact counts already, gathered as the rows
2990        // went past, so the pages are not read back to count them again. On `hits` that is most of
2991        // the flag and enum columns. The tally only speaks for the whole column when it saw every
2992        // row, which is the same check the statistics make before they are written.
2993        let tallied = self
2994            .gathers
2995            .get(column)
2996            .and_then(Option::as_ref)
2997            .filter(|gather| gather.rows() == self.table.rows as u64)
2998            .and_then(stats::Gather::frequencies)
2999            .and_then(|(values, nulls)| {
3000                let entries = values
3001                    .iter()
3002                    .map(|(value, count)| {
3003                        let value = value_of(Some(frequency_bits(value)?));
3004                        Some(FrequencyEntry { value, count: *count })
3005                    })
3006                    .chain((nulls != 0).then_some(Some(FrequencyEntry {
3007                        value: FrequencyValue::Null,
3008                        count: nulls,
3009                    })))
3010                    .collect::<Option<Vec<_>>>()?;
3011                Some((entries, values.len() as u64))
3012            });
3013        // A column the sketch expects to fit the exact set is counted there, every value with the
3014        // rows holding it, which is its distinct count and its frequencies from one read of its
3015        // pages. Only a column past the set's cap goes through the candidate table.
3016        let exact = match (&tallied, counted) {
3017            (None, true) => self.exact_frequency(column, signed, dense)?,
3018            _ => None,
3019        };
3020        let (mut entries, decrements, distinct_count) = match (tallied, exact) {
3021            (Some((entries, distinct)), _) => (entries, 0, Some(distinct)),
3022            (None, Some((Some(entries), distinct))) => (entries, 0, Some(distinct)),
3023            (None, Some((None, distinct))) => return Ok((None, Some(distinct))),
3024            (None, None) => {
3025                // Rows arrive a run of equal values at a time, because a sorted column is runs and
3026                // a flag column is mostly one value, so a run is counted and inserted once rather
3027                // than per row.
3028                let mut first = Candidates::default();
3029                let mut run = Run::default();
3030                self.visit_numeric(column, signed, |_, bits| {
3031                    if let Some((ended, times)) = run.push(bits) {
3032                        first.add(ended, times);
3033                    }
3034                })?;
3035                if let Some((bits, times)) = run.take() {
3036                    first.add(bits, times);
3037                }
3038                // Until a candidate is turned away the table holds every value the column has, so
3039                // its size is the count.
3040                let (nulls, decrements) = (first.nulls, first.decrements);
3041                let distinct_count = (decrements == 0).then_some(first.held as u64);
3042                let (exact, null_count) = if decrements == 0 {
3043                    let exact = first
3044                        .pairs()
3045                        .map(|(bits, count)| (bits, u64::from(count)))
3046                        .collect::<FrequencyMap<_>>();
3047                    (exact, (nulls != 0).then_some(u64::from(nulls)))
3048                } else {
3049                    let mut lower = first.pairs().map(|(_, count)| count).collect::<Vec<_>>();
3050                    if nulls != 0 {
3051                        lower.push(nulls);
3052                    }
3053                    lower.sort_unstable_by(|left, right| right.cmp(left));
3054                    if lower.len() < FREQUENCY_BUILD_RANK
3055                        || u64::from(lower[FREQUENCY_BUILD_RANK - 1]) <= decrements
3056                    {
3057                        return Ok((None, distinct_count));
3058                    }
3059                    // Counted beside the slot each candidate sits in, since the table is not
3060                    // changed again and a lookup in it is the one probe the first pass made.
3061                    let mut recounts = vec![0_u64; first.slots.len()];
3062                    let mut null_count = (nulls != 0).then_some(0_u64);
3063                    let mut recount = |bits: Option<u64>, times: u32| {
3064                        let held = match bits {
3065                            Some(bits) => first.position(bits).map(|at| &mut recounts[at]),
3066                            None => null_count.as_mut(),
3067                        };
3068                        if let Some(count) = held {
3069                            *count = count.saturating_add(u64::from(times));
3070                        }
3071                    };
3072                    let mut run = Run::default();
3073                    self.visit_numeric(column, signed, |_, bits| {
3074                        if let Some((bits, times)) = run.push(bits) {
3075                            recount(bits, times);
3076                        }
3077                    })?;
3078                    if let Some((bits, times)) = run.take() {
3079                        recount(bits, times);
3080                    }
3081                    let exact = first
3082                        .slots
3083                        .iter()
3084                        .zip(&recounts)
3085                        .filter(|(slot, _)| slot.count != 0)
3086                        .map(|(slot, &count)| (slot.bits, count))
3087                        .collect::<FrequencyMap<_>>();
3088                    (exact, null_count)
3089                };
3090                let entries = exact
3091                    .into_iter()
3092                    .map(|(bits, count)| FrequencyEntry { value: value_of(Some(bits)), count })
3093                    .chain(
3094                        null_count
3095                            .map(|count| FrequencyEntry { value: FrequencyValue::Null, count }),
3096                    )
3097                    .collect::<Vec<_>>();
3098                (entries, decrements, distinct_count)
3099            }
3100        };
3101        let mut omitted_max = keep_most_frequent(&mut entries).max(decrements);
3102        // A complete value-to-count table is also the result of grouping this column.
3103        // Keep up to two leading frequencies for selectivity and equality predicates,
3104        // but leave multi-value grouped counts to the encoded rows at query time.
3105        if omitted_max == 0 && entries.len() > 1 {
3106            let retained = entries.len().saturating_sub(1).min(2);
3107            omitted_max = entries[retained].count;
3108            entries.truncate(retained);
3109        }
3110        let kept_rows = entries.iter().try_fold(0_u64, |total, entry| {
3111            total.checked_add(entry.count).filter(|&total| total <= FREQUENCY_ORDINALS as u64)
3112        });
3113        let mut ordinals = Vec::new();
3114        let mut ordinal_entries = Vec::new();
3115        if let Some(kept_rows) = kept_rows {
3116            let mut kept = FrequencyMap::default();
3117            let mut null_kept = None;
3118            for (at, entry) in entries.iter().enumerate() {
3119                let at = u16::try_from(at)
3120                    .map_err(|_| invalid("too many retained frequency entries"))?;
3121                match entry.value {
3122                    FrequencyValue::Integer(value) => {
3123                        kept.insert(value as u64, at);
3124                    }
3125                    FrequencyValue::Null => null_kept = Some(at),
3126                    FrequencyValue::Code(_) => {}
3127                }
3128            }
3129            ordinals.reserve(usize::try_from(kept_rows).unwrap_or(FREQUENCY_ORDINALS));
3130            ordinal_entries.reserve(usize::try_from(kept_rows).unwrap_or(FREQUENCY_ORDINALS));
3131            self.visit_numeric(column, signed, |ordinal, bits| {
3132                let held = match bits {
3133                    Some(bits) => kept.get(&bits).copied(),
3134                    None => null_kept,
3135                };
3136                if let Some(entry) = held {
3137                    ordinals.push(ordinal);
3138                    ordinal_entries.push(entry);
3139                }
3140            })?;
3141        }
3142        Ok((
3143            Some(FrequencySummary { entries, omitted_max, ordinals, ordinal_entries }),
3144            distinct_count,
3145        ))
3146    }
3147
3148    /// The bits of a column's lowest value and how many values its range holds, when counting it in
3149    /// a [`distinct::DenseCounts`] would take no more memory than the set it would otherwise be
3150    /// charged, or a mebibyte, whichever is more.
3151    ///
3152    /// Only for a column the statistics saw every row of, since otherwise its ends may not be its
3153    /// ends, and a table of fewer than `u32::MAX` rows, so that a count fits in its slot.
3154    fn dense_range(&self, gather: &stats::Gather, set: usize) -> Option<(u64, usize)> {
3155        let rows = self.table.rows;
3156        if gather.rows() != rows as u64 || u32::try_from(rows).is_err() {
3157            return None;
3158        }
3159        let (low, high) = gather.span()?;
3160        let len = usize::try_from(high.checked_sub(low)?.checked_add(1)?).ok()?;
3161        #[allow(clippy::cast_sign_loss, clippy::cast_possible_truncation)]
3162        let bits = low as u64;
3163        (len.checked_mul(size_of::<u32>())? <= set.max(1 << 20)).then_some((bits, len))
3164    }
3165
3166    /// Counts every value of an integer column and the rows holding it, and hands back the
3167    /// frequency entries worth keeping beside the distinct count, or nothing for a column with more
3168    /// values than [`distinct::ExactCounts`] keeps.
3169    ///
3170    /// The entries are `None` for a column with no value common enough to be worth a synopsis. The
3171    /// rule is the one the candidate table applied. A column with more values than that table holds
3172    /// keeps its frequencies only if its tenth commonest value is held by more rows than a
3173    /// Misra-Gries table of [`FREQUENCY_CANDIDATES`] could have decremented it by, which is its rows
3174    /// over one more than the candidates. The counts kept are exact either way, so the largest one
3175    /// left out is exact too and not the table's bound on it.
3176    ///
3177    /// Only the commonest entries and the ones tied with the first left out are built, since on a
3178    /// column of a million values the rest are thrown away the moment they are ranked.
3179    fn exact_frequency(
3180        &self,
3181        column: usize,
3182        signed: bool,
3183        dense: Option<(u64, usize)>,
3184    ) -> Result<Option<(Option<Vec<FrequencyEntry>>, u64)>> {
3185        // A column whose ends are close together is counted in a flat array. A value outside the
3186        // ends it was given, which would be a bug in the statistics, sends it to the set instead.
3187        if let Some((low, len)) = dense {
3188            let mut counts = distinct::DenseCounts::new(low, len);
3189            let nulls =
3190                self.count_numeric(column, signed, |bits, times| counts.insert(bits, times))?;
3191            if let Some(distinct) = counts.count() {
3192                let Some(distinct) = distinct else { return Ok(None) };
3193                return Ok(Some(self.frequent_entries(signed, distinct, nulls, |visit| {
3194                    counts.visit(visit);
3195                })));
3196            }
3197        }
3198        let mut set = distinct::ExactCounts::new();
3199        let nulls = self.count_numeric(column, signed, |bits, times| set.insert(bits, times))?;
3200        let Some(distinct) = set.count() else {
3201            return Ok(None);
3202        };
3203        Ok(Some(self.frequent_entries(signed, distinct, nulls, |visit| {
3204            set.visit(visit);
3205        })))
3206    }
3207
3208    /// Hands every run of equal non-null values in an integer column to `add` as its bits and its
3209    /// length, and answers how many rows were null.
3210    fn count_numeric(
3211        &self,
3212        column: usize,
3213        signed: bool,
3214        mut add: impl FnMut(u64, u32),
3215    ) -> Result<u64> {
3216        let mut nulls = 0_u64;
3217        let mut run = Run::default();
3218        let mut take = |bits: Option<u64>, times: u32| match bits {
3219            Some(bits) => add(bits, times),
3220            None => nulls += u64::from(times),
3221        };
3222        self.visit_numeric(column, signed, |_, bits| {
3223            if let Some((bits, times)) = run.push(bits) {
3224                take(bits, times);
3225            }
3226        })?;
3227        if let Some((bits, times)) = run.take() {
3228            take(bits, times);
3229        }
3230        Ok(nulls)
3231    }
3232
3233    /// The frequency entries worth keeping out of a column's exact counts, which `visit` hands over
3234    /// as bits and rows once for each call it gets. See [`Self::exact_frequency`] for the rule.
3235    fn frequent_entries(
3236        &self,
3237        signed: bool,
3238        distinct: u64,
3239        nulls: u64,
3240        mut visit: impl FnMut(&mut dyn FnMut(u64, u64)),
3241    ) -> (Option<Vec<FrequencyEntry>>, u64) {
3242        // The commonest counts, one more than the entries kept so that the first left out is here.
3243        let mut top = std::collections::BinaryHeap::with_capacity(FREQUENCY_ENTRIES + 2);
3244        let mut rank = |count: u64| {
3245            if top.len() <= FREQUENCY_ENTRIES {
3246                top.push(Reverse(count));
3247            } else if top.peek().is_some_and(|&Reverse(least)| count > least) {
3248                top.pop();
3249                top.push(Reverse(count));
3250            }
3251        };
3252        visit(&mut |_, count| rank(count));
3253        if nulls != 0 {
3254            rank(nulls);
3255        }
3256        let top = top.into_sorted_vec();
3257        let values = distinct + u64::from(nulls != 0);
3258        if values > FREQUENCY_CANDIDATES as u64 {
3259            let bound = self.table.rows as u64 / (FREQUENCY_CANDIDATES as u64 + 1);
3260            if top.get(FREQUENCY_BUILD_RANK - 1).is_none_or(|&Reverse(count)| count <= bound) {
3261                return (None, distinct);
3262            }
3263        }
3264        let least = top.get(FREQUENCY_ENTRIES).map_or(0, |&Reverse(count)| count);
3265        let mut entries = Vec::with_capacity(FREQUENCY_ENTRIES + 1);
3266        visit(&mut |bits, count| {
3267            if count >= least {
3268                entries.push(FrequencyEntry { value: integer_value(bits, signed), count });
3269            }
3270        });
3271        if nulls != 0 && nulls >= least {
3272            entries.push(FrequencyEntry { value: FrequencyValue::Null, count: nulls });
3273        }
3274        (Some(entries), distinct)
3275    }
3276
3277    /// Hands every row of an integer column to `visit` as its ordinal and its sixty four bits, or
3278    /// `None` for a null.
3279    ///
3280    /// `signed` says which of the two readings the column has. A packed unsigned column would come
3281    /// back from `signed_block` as a base plus a code in `i64`, which wraps for a value past the top
3282    /// of `BIGINT`, so only a signed column takes the block path.
3283    fn visit_numeric(
3284        &self,
3285        column: usize,
3286        signed: bool,
3287        mut visit: impl FnMut(u64, Option<u64>),
3288    ) -> Result<()> {
3289        let ty = &self.table.fields[column].ty;
3290        let mut start = 0_u64;
3291        let mut block = Vec::new();
3292        for stripe in &self.table.stripes {
3293            let spans = read_index(&self.file, stripe, column)?;
3294            let page = stripe.pages[column];
3295            let mut bytes = vec![0; page.length as usize];
3296            read_at(&self.file, page.offset, &mut bytes)?;
3297            for (span, &rows) in spans.iter().zip(&stripe.parts) {
3298                let part = part_bytes(&bytes, *span)?;
3299                if checksum(part) != span.hash {
3300                    return Err(invalid("column page checksum differs while building frequencies"));
3301                }
3302                let rows = rows as usize;
3303                let vector = decode(ty, rows, part, None)?;
3304                // Every signed layout a numeric column decodes to, which is every column of `hits`,
3305                // comes out as one run of `i64` and is walked as a slice. The row path below is for
3306                // the unsigned types and anything else that cannot be handed over that way.
3307                if signed && vector.signed_block(&mut block) && block.len() == rows {
3308                    if vector.none_null() {
3309                        for (row, &value) in block.iter().enumerate() {
3310                            visit(start.saturating_add(row as u64), Some(value as u64));
3311                        }
3312                    } else {
3313                        for (row, &value) in block.iter().enumerate() {
3314                            let bits = (!vector.is_null_at(row)).then_some(value as u64);
3315                            visit(start.saturating_add(row as u64), bits);
3316                        }
3317                    }
3318                    start = start.saturating_add(rows as u64);
3319                    continue;
3320                }
3321                // row at a time: frequency construction visits decoded values to update bounded candidates.
3322                for row in 0..rows {
3323                    let bits = if vector.is_null_at(row) {
3324                        None
3325                    } else {
3326                        // An unsigned column has no signed reading, and the documented fallback is
3327                        // the value itself. Every width the format stores fits in sixty four bits,
3328                        // so nothing is lost on the way through.
3329                        let widened = match vector.signed_at(row) {
3330                            Some(value) => Some(value as u64),
3331                            None => match vector.value_at(row) {
3332                                Value::UTinyInt(value) => Some(u64::from(value)),
3333                                Value::USmallInt(value) => Some(u64::from(value)),
3334                                Value::UInteger(value) => Some(u64::from(value)),
3335                                Value::UBigInt(value) => Some(value),
3336                                _ => None,
3337                            },
3338                        };
3339                        Some(widened.ok_or_else(|| {
3340                            invalid("numeric frequency page did not contain an integer value")
3341                        })?)
3342                    };
3343                    visit(start.saturating_add(row as u64), bits);
3344                }
3345                start = start.saturating_add(rows as u64);
3346            }
3347        }
3348        Ok(())
3349    }
3350
3351    /// The columns that get numeric frequencies, which are the integer, date and timestamp ones.
3352    fn numeric_columns(&self) -> Vec<usize> {
3353        self.table
3354            .fields
3355            .iter()
3356            .enumerate()
3357            .filter_map(|(column, field)| {
3358                matches!(
3359                    field.ty,
3360                    LogicalType::TinyInt
3361                        | LogicalType::SmallInt
3362                        | LogicalType::Integer
3363                        | LogicalType::BigInt
3364                        | LogicalType::UTinyInt
3365                        | LogicalType::USmallInt
3366                        | LogicalType::UInteger
3367                        | LogicalType::UBigInt
3368                        | LogicalType::Date
3369                        | LogicalType::Timestamp
3370                )
3371                .then_some(column)
3372            })
3373            .collect()
3374    }
3375
3376    /// Reads one stable dictionary code column only at sorted table-wide row ordinals.
3377    #[allow(dead_code)]
3378    fn stable_codes_at(&self, column: usize, ordinals: &[u64]) -> Result<Option<Vec<Option<u32>>>> {
3379        if self.dictionaries.get(column).and_then(Option::as_ref).is_none() {
3380            return Ok(None);
3381        }
3382        if ordinals.windows(2).any(|pair| pair[0] >= pair[1]) {
3383            return Err(invalid("frequency ordinals are not sorted and unique"));
3384        }
3385        let mut out = Vec::with_capacity(ordinals.len());
3386        let mut wanted = 0;
3387        let mut stripe_start = 0_u64;
3388        for stripe in &self.table.stripes {
3389            let stripe_end = stripe_start.saturating_add(stripe.rows as u64);
3390            if wanted == ordinals.len() || ordinals[wanted] >= stripe_end {
3391                stripe_start = stripe_end;
3392                continue;
3393            }
3394            let spans = read_index(&self.file, stripe, column)?;
3395            let page = stripe.pages[column];
3396            let mut bytes = vec![0; page.length as usize];
3397            read_at(&self.file, page.offset, &mut bytes)?;
3398            let mut part_start = stripe_start;
3399            for (span, &rows) in spans.iter().zip(&stripe.parts) {
3400                let part_end = part_start.saturating_add(u64::from(rows));
3401                if wanted < ordinals.len() && ordinals[wanted] < part_end {
3402                    let part = part_bytes(&bytes, *span)?;
3403                    if checksum(part) != span.hash {
3404                        return Err(invalid(
3405                            "column page checksum differs while building pair frequencies",
3406                        ));
3407                    }
3408                    let upto = ordinals.partition_point(|&ordinal| ordinal < part_end);
3409                    let positions = ordinals[wanted..upto]
3410                        .iter()
3411                        .map(|&ordinal| {
3412                            usize::try_from(ordinal.saturating_sub(part_start))
3413                                .map_err(|_| invalid("frequency row offset does not fit in memory"))
3414                        })
3415                        .collect::<Result<Vec<_>>>()?;
3416                    if !decode_selected_stable_codes(rows as usize, part, &positions, &mut out)? {
3417                        return Ok(None);
3418                    }
3419                    wanted = upto;
3420                }
3421                part_start = part_end;
3422            }
3423            stripe_start = stripe_end;
3424        }
3425        if wanted != ordinals.len() {
3426            return Err(invalid("frequency ordinal is outside the table"));
3427        }
3428        Ok(Some(out))
3429    }
3430
3431    /// Derives bounded two-key leaders from numeric anchor ordinals and stable string codes.
3432    #[allow(dead_code)]
3433    fn pair_frequencies(
3434        &self,
3435        frequencies: &[Option<Frequencies>],
3436    ) -> Result<Vec<PairFrequencySummary>> {
3437        let anchors = frequencies
3438            .iter()
3439            .enumerate()
3440            .filter_map(|(column, summary)| {
3441                // A writer holds every synopsis it counted, so there is nothing stored to skip.
3442                match summary {
3443                    Some(Frequencies::Held(summary)) => Some(summary),
3444                    _ => None,
3445                }
3446                .filter(|summary| {
3447                    !summary.ordinals.is_empty()
3448                        && summary.ordinal_entries.len() == summary.ordinals.len()
3449                })
3450                .cloned()
3451                .map(|summary| (column, summary))
3452            })
3453            .collect::<Vec<_>>();
3454        let strings = self
3455            .dictionaries
3456            .iter()
3457            .enumerate()
3458            .filter_map(|(column, dictionary)| dictionary.as_ref().map(|_| column))
3459            .collect::<Vec<_>>();
3460        let mut summaries = Vec::new();
3461        for (first, anchors) in anchors {
3462            for &second in &strings {
3463                if summaries.len() == MAX_PAIR_FREQUENCIES {
3464                    return Ok(summaries);
3465                }
3466                let Some(codes) = self.stable_codes_at(second, &anchors.ordinals)? else {
3467                    continue;
3468                };
3469                if codes.len() != anchors.ordinal_entries.len() {
3470                    return Err(invalid("pair frequency columns have different lengths"));
3471                }
3472                let mut counts = HashMap::<(u16, Option<u32>), u64>::new();
3473                for (&anchor, code) in anchors.ordinal_entries.iter().zip(codes) {
3474                    *counts.entry((anchor, code)).or_default() += 1;
3475                }
3476                let mut entries = counts
3477                    .into_iter()
3478                    .map(|((first_entry, second), count)| PairFrequencyEntry {
3479                        first_entry,
3480                        second,
3481                        count,
3482                    })
3483                    .collect::<Vec<_>>();
3484                entries.sort_unstable_by(|left, right| {
3485                    right
3486                        .count
3487                        .cmp(&left.count)
3488                        .then_with(|| left.first_entry.cmp(&right.first_entry))
3489                        .then_with(|| left.second.cmp(&right.second))
3490                });
3491                let pair_omitted = entries.get(FREQUENCY_ENTRIES).map_or(0, |entry| entry.count);
3492                entries.truncate(FREQUENCY_ENTRIES);
3493                summaries.push(PairFrequencySummary {
3494                    first: u16::try_from(first)
3495                        .map_err(|_| invalid("pair frequency column index overflows"))?,
3496                    second: u16::try_from(second)
3497                        .map_err(|_| invalid("pair frequency column index overflows"))?,
3498                    entries,
3499                    omitted_max: anchors.omitted_max.max(pair_omitted),
3500                });
3501            }
3502        }
3503        Ok(summaries)
3504    }
3505
3506    /// Writes the directory of the table this writer is on and says where it went.
3507    ///
3508    /// Everything [`Writer::finish`] used to do except the two writes that publish. Pulling it out
3509    /// is what lets a second table follow a first: the bytes of a closed table are complete and
3510    /// addressable while nothing yet points at them, and the pointer is the last write of the
3511    /// commit.
3512    ///
3513    /// # Errors
3514    ///
3515    /// If directory encoding or writing fails.
3516    fn close(&mut self) -> Result<Entry> {
3517        self.reclaim()?;
3518        self.flush_pending()?;
3519        // The rest of a table is its statistics, its dictionaries and its directory. The dictionary
3520        // work is charged as its own stage, because ranking a global dictionary can be most of what
3521        // this costs, and the rest as publish.
3522        let profile = self.profile.clone();
3523        let timing = profile.as_deref().map(|profile| profile.span(Stage::Publish));
3524        let before = self.at;
3525        let mut stripes = std::mem::take(&mut self.order)
3526            .into_iter()
3527            .zip(std::mem::take(&mut self.table.stripes))
3528            .collect::<Vec<_>>();
3529        stripes.sort_by_key(|(order, _)| order.0);
3530        let mut previous: Option<(u64, u64)> = None;
3531        for ((first, last), _) in &stripes {
3532            if previous.is_some_and(|previous| previous >= *first) {
3533                return Err(invalid("chunks did not arrive in source order"));
3534            }
3535            previous = Some(*last);
3536        }
3537        self.table.stripes = stripes.into_iter().map(|(_, stripe)| stripe).collect();
3538        drop(timing);
3539        let timing = profile.as_deref().map(|profile| profile.span(Stage::Dictionary));
3540        let placing = self.at;
3541        finish_dictionaries(&mut self.dictionaries)?;
3542        self.place_blocks()?;
3543        for dictionary in self.dictionaries.iter_mut().flatten() {
3544            dictionary.release_lookup();
3545            dictionary.recharge(profile.as_deref());
3546        }
3547        let (numeric, closed) = self.close_columns()?;
3548        let (frequencies, distincts): (Vec<Option<FrequencySummary>>, Vec<_>) =
3549            numeric.into_iter().unzip();
3550        let frequencies =
3551            frequencies.into_iter().map(|held| held.map(Frequencies::Held)).collect::<Vec<_>>();
3552        // Pair leaders are query results, not reusable column statistics.
3553        let pairs = Vec::new();
3554        self.table.frequencies = frequencies;
3555        self.table.distincts = distincts;
3556        self.table.pair_frequencies = pairs;
3557        if let Some(profile) = &profile {
3558            profile.release(self.dictionaries.iter().flatten().map(|held| held.charged).sum());
3559        }
3560        self.table.demoted = self
3561            .dictionaries
3562            .iter()
3563            .map(|dictionary| dictionary.as_ref().is_some_and(|held| held.demoted))
3564            .collect();
3565        if !self.table.demoted.contains(&true) {
3566            self.table.demoted = Vec::new();
3567        }
3568        self.dictionaries = Vec::new();
3569        self.table.dictionary_payloads = vec![0; self.table.fields.len()];
3570        self.table.frequency_texts = vec![Vec::new(); self.table.fields.len()];
3571        self.table.host_groups = None;
3572        for (index, closed) in closed.into_iter().enumerate() {
3573            let Some(closed) = closed else { continue };
3574            let ClosedDictionary { distinct, frequencies, texts, hosts, encoded, payload } = closed;
3575            self.table.distincts[index] = distinct;
3576            self.table.frequencies[index] = frequencies.map(Frequencies::Held);
3577            self.table.frequency_texts[index] = texts;
3578            if hosts.is_some() {
3579                self.table.host_groups = hosts;
3580            }
3581            let offset = self.at;
3582            self.put(&encoded.index)?;
3583            self.put(&encoded.ranks)?;
3584            self.put(&encoded.grams)?;
3585            self.table.dictionary_payloads[index] = payload;
3586            let length = encoded
3587                .index
3588                .len()
3589                .checked_add(encoded.ranks.len())
3590                .and_then(|len| len.checked_add(encoded.grams.len()))
3591                .ok_or_else(|| invalid("dictionary page length overflow"))?;
3592            self.table.dictionaries[index] = Some(Page {
3593                offset,
3594                length: u32::try_from(length)
3595                    .map_err(|_| invalid("dictionary page length overflow"))?,
3596                hash: checksum(&encoded.index),
3597            });
3598        }
3599        drop(timing);
3600        let timing = profile.as_deref().map(|profile| profile.span(Stage::Publish));
3601        let placed = self.at - placing;
3602        self.write_stats()?;
3603        let directory = encode_directory(&self.table)?;
3604        if directory.len() > MAX_DIRECTORY {
3605            return Err(invalid("directory exceeds the configured bound"));
3606        }
3607        let offset = self.at;
3608        self.put(&directory)?;
3609        drop(timing);
3610        if let Some(profile) = &profile {
3611            profile.moved(Stage::Dictionary, 0, placed, 0);
3612            profile.moved(Stage::Publish, 0, self.at - before - placed, 0);
3613        }
3614        Ok(Entry {
3615            name: self.table.name.clone(),
3616            fields: self.table.fields.clone(),
3617            rows: self.table.rows,
3618            nonzero: vec![None; self.table.fields.len()],
3619            aggregates: table_aggregate_sums(&self.table),
3620            distincts: self.table.distincts.clone(),
3621            extremes: table_integer_extremes(&self.table),
3622            frequencies: table_complete_numeric_frequencies(&self.table),
3623            directory: Page {
3624                offset,
3625                length: u32::try_from(directory.len())
3626                    .map_err(|_| invalid("directory length overflow"))?,
3627                hash: checksum(&directory),
3628            },
3629        })
3630    }
3631
3632    /// Every numeric column's frequencies and every global dictionary's page and statistics, by
3633    /// column, as many columns at a time as [`CLOSE_BYTES`] allows.
3634    ///
3635    /// The two kinds read what is already written and write nothing, so they share one set of
3636    /// threads. Each was most of a second on `hits` with the other waiting for it, and neither keeps
3637    /// every core busy on its own. The most expensive column that fits is the one taken next, so
3638    /// the long ones start first and the short ones fill in behind them. A column that does not fit
3639    /// waits for one that is closing to finish, unless nothing is closing, in which case it goes
3640    /// alone.
3641    ///
3642    /// A numeric column is charged the exact distinct set its sketch says it will need, and one the
3643    /// sketch puts far past what that set can hold does not build it, because the set would fill,
3644    /// give up and have held 512 MiB for nothing. A column with no sketch is charged the whole set.
3645    /// Each job charges itself as its own span, publish for the numeric ones and dictionary for the
3646    /// rest, because it runs on a thread of its own and a span on this one would see the wall time
3647    /// and none of the CPU.
3648    #[allow(clippy::type_complexity)]
3649    fn close_columns(
3650        &self,
3651    ) -> Result<(Vec<(Option<FrequencySummary>, Option<u64>)>, Vec<Option<ClosedDictionary>>)> {
3652        let numeric = self.numeric_columns().into_iter().map(|column| {
3653            let gather = self.gathers.get(column).and_then(Option::as_ref);
3654            let estimate = gather.and_then(stats::Gather::distinct);
3655            let counted = !estimate.is_some_and(distinct::beyond);
3656            let set =
3657                if counted { distinct::bytes_for(estimate.unwrap_or(f64::INFINITY)) } else { 0 };
3658            let dense = gather.filter(|_| counted).and_then(|gather| self.dense_range(gather, set));
3659            let set = dense.map_or(set, |(_, len)| len * size_of::<u32>());
3660            let cost = self.table.rows.saturating_mul(weight(&self.table.fields[column].ty));
3661            (Closing::Numeric { column, counted, dense }, NUMERIC_CLOSE_BYTES + set, cost)
3662        });
3663        let dictionaries =
3664            self.dictionaries.iter().enumerate().filter_map(|(index, dictionary)| {
3665                let dictionary = dictionary.as_ref()?;
3666                let bytes = dictionary.closing_bytes();
3667                Some((Closing::Dictionary { index, dictionary }, bytes, bytes))
3668            });
3669        let mut jobs = numeric.chain(dictionaries).collect::<Vec<_>>();
3670        jobs.sort_by_key(|&(_, _, cost)| cost);
3671        let columns = self.table.fields.len();
3672        let mut frequencies = vec![(None, None); columns];
3673        let mut closed = (0..columns).map(|_| None).collect::<Vec<_>>();
3674        let profile = self.profile.as_deref();
3675        let run = |job: Closing<'_>, bytes: usize| -> Result<Closed> {
3676            let _holding = profile.map(|profile| profile.holding(bytes as u64));
3677            let closed = match job {
3678                Closing::Numeric { column, counted, dense } => {
3679                    let _timing = profile.map(|profile| profile.span(Stage::Publish));
3680                    Closed::Numeric(column, self.numeric_frequency(column, counted, dense)?)
3681                }
3682                Closing::Dictionary { index, dictionary } => {
3683                    let _timing = profile.map(|profile| profile.span(Stage::Dictionary));
3684                    Closed::Dictionary(index, self.close_dictionary(index, dictionary)?)
3685                }
3686            };
3687            // A dictionary's decoded values or a column's distinct set were just dropped, and the
3688            // next job is about to take as much again.
3689            rudb_common::heap::release();
3690            Ok(closed)
3691        };
3692        let workers = close_workers().min(jobs.len());
3693        let pieces = if workers <= 1 {
3694            jobs.into_iter().map(|(job, bytes, _)| run(job, bytes)).collect::<Result<Vec<_>>>()?
3695        } else {
3696            // The columns not taken yet, cheapest first, and the bytes the ones closing now hold.
3697            let state = Mutex::new((jobs, 0_usize));
3698            let finished = Condvar::new();
3699            std::thread::scope(|scope| {
3700                (0..workers)
3701                    .map(|_| {
3702                        scope.spawn(|| {
3703                            let mut mine = Vec::new();
3704                            loop {
3705                                let mut held = state.lock().map_err(|_| {
3706                                    Error::internal("a native close worker panicked")
3707                                })?;
3708                                let (job, bytes) = loop {
3709                                    let (jobs, busy) = &mut *held;
3710                                    if jobs.is_empty() {
3711                                        return Ok(mine);
3712                                    }
3713                                    let fits = jobs.iter().rposition(|&(_, bytes, _)| {
3714                                        *busy == 0 || busy.saturating_add(bytes) <= CLOSE_BYTES
3715                                    });
3716                                    if let Some(at) = fits {
3717                                        let (job, bytes, _) = jobs.remove(at);
3718                                        *busy += bytes;
3719                                        break (job, bytes);
3720                                    }
3721                                    held = finished.wait(held).map_err(|_| {
3722                                        Error::internal("a native close worker panicked")
3723                                    })?;
3724                                };
3725                                drop(held);
3726                                // Given back on the way out whether the close worked, failed or
3727                                // panicked, so that a worker waiting for room is never left waiting.
3728                                let _room = Room { state: &state, finished: &finished, bytes };
3729                                mine.push(run(job, bytes)?);
3730                            }
3731                        })
3732                    })
3733                    .collect::<Vec<_>>()
3734                    .into_iter()
3735                    .map(|handle| {
3736                        handle
3737                            .join()
3738                            .map_err(|_| Error::internal("a native close worker panicked"))?
3739                    })
3740                    .collect::<Result<Vec<_>>>()
3741            })?
3742            .into_iter()
3743            .flatten()
3744            .collect()
3745        };
3746        for piece in pieces {
3747            match piece {
3748                Closed::Numeric(column, summary) => frequencies[column] = summary,
3749                Closed::Dictionary(index, one) => closed[index] = Some(one),
3750            }
3751        }
3752        Ok((frequencies, closed))
3753    }
3754
3755    /// One global dictionary's page and statistics, built from what is already in the file.
3756    ///
3757    /// Nothing is written here, so that [`Self::close`] can run this beside the numeric frequencies
3758    /// and put the pages down afterwards in column order, which is where they always went. The
3759    /// column's values are decoded in here and dropped before it returns, and
3760    /// [`Self::close_columns`] decides how many columns are in here at once.
3761    fn close_dictionary(
3762        &self,
3763        _index: usize,
3764        dictionary: &GlobalDictionary,
3765    ) -> Result<ClosedDictionary> {
3766        let (order, flat, bases) = dictionary.ranked_with_values(Some(&*self.file))?;
3767        // A code nothing counted is a code no non-null row of this column holds, which is the
3768        // empty string a null was written as and nothing else, because a code is only ever made by
3769        // a row asking for one. A demoted dictionary counted the stripes before its demotion and
3770        // none after, so it has no count or frequency of the column to give.
3771        let (distinct, frequencies, texts) = if dictionary.demoted {
3772            (None, None, Vec::new())
3773        } else {
3774            let distinct = dictionary.counts.iter().filter(|count| **count != 0).count() as u64;
3775            let (frequencies, texts) = code_frequency(dictionary, &flat, &bases)?;
3776            (Some(distinct), Some(frequencies), texts)
3777        };
3778        // Deriving a fixed SQL host expression at load time materializes its answer.
3779        let hosts = None;
3780        drop(flat);
3781        drop(bases);
3782        let encoded = encode_global_dictionary(dictionary, &order, &dictionary.placed, true)?;
3783        let payload = dictionary
3784            .placed
3785            .iter()
3786            .try_fold(0_u64, |sum, place| sum.checked_add(place.length))
3787            .ok_or_else(|| invalid("global dictionary payload overflow"))?;
3788        Ok(ClosedDictionary { distinct, frequencies, texts, hosts, encoded, payload })
3789    }
3790
3791    /// Writes the statistics sections for the table being closed, as far as the budget reaches.
3792    ///
3793    /// Called from [`Self::close`] after the last stripe and after the dictionaries, which is the
3794    /// first moment the table's column bytes are final and the last moment before the directory is
3795    /// encoded. Both halves matter: the budget is a share of the column bytes, and a section that
3796    /// went in after the directory would be a section the directory does not name.
3797    ///
3798    /// Nothing here can fail the write. A column whose gather came back blind gets no sections, a
3799    /// column the budget could not reach gets none, and section 3.1 says both of those plan the way
3800    /// they planned before statistics existed. The two errors that are returned are an encode
3801    /// failure and a section count past the bound, and neither is a thing a column can cause.
3802    fn write_stats(&mut self) -> Result<()> {
3803        let gathers = std::mem::take(&mut self.gathers);
3804        let rows = self.table.rows as u64;
3805        let mut payloads = Vec::new();
3806        for (column, gather) in gathers.into_iter().enumerate() {
3807            let Some(gather) = gather else { continue };
3808            // A gather that saw a different number of rows than the table committed is a gather
3809            // that missed some, and a distinct count over some of a column is the one error an
3810            // estimator cannot see coming. This has no way of happening today, since a table is
3811            // written once and every chunk goes through `flush_pending`, and that is exactly why it
3812            // is worth a line: it stays true only while that stays true.
3813            if gather.rows() != rows {
3814                continue;
3815            }
3816            let Some(stats) = gather.finish() else { continue };
3817            let mut summary = Vec::new();
3818            stats.summary.encode(&mut summary)?;
3819            let mut sketches = Vec::new();
3820            stats.sketches.encode(&mut sketches)?;
3821            payloads.push((column, summary, sketches));
3822        }
3823        if payloads.is_empty() {
3824            return Ok(());
3825        }
3826        let summaries = payloads.iter().map(|(_, summary, _)| summary.len()).collect::<Vec<_>>();
3827        let sketches = payloads.iter().map(|(_, _, sketches)| sketches.len()).collect::<Vec<_>>();
3828        let allowance = stats::allowance(stats::column_bytes(&self.table), stats::BUDGET_SHARE);
3829        // Nothing is spent yet. A table this writer is closing is one it wrote from nothing, so the
3830        // only statistics sections it can have are the ones about to go in.
3831        let keep = stats::kept(&summaries, &sketches, allowance, 0);
3832        for ((column, summary, sketches), &(built, sketched)) in payloads.iter().zip(&keep) {
3833            if !built {
3834                continue;
3835            }
3836            let id = u64::try_from(*column).map_err(|_| invalid("column index overflow"))?;
3837            let sections = [
3838                // A summary is a header the whole way down: there is nothing behind it a reader
3839                // could decide not to read.
3840                (*section::SUMMARY, summary, summary.len() as u32),
3841                (*section::SKETCHES, sketches, rudb_stats::sketches::HEADER_BYTES),
3842            ];
3843            let wanted = 1 + usize::from(sketched);
3844            for (kind, bytes, header_bytes) in sections.into_iter().take(wanted) {
3845                let written = write_section(
3846                    &*self.file,
3847                    &mut self.at,
3848                    &section::Attachment { kind, id, flags: 0, header_bytes, bytes },
3849                    self.generation,
3850                )?;
3851                self.table.sections.push(written);
3852            }
3853        }
3854        if self.table.sections.len() > MAX_SECTIONS {
3855            return Err(invalid("the table would name more sections than the bound allows"));
3856        }
3857        Ok(())
3858    }
3859
3860    /// Commits every table this writer has written and syncs the file before publishing its header
3861    /// slot.
3862    ///
3863    /// The table handed back is the one the writer was on, which is the last of them. Callers that
3864    /// wrote several already know the others, since they named them.
3865    ///
3866    /// # Errors
3867    ///
3868    /// If directory encoding, writing, or syncing fails.
3869    pub fn finish(mut self) -> Result<Table> {
3870        let entry = self.close()?;
3871        let profile = self.profile.take();
3872        let _timing = profile.as_deref().map(|profile| profile.span(Stage::Publish));
3873        let mut tables = std::mem::take(&mut self.closed);
3874        tables.push(entry);
3875        let catalog = encode_catalog(&tables, &self.views, self.card.as_ref())?;
3876        if catalog.len() > MAX_DIRECTORY {
3877            return Err(invalid("catalog exceeds the configured bound"));
3878        }
3879        let offset = self.at;
3880        self.put(&catalog)?;
3881        if let Some(profile) = &profile {
3882            profile.moved(Stage::Publish, 0, catalog.len() as u64, 0);
3883        }
3884        // Every page and every table directory is on the disk before anything points at them. The
3885        // slot write below is what makes this generation the one a reader picks, so the order of
3886        // these two syncs is the whole of the commit.
3887        synced(&*self.file, profile.as_deref())?;
3888        let slot = Slot {
3889            offset,
3890            length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
3891            generation: self.generation,
3892            hash: checksum(&catalog),
3893        };
3894        // The one write that is not an append, and the last one. It goes back over the slot in the
3895        // header, so it names its offset rather than going through `put`, and `at` does not move.
3896        // Which of the two slots it is alternates with the generation, so the one naming the
3897        // generation before this is still intact and still valid until this write lands.
3898        self.file.write_at(slot_offset(self.generation), &slot.bytes())?;
3899        synced(&*self.file, profile.as_deref())?;
3900        Ok(self.table)
3901    }
3902
3903    /// Commits a generation that changes the views and leaves every table exactly where it is.
3904    ///
3905    /// There was no way to do this before views existed, because everything that could change the
3906    /// catalog also wrote a table, so the only way to say something new about a file was to go
3907    /// through a table. A view is the first thing that can change on its own. Without this, adding
3908    /// a view to a database with eight tables in it would rewrite all eight, since the append path
3909    /// needs a table to append and the fallback is the whole file.
3910    ///
3911    /// It is the same commit as [`Writer::finish`] with nothing appended before it. The table
3912    /// entries are carried forward by directory pointer the way an append carries them, the new
3913    /// catalog goes on the end, and the slot write at the end is what publishes it.
3914    ///
3915    /// # Errors
3916    ///
3917    /// If the file has no valid committed directory, is not this build's format, or cannot be
3918    /// written.
3919    pub fn restate(path: impl AsRef<Path>, views: &[ViewEntry]) -> Result<()> {
3920        let file = RealFilesystem::new().open(path.as_ref(), OpenMode::ReadWrite)?;
3921        let size = file.len()?;
3922        let (slot, bytes, _) = committed_slot(&*file, size)?;
3923        let (closed, _, card) = decode_catalog(&bytes, size)?;
3924        let generation = slot
3925            .generation
3926            .checked_add(1)
3927            .ok_or_else(|| invalid("native file generation overflow"))?;
3928        let catalog = encode_catalog(&closed, views, card_for(path.as_ref(), card).as_ref())?;
3929        if catalog.len() > MAX_DIRECTORY {
3930            return Err(invalid("catalog exceeds the configured bound"));
3931        }
3932        file.write_at(size, &catalog)?;
3933        file.sync()?;
3934        let slot = Slot {
3935            offset: size,
3936            length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
3937            generation,
3938            hash: checksum(&catalog),
3939        };
3940        file.write_at(slot_offset(generation), &slot.bytes())?;
3941        file.sync()?;
3942        Ok(())
3943    }
3944
3945    /// Commits a generation that writes down the device card this process has for the device the
3946    /// file is on, and changes nothing else. It writes nothing when the file already holds that
3947    /// card or the process has none.
3948    ///
3949    /// This is what `PRAGMA device_card_refresh` calls after it measures. Any other commit writes
3950    /// the card too, but a refresh that changes nothing else has no commit to ride on.
3951    ///
3952    /// # Errors
3953    ///
3954    /// The same as [`Writer::restate`].
3955    pub fn keep_device_card(path: impl AsRef<Path>) -> Result<()> {
3956        let path = path.as_ref();
3957        let (_, size, _, bytes, _) = slot_bytes(path)?;
3958        let (_, views, held) = decode_catalog(&bytes, size)?;
3959        if card_for(path, held.clone()) == held {
3960            return Ok(());
3961        }
3962        Self::restate(path, &views)
3963    }
3964
3965    /// Adds exact count, sum, distinct, bound, and bounded frequency certificates to an older file without
3966    /// rewriting table pages. The old slot remains readable until the new catalog is synced.
3967    pub fn certify_summaries(path: impl AsRef<Path>) -> Result<()> {
3968        let path = path.as_ref();
3969        let (_, size, slot, bytes, _) = slot_bytes(path)?;
3970        let (mut entries, views, card) = decode_catalog(&bytes, size)?;
3971        let native = Catalog::open(path)?;
3972        for entry in &mut entries {
3973            let reader = native.table(&entry.name)?;
3974            entry.nonzero.fill(None);
3975            entry.aggregates = reader_aggregate_sums(&reader)?;
3976            entry.distincts = (0..entry.fields.len())
3977                .map(|column| reader.distinct_values(column))
3978                .collect::<Result<Vec<_>>>()?;
3979            entry.extremes = reader_integer_extremes(&reader)?;
3980            entry.frequencies = reader_complete_numeric_frequencies(&reader)?;
3981        }
3982        let generation = slot
3983            .generation
3984            .checked_add(1)
3985            .ok_or_else(|| invalid("native file generation overflow"))?;
3986        let catalog = encode_catalog(&entries, &views, card_for(path, card).as_ref())?;
3987        if catalog.len() > MAX_DIRECTORY {
3988            return Err(invalid("catalog exceeds the configured bound"));
3989        }
3990        let file = RealFilesystem::new().open(path, OpenMode::ReadWrite)?;
3991        file.write_at(size, &catalog)?;
3992        file.sync()?;
3993        let slot = Slot {
3994            offset: size,
3995            length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
3996            generation,
3997            hash: checksum(&catalog),
3998        };
3999        file.write_at(slot_offset(generation), &slot.bytes())?;
4000        file.sync()?;
4001        Ok(())
4002    }
4003
4004    /// The earlier name for [`Self::certify_summaries`].
4005    pub fn certify_counts(path: impl AsRef<Path>) -> Result<()> {
4006        Self::certify_summaries(path)
4007    }
4008}
4009
4010/// Appends one run of bytes at `at` and moves it past them, answering where they went.
4011///
4012/// The append half of [`attach`], which cannot use [`Writer::put`] because it is not writing a
4013/// table. Every byte a section costs goes through here, so the offsets in an extent table come
4014/// from one place.
4015fn append(file: &dyn rudb_io::File, at: &mut u64, bytes: &[u8]) -> Result<u64> {
4016    let offset = *at;
4017    file.write_at(offset, bytes)?;
4018    *at =
4019        at.checked_add(bytes.len() as u64).ok_or_else(|| invalid("native file length overflow"))?;
4020    Ok(offset)
4021}
4022
4023/// Writes one attachment's payload as extents and returns the entry that names it.
4024///
4025/// The split is by bytes, and the `first` of each extent is therefore a byte count. A section kind
4026/// whose extents should break on a row boundary instead will want to hand its extents over already
4027/// split; nothing needs that yet, and guessing at the shape of it now would be guessing.
4028fn write_section(
4029    file: &dyn rudb_io::File,
4030    at: &mut u64,
4031    one: &section::Attachment<'_>,
4032    generation: u64,
4033) -> Result<Section> {
4034    // A payload of nothing is the exception, and it is not a special case so much as a different
4035    // reading of the same field: an entry with no bytes has no header to be longer than them, and
4036    // `header_bytes` is what the structure would have cost. See `Section::refused`.
4037    if !one.bytes.is_empty() && one.header_bytes as usize > one.bytes.len() {
4038        return Err(invalid("a section's header is longer than its payload"));
4039    }
4040    let mut extents = Vec::new();
4041    let mut first = 0_u64;
4042    let extent_size =
4043        if one.kind == *section::RUN_PROJECTION && one.flags == run_projection::RLE_PAGES {
4044            run_projection::RLE_PAGE_BYTES
4045        } else if one.kind == *section::SORTED_PROJECTION || one.kind == *section::RUN_PROJECTION {
4046            1 << 19
4047        } else {
4048            section::MAX_EXTENT as usize
4049        };
4050    for chunk in one.bytes.chunks(extent_size) {
4051        let offset = append(file, at, chunk)?;
4052        extents.push(section::Extent {
4053            offset,
4054            length: u32::try_from(chunk.len()).map_err(|_| invalid("extent length overflow"))?,
4055            hash: checksum(chunk),
4056            first,
4057        });
4058        first += chunk.len() as u64;
4059    }
4060    let mut table = Vec::with_capacity(extents.len() * section::EXTENT_BYTES);
4061    section::encode_extents(&extents, &mut table)?;
4062    // A payload of nothing is a section of no extents and no extent table, and its `extent_page`
4063    // is zero rather than the end of the file. Section 3.7 wants that entry to exist: it is how a
4064    // relationship that did not fit the budget is recorded as not built rather than forgotten.
4065    let extent_page = if table.is_empty() { 0 } else { append(file, at, &table)? };
4066    Ok(Section {
4067        kind: one.kind,
4068        id: one.id,
4069        generation,
4070        extents: u32::try_from(extents.len()).map_err(|_| invalid("too many extents"))?,
4071        extent_page,
4072        extent_bytes: u32::try_from(table.len()).map_err(|_| invalid("extent table overflow"))?,
4073        hash: checksum(&table),
4074        flags: one.flags,
4075        header_bytes: one.header_bytes,
4076    })
4077}
4078
4079/// Attaches graph sections to a table already committed in a file, without rewriting a page.
4080///
4081/// This is the second pass spec/graph/03-the-file-format.md section 3.8 asks for. A key map has to
4082/// exist before the link that uses it can be built, and it is built by reading the key column back,
4083/// so the structures of a table cannot be written during the load that wrote the table. They are
4084/// written afterwards, by this, and the file in between the two is a correct file that answers
4085/// every query more slowly.
4086///
4087/// Nothing is overwritten. The payloads, the extent tables, the new directory for this table and
4088/// the new catalog all go on the end of the file past the committed generation, and the last write
4089/// is the header slot, exactly as [`Writer::finish`] does it. So a crash anywhere in here leaves
4090/// the generation before it intact, and leaves unreferenced trailing bytes that the next commit
4091/// writes past.
4092///
4093/// An attachment replaces any section of the same kind and id, and every other section is carried
4094/// through untouched, including one whose kind this build does not know. The table's own generation
4095/// is carried through too, because attaching a section moves no row: see [`Table::generation`].
4096///
4097/// # Errors
4098///
4099/// If the file has no valid committed directory, is an older format than this build writes, holds
4100/// no table of that name, names a section whose payload cannot be written, or would end up naming
4101/// more sections than the format allows.
4102pub fn attach(
4103    path: impl AsRef<Path>,
4104    table: &str,
4105    attachments: &[section::Attachment<'_>],
4106) -> Result<Table> {
4107    let file = RealFilesystem::new().open(path.as_ref(), OpenMode::ReadWrite)?;
4108    let file = &*file;
4109    let size = file.len()?;
4110    let (slot, bytes, _) = committed_slot(file, size)?;
4111    let (mut entries, views, card) = decode_catalog(&bytes, size)?;
4112    let card = card_for(path.as_ref(), card);
4113    let at = entries
4114        .iter()
4115        .position(|entry| entry.name == table)
4116        .ok_or_else(|| invalid(&format!("the file holds no table called {table}")))?;
4117    let mut version = [0; 4];
4118    read_at(file, 8, &mut version)?;
4119    let version = u32::from_le_bytes(version);
4120    // Readable is not the same as writable. A format 22 file has no section table, and giving its
4121    // directory one without moving the number in its header would leave a file that claims to be
4122    // format 22 and is not, which is worse than refusing. Rewriting it with this build is the
4123    // answer, and the format is at 0.3.x, so nobody has one of these that this project did not
4124    // just make.
4125    if version != FORMAT {
4126        return Err(invalid(&format!(
4127            "the file is format {version} and a graph section needs format {FORMAT}, so it has \
4128             to be written again"
4129        )));
4130    }
4131    let mut directory = vec![0; entries[at].directory.length as usize];
4132    read_at(file, entries[at].directory.offset, &mut directory)?;
4133    if checksum(&directory) != entries[at].directory.hash {
4134        return Err(invalid(&format!("the directory of table {table} does not checksum")));
4135    }
4136    let mut held = decode_directory(&directory, size)?;
4137    let mut cursor = size;
4138    for one in attachments {
4139        let written = write_section(file, &mut cursor, one, held.generation)?;
4140        held.sections.retain(|old| !(old.kind == one.kind && old.id == one.id));
4141        held.sections.push(written);
4142    }
4143    if held.sections.len() > MAX_SECTIONS {
4144        return Err(invalid("the table would name more sections than the bound allows"));
4145    }
4146    let encoded = encode_directory(&held)?;
4147    if encoded.len() > MAX_DIRECTORY {
4148        return Err(invalid("directory exceeds the configured bound"));
4149    }
4150    let offset = append(file, &mut cursor, &encoded)?;
4151    entries[at].directory = Page {
4152        offset,
4153        length: u32::try_from(encoded.len()).map_err(|_| invalid("directory length overflow"))?,
4154        hash: checksum(&encoded),
4155    };
4156    // The views the file already had, written back unchanged. Attaching a section to a table says
4157    // nothing about a view and must not drop one.
4158    let catalog = encode_catalog(&entries, &views, card.as_ref())?;
4159    if catalog.len() > MAX_DIRECTORY {
4160        return Err(invalid("catalog exceeds the configured bound"));
4161    }
4162    let offset = append(file, &mut cursor, &catalog)?;
4163    file.sync()?;
4164    let generation =
4165        slot.generation.checked_add(1).ok_or_else(|| invalid("native file generation overflow"))?;
4166    let committed = Slot {
4167        offset,
4168        length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
4169        generation,
4170        hash: checksum(&catalog),
4171    };
4172    file.write_at(slot_offset(generation), &committed.bytes())?;
4173    file.sync()?;
4174    Ok(held)
4175}
4176
4177/// One column's frequency synopsis as values with their row counts, shared by every clone of a
4178/// reader.
4179type Synopsis = Arc<Vec<(Value, u64)>>;
4180
4181/// Reads committed native column pages without holding the table in memory.
4182#[derive(Debug, Clone)]
4183pub struct Reader {
4184    file: Arc<File>,
4185    table: Arc<Table>,
4186    dictionaries: Arc<Vec<OnceLock<Arc<Vector>>>>,
4187    /// Held while a global dictionary is being opened, one per column.
4188    ///
4189    /// The [`OnceLock`] above says whether one has been opened, which is the question a reader that
4190    /// already has it needs answered and is free. It does not say whether one is being opened, and
4191    /// the difference matters because every worker of a scan wants the same dictionary at the same
4192    /// moment. Without this they all miss, all read the page, all verify it and all decode it, and
4193    /// all but one throw the answer away. ClickBench 38 reads the URL dictionary, which is 515,958
4194    /// entries, and was paying for it twice.
4195    loading: Arc<Vec<Mutex<()>>>,
4196    /// Each column's frequency synopsis as values, the first time anything asks for it. See
4197    /// [`Reader::decode_frequencies`].
4198    frequency_values: Arc<Vec<OnceLock<Synopsis>>>,
4199    /// Stored frequency sections are decoded once per open table. A small directory can hold the
4200    /// summary inline, but a larger one otherwise rereads and decodes the same section on every
4201    /// plan and every summary-backed aggregate.
4202    frequency_summaries: Arc<Vec<OnceLock<Arc<FrequencySummary>>>>,
4203    /// Each column's summary, the first time anything asks for it. See `stats::held_summary`.
4204    summaries: Arc<Vec<OnceLock<Option<Arc<rudb_stats::Summary>>>>>,
4205    /// How many global dictionaries have been opened. A scan of a dictionary column should open its
4206    /// dictionary once however many workers it has, and the test that says so is the only thing
4207    /// keeping it that way.
4208    opened: Arc<AtomicUsize>,
4209    /// The membership sieves of one stripe of one column, by column and then by stripe, read the
4210    /// first time a probe asks about them. A query filters on one or two columns and never looks at
4211    /// the rest, so reading these at open would be the whole index for the sake of a fraction of it.
4212    sieves: Arc<Vec<Vec<SieveSlot>>>,
4213    /// The per part ranges of one stripe of one column, by column and then by stripe, read the
4214    /// first time something compares that column and kept after that.
4215    part_ranges: Arc<Vec<Vec<RangeSlot>>>,
4216    /// Which stripe and which part of it every part of the table is, by table wide part number.
4217    places: Arc<Vec<Place>>,
4218    cache: Arc<Shelf>,
4219    /// Where the pages above are counted against the database's budget. See [`PagePool`].
4220    pool: PagePool,
4221    /// How many whole stripe pages have been read, which is what the sharing above is judged on. A
4222    /// scan of a column should read each of its stripes once however many workers it has.
4223    pages: Arc<AtomicUsize>,
4224    /// How many index sections have been read. A scan of a column should read each of its stripes
4225    /// once here too, and the test that says so is the only thing keeping it that way.
4226    indexes: Arc<AtomicUsize>,
4227    /// Which parts of which columns have matched their checksums, a bit per part of the table for
4228    /// each column in turn.
4229    ///
4230    /// A part is written once and a later generation writes its parts somewhere else, so bytes
4231    /// that matched once match for as long as this reader is open. The page cache keeps the same
4232    /// promise for as long as it holds a page, and this one outlives the page. A scan the graph
4233    /// layer reduces reads a part at the rows it keeps and not the stripe's page, and each of those
4234    /// reads hashed the whole part again: on TPC-H q21, which reads `lineitem` three times, that was
4235    /// 4 percent of the query.
4236    verified: Arc<Vec<AtomicU64>>,
4237    /// Each text column's [`grams`] sketch in row id order, read the first time a `LIKE` asks
4238    /// about the column, and `None` when the table carries none for it.
4239    text_grams: Arc<Vec<OnceLock<Option<Vec<u64>>>>>,
4240    /// The row id of every part's first row, by table wide part number.
4241    firsts: Arc<Vec<usize>>,
4242    /// The file's size when it was opened, for [`Reader::layout`].
4243    size: u64,
4244    /// The committed directory's size, for [`Reader::layout`].
4245    directory: u64,
4246    /// What opening the file cost, which is a number rather than a claim.
4247    opening: Opening,
4248}
4249
4250/// What [`Reader::open`] read before it returned.
4251///
4252/// `spec/stats/04-in-memory.md` section 4.2 says opening a table reads the header and the directory
4253/// and nothing else, and once that document's statistics are in the file the tempting change is to
4254/// load a column summary or two on the way past, because they are small and the next query will
4255/// want them. A hundred milliseconds of that is a hundred milliseconds nobody asked for, and an
4256/// embedded database is opened by processes that are about to run one trivial query.
4257///
4258/// So the claim gets a number. Both of these are fixed by the schema and the stripe count and are
4259/// independent of how many rows the file holds, and the test that says so is what stops the
4260/// tempting change from landing quietly.
4261#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
4262pub struct Opening {
4263    /// How many times the file was read. The header, then each directory slot that looked valid
4264    /// enough to check, so three at the most.
4265    pub reads: u32,
4266    /// How many bytes those reads asked for.
4267    pub bytes: u64,
4268}
4269
4270/// What a reader has read, while it was being opened and since.
4271#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
4272pub struct Reads {
4273    /// What opening cost, before any query had been planned.
4274    pub opening: Opening,
4275    /// Whole stripe pages read since.
4276    pub pages: usize,
4277    /// Index sections read since.
4278    pub indexes: usize,
4279    /// Global dictionaries opened since. One per dictionary column that a query touched, however
4280    /// many workers touched it, which is a claim only a test can keep true.
4281    pub dictionaries: usize,
4282}
4283
4284/// Where one table wide part number lands.
4285#[derive(Debug, Clone, Copy)]
4286struct Place {
4287    stripe: u32,
4288    part: u32,
4289    rows: u32,
4290}
4291
4292/// One part's bytes inside one column page.
4293#[derive(Debug, Clone, Copy)]
4294struct PartSpan {
4295    start: usize,
4296    length: usize,
4297    hash: u64,
4298}
4299
4300/// What a reader holds for one stripe of one column.
4301///
4302/// The index is small and is loaded whether the caller wants the whole page or one part of it. The
4303/// page is loaded only by a scan, because a sparse fetch that wants a thousand rows out of sixty
4304/// four thousand would be reading sixty four times what it uses.
4305#[derive(Debug, Clone)]
4306struct CachedColumn {
4307    stripe: usize,
4308    index: Arc<Vec<PartSpan>>,
4309    page: Option<Arc<HeldPage>>,
4310}
4311
4312/// One stripe's page of one column, with which of its parts have already matched their checksums.
4313///
4314/// The bytes never change once they are read, so a part that matched once matches for as long as
4315/// the page is held. Hashing it again on every read was 3.5% of a `GROUP BY CounterID` over the
4316/// held pages of the ClickBench sample, run seventy times in one process. A part read without its
4317/// page is still checked every time, since those bytes come fresh off the file.
4318#[derive(Debug)]
4319struct HeldPage {
4320    bytes: Vec<u8>,
4321    checked: Vec<AtomicBool>,
4322}
4323
4324impl HeldPage {
4325    /// The bytes of part `part`, checked against `span` the first time anyone asks for them.
4326    fn part(&self, part: usize, span: PartSpan) -> Result<&[u8]> {
4327        let bytes = part_bytes(&self.bytes, span)?;
4328        let checked = self.checked.get(part).ok_or_else(|| invalid("part index out of range"))?;
4329        if !checked.load(Atomic::Relaxed) {
4330            verify_part(bytes, span)?;
4331            checked.store(true, Atomic::Relaxed);
4332        }
4333        Ok(bytes)
4334    }
4335}
4336
4337/// Checks one part's bytes against the hash its index carries for them.
4338fn verify_part(bytes: &[u8], span: PartSpan) -> Result<()> {
4339    let got = checksum(bytes);
4340    if got != span.hash {
4341        return Err(invalid(&format!(
4342            "column page checksum differs, part at {}+{} bytes, wanted {:016x} and got {got:016x}",
4343            span.start, span.length, span.hash,
4344        )));
4345    }
4346    Ok(())
4347}
4348
4349/// One column's stripes a reader holds, and which of them somebody is reading right now.
4350///
4351/// The pages are one slot per stripe of the table rather than a list of the ones being kept, so
4352/// finding a page is an index and not a walk. That matters because the walk happened under the
4353/// lock, once per part per column, and a scan that gives a whole stripe to each of thirty two
4354/// workers keeps enough pages that walking them was the longest thing the lock was held for. The
4355/// slots cost a pointer per stripe per column, which on the ClickBench file is eight kilobytes
4356/// against the forty megabytes of pages they point at. `order` is which of them are filled, oldest
4357/// first, because that is the one thing the slots cannot say by themselves.
4358///
4359/// `loading` is what keeps a scan from reading the same page once per worker. It is a list and not
4360/// a set because it holds at most one stripe per worker on the column and is walked far less often
4361/// than a hash of it would be built.
4362///
4363/// `index` is every index this reader has ever read for the column, one slot per stripe, and it is
4364/// never evicted. An index is a few hundred bytes and a page is a quarter of a megabyte, so the two
4365/// do not belong under the same budget. Riding in the page cache meant a worker that came back to a
4366/// stripe after its page had been evicted read the index again with it, which on the full
4367/// ClickBench file was about thirteen hundred reads out of a hundred and fourteen thousand.
4368///
4369/// `touched` is which parts of each stripe have been read, a bit a part. A part is read on its own
4370/// the first time and the stripe's page is read whole only when one of its parts is asked for again.
4371/// That is the rule [`NativeText`] follows for its decoded blocks: a page earns its memory by being
4372/// wanted a second time. A process that runs one statement, which is how a script or a benchmark
4373/// uses the engine, wants each part once, and holding whole pages for it is what its peak was made
4374/// of. On ClickBench 32 the pages of `WatchID` and `ClientIP`, each two megabytes a stripe, were half
4375/// of the 108 MB the query peaked at, and ClickBench 41 held a stripe of every filter column to use
4376/// a handful of parts out of each. Read a part at a time a scan costs more calls to read the same
4377/// bytes, which on the whole suite was lost in the noise.
4378///
4379/// A page read whole goes into the pool when every part of its stripe had been read before, which
4380/// is a second scan. When only some had, it is one scan asking for a part twice, the way a `LIKE`
4381/// asks a compressed text part whether it can answer and then reads it, and `passing` holds those
4382/// pages, oldest first, down to the column's floor. That is what keeps ClickBench 21 from pooling
4383/// every page of `URL` for a second scan that never comes.
4384#[derive(Debug, Default)]
4385struct Cached {
4386    pages: Vec<Option<Resident>>,
4387    loading: Vec<usize>,
4388    index: Vec<Option<Arc<Vec<PartSpan>>>>,
4389    touched: Vec<Vec<u64>>,
4390    passing: VecDeque<usize>,
4391}
4392
4393/// One page a reader holds, and whether anyone has read it since the pool last looked.
4394#[derive(Debug, Clone)]
4395struct Resident {
4396    page: Arc<HeldPage>,
4397    used: Arc<AtomicBool>,
4398}
4399
4400/// Every column's pages of one reader, with how many each column holds and the floor under that.
4401#[derive(Debug)]
4402struct Shelf {
4403    columns: Vec<Mutex<Cached>>,
4404    /// How many pages each column holds right now. Counted outside the column locks so that the
4405    /// pool can tell whether a column is at its floor without taking a lock it might be under.
4406    held: Vec<AtomicUsize>,
4407    /// How many stripes of one column are kept whatever the budget says. See
4408    /// [`CACHED_STRIPES_PER_COLUMN`] for what sets it and [`Reader::keep_stripes`] for who raises it.
4409    kept: AtomicUsize,
4410}
4411
4412/// The pages every reader of one database keeps, under one budget in bytes.
4413///
4414/// A reader lives as long as the database does, so the pages it holds are what the next query finds
4415/// already in memory. They used to be four stripes a column, oldest out first, which on TPC-H SF1
4416/// meant every query read every page of lineitem off the file again and paid the system call for
4417/// it. Keeping every page there costs 38 MB and took a third of the system time off the suite.
4418///
4419/// So the question is no longer how many stripes a column keeps but how many bytes the database
4420/// does, and one budget answers it for every reader at once. A table nobody queries gives its pages
4421/// up to one that is being queried, which a count per column cannot do.
4422///
4423/// Pages leave by the clock. Each has a bit a read sets, and when the pool is over budget it walks
4424/// from the oldest: a page with the bit set loses the bit and goes round again, and a page without
4425/// it goes. That keeps what is read over and over and lets a page one scan read once go first.
4426///
4427/// The old count is still a floor. A column never gives up a page while it holds four or fewer,
4428/// because a scan whose workers evict each other's pages reads a quarter of a megabyte for every
4429/// part it takes, and a budget of zero is the cache as it was before the pool existed.
4430#[derive(Debug, Clone, Default)]
4431pub struct PagePool {
4432    ring: Arc<Mutex<Ring>>,
4433    budget: Arc<AtomicUsize>,
4434}
4435
4436#[derive(Debug, Default)]
4437struct Ring {
4438    held: VecDeque<Held>,
4439    bytes: usize,
4440}
4441
4442/// One page in the pool, pointing back at the reader that holds it.
4443///
4444/// Weak, because a reader that has gone, which every reader does at a checkpoint, should take its
4445/// pages with it and not have them kept alive by the pool.
4446#[derive(Debug)]
4447struct Held {
4448    shelf: Weak<Shelf>,
4449    column: usize,
4450    stripe: usize,
4451    bytes: usize,
4452    used: Arc<AtomicBool>,
4453}
4454
4455impl PagePool {
4456    /// A pool that keeps up to `budget` bytes of pages beyond each column's floor.
4457    #[must_use]
4458    pub fn new(budget: usize) -> Self {
4459        let pool = Self::default();
4460        pool.budget.store(budget, Atomic::Relaxed);
4461        pool
4462    }
4463
4464    /// The bytes of pages the pool is counting now.
4465    ///
4466    /// # Panics
4467    ///
4468    /// If the pool's lock is poisoned, which takes a panic while it was held.
4469    #[must_use]
4470    pub fn bytes(&self) -> usize {
4471        self.ring.lock().map_or(0, |ring| ring.bytes)
4472    }
4473
4474    /// Counts a page a reader has just taken in, and lets pages go until the pool is back under its
4475    /// budget or it has looked at every page once.
4476    ///
4477    /// Called with no column lock held. The pages that go are chosen under the pool's lock and
4478    /// dropped under their column's lock afterwards, so no thread ever holds both.
4479    fn admit(&self, held: Held) {
4480        let budget = self.budget.load(Atomic::Relaxed);
4481        let mut gone = Vec::new();
4482        {
4483            let Ok(mut ring) = self.ring.lock() else { return };
4484            ring.bytes += held.bytes;
4485            ring.held.push_back(held);
4486            // One lap and no more. A page read since the last pass loses its bit on this one and
4487            // can only go on a later one, which is the second chance the clock is named for.
4488            let mut looked = 0;
4489            let limit = ring.held.len();
4490            while ring.bytes > budget && looked < limit {
4491                looked += 1;
4492                let Some(entry) = ring.held.pop_front() else { break };
4493                let Some(shelf) = entry.shelf.upgrade() else {
4494                    ring.bytes -= entry.bytes;
4495                    continue;
4496                };
4497                if entry.used.swap(false, Atomic::Relaxed) {
4498                    ring.held.push_back(entry);
4499                    continue;
4500                }
4501                let count = &shelf.held[entry.column];
4502                if count.load(Atomic::Relaxed) <= shelf.kept.load(Atomic::Relaxed).max(1) {
4503                    ring.held.push_back(entry);
4504                    continue;
4505                }
4506                count.fetch_sub(1, Atomic::Relaxed);
4507                ring.bytes -= entry.bytes;
4508                gone.push((shelf, entry));
4509            }
4510            // A reader that has gone leaves its entries behind, and with a budget nobody reaches
4511            // they would pile up one checkpoint after another. The front is where the oldest are.
4512            while ring.held.front().is_some_and(|entry| entry.shelf.strong_count() == 0) {
4513                if let Some(entry) = ring.held.pop_front() {
4514                    ring.bytes -= entry.bytes;
4515                }
4516            }
4517        }
4518        for (shelf, entry) in gone {
4519            let Ok(mut cached) = shelf.columns[entry.column].lock() else { continue };
4520            if let Some(slot) = cached.pages.get_mut(entry.stripe)
4521                && slot.as_ref().is_some_and(|slot| Arc::ptr_eq(&slot.used, &entry.used))
4522            {
4523                *slot = None;
4524            }
4525        }
4526    }
4527}
4528
4529/// Stripes of one column a reader keeps the bytes of, when nobody has asked for more.
4530///
4531/// This has to hold at least as many stripes as a column has workers in it at once, or the workers
4532/// evict each other's pages and read them again. Four is what a scan that hands parts out in order
4533/// needs, because then every worker is within a few parts of every other and at most a couple of
4534/// stripes are open at a time. A scan that hands a whole stripe to each worker has one stripe open
4535/// per worker for the length of that stripe, and it says so with [`Reader::keep_stripes`] rather
4536/// than paying for sixteen slots on every table that is read one part at a time.
4537///
4538/// It multiplies by the page size, which is a quarter of a megabyte for a four byte column, and by
4539/// the number of columns a query touches.
4540const CACHED_STRIPES_PER_COLUMN: usize = 4;
4541
4542/// The sieves of one stripe of one column, once somebody has asked for them.
4543type SieveSlot = OnceLock<Arc<Vec<Option<Sieve>>>>;
4544
4545type RangeSlot = OnceLock<Arc<Vec<Range>>>;
4546
4547#[derive(Debug)]
4548struct NativeText {
4549    file: Arc<File>,
4550    /// How many values the dictionary holds.
4551    values: usize,
4552    /// Where each value ends inside its payload block, packed at `offset_bits` in runs of
4553    /// [`TEXT_OFFSET_RUN`].
4554    ///
4555    /// Ends rather than starts, because then a block of 1,024 values is 1,024 numbers rather than
4556    /// 1,025: the start of a value is the end of the one before it, and the first value of a block
4557    /// starts at zero by construction. Relative to the block rather than to the payload, because a
4558    /// reader decodes a whole block and slices it, so an offset into the payload is a number it
4559    /// would have to subtract a base from anyway.
4560    ///
4561    /// The vector is the index as it was read, so the offsets start after the header, and
4562    /// [`Self::packed`] is where they are read from.
4563    offsets: Vec<u8>,
4564    /// Bits one offset is packed at, which is what the largest block of this column spans and is the
4565    /// same for every block of it.
4566    offset_bits: usize,
4567    /// The same ends unpacked, built once enough readers have asked for one at a time.
4568    ///
4569    /// Reading one offset out of the packed form costs about fifty instructions: a division to find
4570    /// the run, a bounds check to slice it, a shift to reach the bit the value starts at and a
4571    /// narrowing on the way out. That is the right price for a reader that wants a handful. It is
4572    /// the wrong price for `STRLEN` over a column, which asks for one per row and nothing else, and
4573    /// where a million of them was a third of ClickBench 28.
4574    ///
4575    /// With the ends unpacked every read is a load, and a vector of lengths is one loop over them.
4576    /// The table is built only once the reads say it will be used, which is what
4577    /// [`Self::ends_worth_unpacking`] decides and [`Self::ends_asked`] counts towards, because a
4578    /// table built for a reader that wanted three values is four bytes a value spent on nothing.
4579    value_ends: OnceLock<Option<Vec<u32>>>,
4580    /// The length of every value, worked out of [`Self::value_ends`] the first time a vector of
4581    /// lengths is asked for.
4582    ///
4583    /// A length out of the ends is two loads, a test for whether the value opens its block and a
4584    /// check that it does not end before it starts, which came to thirteen instructions a row on
4585    /// ClickBench 28. Out of this it is one load. The order is checked once for the whole table
4586    /// while it is built, and a column that fails it gets no table and goes on reading the ends,
4587    /// which is where the error is reported. Two bytes a value where every value is short enough,
4588    /// four otherwise, and only for a column something has asked the length of a vector at a time.
4589    value_lens: OnceLock<Option<Lengths>>,
4590    /// How many single offset reads have come in while the table is not built.
4591    ///
4592    /// Relaxed, and read only against a threshold, so two threads racing here means the table is
4593    /// built one read early or one read late. Counting stops the moment the table exists, because
4594    /// [`OnceLock::get`] settles it before this is touched.
4595    ends_asked: AtomicUsize,
4596    /// How many entries the sorted order has, which is the value count.
4597    ranks: usize,
4598    /// Where the sorted order starts in the file. It is read a block at a time and only when
4599    /// something searches it, so a query that never compares this column against a literal never
4600    /// touches it at all.
4601    rank_at: u64,
4602    /// Where each block of the sorted order ends, as a byte offset from `rank_at`. A block is packed
4603    /// at whatever width its own heads need, so unlike the entries it replaced its length is not
4604    /// arithmetic on the block number.
4605    rank_ends: Vec<u64>,
4606    rank_hashes: Vec<u64>,
4607    rank_blocks: Vec<OnceLock<Result<Vec<u8>>>>,
4608    /// Bits one code is packed at, which is what the value count needs and is the same for every
4609    /// block of the column.
4610    code_bits: usize,
4611    /// The sorted order turned round, built the first time a reader asks for it.
4612    ///
4613    /// Four bytes per value against the four the offsets already hold, so a column that has this is
4614    /// carrying half again what it carried before rather than something of a new order. It is built
4615    /// only when something asks, which is a grouped min or max over this column and nothing else,
4616    /// and that reader was going to read the payload of this column once per row otherwise.
4617    code_ranks: OnceLock<Option<Vec<u32>>>,
4618    /// Where each block of the payload starts in the file, and how many stored bytes it is.
4619    ///
4620    /// Absolute rather than an offset from a base the blocks share, because a block is written the
4621    /// moment it fills and what comes after it in the file is whatever the load wrote next. A file
4622    /// old enough to have them back to back is read into these same two lists by adding the base to
4623    /// the ends it carries, so nothing below here knows which kind of file it came from.
4624    starts: Vec<u64>,
4625    lengths: Vec<u64>,
4626    hashes: Vec<u64>,
4627    /// Conservative four-byte substring signatures, read only by a compatible LIKE filter.
4628    grams: Option<NativeGrams>,
4629    /// The payload, read and decoded a block at a time and kept after that.
4630    blocks: Vec<OnceLock<Result<Vec<u8>>>>,
4631    /// The length in characters of every value of a block, worked out the first time `length` asks
4632    /// for a value in that block.
4633    ///
4634    /// Kept instead of the block it was counted out of. `length` reads every row of a column, and
4635    /// reading the bytes through [`Self::payload_block`] kept every block it touched, which is every
4636    /// distinct value of the column decoded: seven string columns of ClickBench held 13.9 GB to
4637    /// answer seven `max(length(...))`. The counts are four bytes a value, so the same scan keeps
4638    /// the counts and decodes each block once, the same number of times it did before.
4639    char_lens: Vec<OnceLock<Box<[u32]>>>,
4640    /// How many decoded payload bytes this column keeps before a sweep stops keeping what it reads.
4641    /// [`TEXT_KEEP_BUDGET`] everywhere but in the test of the ceiling.
4642    keep_budget: usize,
4643    /// Roughly how many decoded payload bytes are being kept, which is what [`TEXT_KEEP_BUDGET`]
4644    /// is measured against.
4645    ///
4646    /// Roughly, because two threads that keep the same block at the same time both add its length
4647    /// while [`OnceLock`] keeps one of the two. That makes the count read high and the budget bind
4648    /// a little early, which is the harmless direction, and it costs one relaxed add a block rather
4649    /// than a lock on the path every scan of a string column goes through.
4650    payload_kept: AtomicUsize,
4651    /// Which payload blocks a sweep has decoded before, one flag a block.
4652    ///
4653    /// A sweep keeps a block the second time it decodes it and not the first. A process that runs
4654    /// one statement, which is how a benchmark or a script uses the engine, sweeps each block once
4655    /// and so keeps nothing: on ten million rows a `URL LIKE` held 396 MB with every block kept and
4656    /// 97 MB with none, for the same processor time. A session that asks again pays the decode one
4657    /// more time and reads kept blocks from then on, under the same [`TEXT_KEEP_BUDGET`].
4658    swept: Vec<AtomicBool>,
4659    /// How many blocks [`TextSource::visit_at`] has decoded and dropped because the column was
4660    /// already holding its [`TEXT_KEEP_BUDGET`].
4661    ///
4662    /// A sweep reads the dictionary in order and touches a block once, so dropping what it reads
4663    /// past the budget costs one decode a block and bounds the column. A visit reads a vector of
4664    /// codes, and the codes of a scan land all over the dictionary: on ten million rows of
4665    /// ClickBench each vector of two thousand `URL`s touches about a hundred and forty of its two
4666    /// and a half thousand blocks, and so does the next one. A cache holding a tenth of the column
4667    /// still misses half of those, and dropping every block past the budget would decode the
4668    /// column hundreds of times over to answer one `lower(URL)`. So a visit drops past the budget
4669    /// only until it has dropped as many blocks as the column has, which is what a read whose codes
4670    /// are few or clustered never reaches, and keeps what it reads after that, the way a row at a
4671    /// time read always did. That bounds what a visit can cost over the old read at one more decode
4672    /// of the column.
4673    visit_dropped: AtomicUsize,
4674    /// The boundaries this dictionary has already been searched for, by the value searched for.
4675    ///
4676    /// A search is the expensive thing this type does. It settles a probe on the stored head where
4677    /// it can and reads a value where it cannot, and reading a value decodes the payload block it
4678    /// sits in, so one search can cost several blocks. The thing that makes remembering worth it is
4679    /// that the same search comes back: a top N asks once a chunk whether anything left can beat its
4680    /// worst candidate, and the worst candidate settles long before the chunks run out.
4681    ///
4682    /// Shared across the instances of a scan rather than kept per instance, because each of them has
4683    /// its own worst candidate and all of them are searching the same dictionary. One lock per chunk
4684    /// is nothing next to a probe of a file.
4685    ///
4686    /// Bounded by [`TEXT_SEARCH_MEMO`] and emptied rather than evicted when it is full. What fills
4687    /// it is a top N improving its bound, which happens a few dozen times and then stops, so the
4688    /// bound is there for the filter that searches for a different literal every chunk rather than
4689    /// for anything this is meant to help.
4690    searched: Mutex<HashMap<Vec<u8>, (usize, bool)>>,
4691}
4692
4693#[derive(Debug)]
4694struct NativeGrams {
4695    start: u64,
4696    length: usize,
4697    /// How long one block's signature is.
4698    width: usize,
4699    hash: u64,
4700    /// For each literal asked about lately, whether each block might hold it.
4701    ///
4702    /// The answer for every block at once, worked out by one pass over the signatures a window at a
4703    /// time, rather than the signatures read in and kept. On ClickBench `URL` they are 21 MB for
4704    /// ten million rows and a verdict is 2,650 flags, and a filter asks the same question of every
4705    /// block, so the pass is paid once and what stays resident is the flags.
4706    verdicts: Mutex<Vec<Verdict>>,
4707}
4708
4709/// A literal and whether each block might hold it.
4710type Verdict = (Vec<u8>, Arc<[bool]>);
4711
4712/// How many literals a column remembers the verdicts of.
4713const GRAM_VERDICTS: usize = 8;
4714
4715impl NativeGrams {
4716    /// Whether each block might hold `literal`, remembered or worked out now.
4717    ///
4718    /// The lock is held over the pass so that the threads of one scan, which all ask about the
4719    /// same literal at the start, read the signatures once between them.
4720    fn verdicts(&self, file: &File, literal: &[u8]) -> Result<Arc<[bool]>> {
4721        let mut held = self.verdicts.lock().map_err(|_| invalid("a poisoned signature verdict"))?;
4722        if let Some((_, verdict)) = held.iter().find(|(asked, _)| asked == literal) {
4723            return Ok(Arc::clone(verdict));
4724        }
4725        let wanted = literal.windows(4).map(|gram| gram_bits(gram, self.width)).collect::<Vec<_>>();
4726        let mut verdict = Vec::with_capacity(self.length / self.width);
4727        let window = GRAM_WINDOW / self.width * self.width;
4728        let hash = walk_checksummed(file, self.start, self.length, window, |bytes| {
4729            verdict.extend(bytes.chunks(self.width).map(|bits| {
4730                wanted
4731                    .iter()
4732                    .flatten()
4733                    .all(|&bit| bits.get(bit / 8).is_some_and(|byte| byte & (1 << (bit % 8)) != 0))
4734            }));
4735            Ok(())
4736        })?;
4737        if hash != self.hash {
4738            return Err(invalid("global dictionary substring signatures checksum differs"));
4739        }
4740        let verdict: Arc<[bool]> = verdict.into();
4741        if held.len() >= GRAM_VERDICTS {
4742            held.remove(0);
4743        }
4744        held.push((literal.to_vec(), Arc::clone(&verdict)));
4745        Ok(verdict)
4746    }
4747
4748    fn footprint(&self) -> usize {
4749        self.verdicts.lock().map_or(0, |held| {
4750            held.iter().map(|(asked, verdict)| asked.capacity() + verdict.len()).sum()
4751        })
4752    }
4753}
4754
4755/// How many searched for values a column's dictionary remembers the boundary of.
4756///
4757/// See [`NativeText::searched`]. Small because the case it is for repeats one value, not because a
4758/// larger one would be wrong.
4759const TEXT_SEARCH_MEMO: usize = 64;
4760
4761/// How many values of a dictionary go in one block of the payload.
4762///
4763/// The block is the unit the string cascade encodes, the unit a checksum covers, and the unit a
4764/// reader has to decode to get at a single value, so it is the one number the payload format turns
4765/// on. Blocking by values rather than by bytes is what keeps a value out of two blocks at once: the
4766/// block holding a code is `code / TEXT_PAYLOAD_VALUES` and nothing has to be stitched.
4767///
4768/// A probe on the five ClickBench columns that have a dictionary worth the name, written up on
4769/// #347, measured the ratio and the decode speed at 128, 256, 512, 1,024 and 4,096 values. Both get
4770/// better all the way up, because front coding and the LZ matcher have more to look back at and
4771/// because the per chunk setup is spread over more values. What stops it is the point read: a query
4772/// that wants ten values has to decode ten blocks, so the block is what a lookup costs. At 1,024
4773/// values a block is between 67 KB and 394 KB decoded across those five columns, and the ratios are
4774/// 2.3 to 4.5. Going up to 4,096 buys two to six percent more and makes a block as much as 1.5 MB.
4775/// Going down to 512 gives up five to nine percent.
4776const TEXT_PAYLOAD_VALUES: usize = 1024;
4777
4778/// Eight KiB per payload block, which is what makes a four-byte substring a useful negative test on
4779/// a column of URLs.
4780///
4781/// Two KiB was the first answer and on ClickBench `URL` it proved almost nothing. A block of 1,024
4782/// sorted URLs holds about seventeen thousand distinct four-byte grams, and at two bits each that
4783/// set nine in ten of the sixteen thousand bits there were, so `LIKE '%google%'` passed most blocks
4784/// it had no match in and decoded them. At eight KiB four bits in ten are set, and of the 2,650
4785/// blocks of `URL` in ten million rows a needle that is in none of them passes 36. The signatures
4786/// are not read into memory, see [`NativeGrams::verdicts`], so the width costs file and not
4787/// resident memory.
4788const TEXT_GRAM_BYTES: usize = 8192;
4789
4790/// The signature width of a format 28 file, which is still read.
4791const NARROW_GRAM_BYTES: usize = 2048;
4792
4793/// How much of a column's signatures a verdict reads at a time.
4794const GRAM_WINDOW: usize = 256 << 10;
4795
4796/// A fast mixing step for exactly four bytes, shared by load and query, into a signature of
4797/// `width` bytes.
4798fn gram_bits(bytes: &[u8], width: usize) -> [usize; 2] {
4799    let original = u32::from_le_bytes(bytes.try_into().expect("a four-byte gram"));
4800    let mut first = original ^ (original >> 16);
4801    first = first.wrapping_mul(0x7feb_352d);
4802    first ^= first >> 15;
4803    let mut second = original ^ (original >> 17);
4804    second = second.wrapping_mul(0x846c_a68b);
4805    second ^= second >> 16;
4806    let mask = width * 8 - 1;
4807    [(first as usize) & mask, (second as usize) & mask]
4808}
4809
4810/// How many decoded payload bytes one dictionary keeps before a sweep stops keeping what it reads.
4811///
4812/// A sweep of the whole dictionary decodes every block whatever it does, and the only question is
4813/// whether it hangs on to them. Keeping all of them is 4.2 GB on ClickBench `URL` at a hundred
4814/// million rows, which is what #997 was right to stop. Keeping none of them means the next query
4815/// asking the same thing decodes all of it again, and on the same column at a million rows that
4816/// took a `LIKE` from 2.7 ms to 16.2 ms, because the decode used to be paid once by a session and
4817/// is now paid by every statement in it. Neither end is the answer. A bound is.
4818///
4819/// So a sweep keeps what it decodes until the column is holding this much and decodes without
4820/// keeping after that. At a million rows the five ClickBench string columns decode to between 8 MB
4821/// and 85 MB, so they sit inside it and a repeated `LIKE` reads a decoded block rather than a
4822/// stored one. At a hundred million rows `URL` fills it and the rest of that column is read and
4823/// dropped, which is the old cost on the part that does not fit and none of the old footprint.
4824///
4825/// Two hundred and fifty six megabytes a column is a number and not a policy, and the policy is
4826/// what should replace it: this wants to be a buffer pool over the whole database, sized against
4827/// the memory limit the session was given, with the blocks of every column competing for it and the
4828/// least useful one evicted. That is F2 work. What is here is the part of it that can be written
4829/// without an eviction order, which is a ceiling.
4830const TEXT_KEEP_BUDGET: usize = 256 * 1024 * 1024;
4831
4832/// The length of every value of a column, as narrow as the longest of them allows.
4833///
4834/// The table is read at the codes a vector holds, which on a column the size of ClickBench `URL`
4835/// land all over it, so what a length costs is whether its line is in cache. Half a million URLs
4836/// are two megabytes at four bytes a length and one at two, which is the difference between the
4837/// table sitting in the second level cache or not.
4838#[derive(Debug)]
4839enum Lengths {
4840    /// Every length fits in sixteen bits.
4841    Narrow(Vec<u16>),
4842    /// Some value is longer than that.
4843    Wide(Vec<u32>),
4844}
4845
4846impl Lengths {
4847    /// The lengths at `indices`, appended to `into`, and zero for a position past the end, which
4848    /// is what a row at a time read says.
4849    fn extend_at(&self, indices: &[u32], into: &mut Vec<i64>) {
4850        match self {
4851            Lengths::Narrow(lens) => into.extend(
4852                indices
4853                    .iter()
4854                    .map(|&index| lens.get(index as usize).map_or(0, |&len| i64::from(len))),
4855            ),
4856            Lengths::Wide(lens) => into.extend(
4857                indices
4858                    .iter()
4859                    .map(|&index| lens.get(index as usize).map_or(0, |&len| i64::from(len))),
4860            ),
4861        }
4862    }
4863
4864    /// The bytes the table holds on to.
4865    fn footprint(&self) -> usize {
4866        match self {
4867            Lengths::Narrow(lens) => lens.capacity() * size_of::<u16>(),
4868            Lengths::Wide(lens) => lens.capacity() * size_of::<u32>(),
4869        }
4870    }
4871}
4872
4873/// The length of every value out of where each one ends inside its payload block, or `None` for
4874/// ends that go backwards somewhere inside a block.
4875///
4876/// A value that opens a block starts at zero and every other one starts where the value before it
4877/// ends, so a block is a run of differences.
4878///
4879/// Built at two bytes a length straight away, and built again at four only when some value turns
4880/// out too long for that, which is rare enough that the second pass is not worth avoiding.
4881fn lengths_of(ends: &[u32]) -> Option<Lengths> {
4882    match lengths_as::<u16>(ends)? {
4883        Some(narrow) => Some(Lengths::Narrow(narrow)),
4884        None => lengths_as::<u32>(ends)?.map(Lengths::Wide),
4885    }
4886}
4887
4888/// [`lengths_of`] at one width: `None` for ends that go backwards, and `Some(None)` for a length
4889/// that does not fit in `T`.
4890fn lengths_as<T: TryFrom<u32>>(ends: &[u32]) -> Option<Option<Vec<T>>> {
4891    let mut lens = Vec::with_capacity(ends.len());
4892    for block in ends.chunks(TEXT_PAYLOAD_VALUES) {
4893        let mut start = 0;
4894        for &end in block {
4895            let Ok(len) = T::try_from(end.checked_sub(start)?) else {
4896                return Some(None);
4897            };
4898            lens.push(len);
4899            start = end;
4900        }
4901    }
4902    Some(Some(lens))
4903}
4904
4905/// How many offsets go in one packed run.
4906///
4907/// A payload block holds 1,024 values and `bitpack::pack_tail` takes fewer than 1,024 at a time,
4908/// since a whole unit of that many belongs in the transposed layout instead. So the offsets of a
4909/// block go in two runs. Five hundred and twelve values at any width is a whole number of bytes, so
4910/// a run starts where a multiply says it does and nothing is padded.
4911const TEXT_OFFSET_RUN: usize = 512;
4912
4913/// Bytes at the front of a global dictionary index: the value count, the values a payload block
4914/// holds, the block count and the bits an offset is packed at.
4915const DICTIONARY_HEADER: usize = 16;
4916
4917/// Set beside the offset width in the fourth word of a global dictionary index, meaning each
4918/// payload block says where in the file it starts and how long it is, rather than sitting directly
4919/// behind the block before it.
4920///
4921/// In that word rather than in a word of its own because the width is at most 32 and lives in a
4922/// `u32`, so the top of it has never been anything. A build old enough not to know the flag reads
4923/// the file's format before it reads any of this and refuses it there, and if it somehow did get
4924/// here it would find an offset width of two billion and say so.
4925///
4926/// The point of the flag is that a block written the moment it fills does not know what will be
4927/// written after it, so the payload of a column cannot be one run of bytes unless the whole column
4928/// is held until the file is closed. That is the memory the load cannot afford. What it costs is
4929/// eight bytes a block, against the block being a thousand values.
4930const DICTIONARY_SCATTERED: u32 = 1 << 31;
4931/// The dictionary index carries one four-byte substring signature per payload block.
4932const DICTIONARY_GRAMS: u32 = 1 << 30;
4933/// Each signature is [`TEXT_GRAM_BYTES`] long rather than the [`NARROW_GRAM_BYTES`] a format 28
4934/// file wrote.
4935const DICTIONARY_WIDE_GRAMS: u32 = 1 << 29;
4936/// Every flag the width word of a dictionary can carry above the offset width.
4937const DICTIONARY_FLAGS: u32 = DICTIONARY_SCATTERED | DICTIONARY_GRAMS | DICTIONARY_WIDE_GRAMS;
4938
4939/// How many entries of a dictionary's sorted order sit in one block that is read and checked as a
4940/// unit.
4941///
4942/// Five hundred and twelve entries is between two and three kilobytes on the ClickBench string
4943/// columns, which is well under a page. A binary search over half a million entries makes nineteen
4944/// probes, and the first ten land in ten different blocks while the last nine land in the one block
4945/// that holds the answer, so the whole search reads about thirty kilobytes of a megabyte of order. A
4946/// smaller block would save a little on the early probes, cost a checksum and an end list four times
4947/// as long, and give the heads less to share a base with. A larger one would read more than it uses
4948/// on every probe.
4949const TEXT_RANK_BLOCK: usize = 512;
4950
4951/// Bytes at the front of a rank block, which is the base of its heads and the width they are packed
4952/// at.
4953///
4954/// An entry used to be twelve bytes flat, eight for the head and four for the code, and on the five
4955/// ClickBench columns that have a dictionary worth the name that was 744 MB of a 12.2 GB file. Both
4956/// halves of it are nearly empty. The heads are the first eight bytes of the values in sorted order,
4957/// so a block of five hundred and twelve of them spans a tiny slice of the column, and on a column of
4958/// URLs they are all `http://w` and the block holds one distinct head. The codes are positions in a
4959/// dictionary of eighteen million, which is twenty five bits and not thirty two.
4960///
4961/// So a block now writes the smallest head in it, the bits the largest is above that, and the heads
4962/// and the codes packed at the width each needs. A block where every head agrees costs nine bytes
4963/// and the codes.
4964const RANK_BLOCK_HEADER: usize = size_of::<u64>() + 1;
4965
4966impl NativeText {
4967    /// One block of the payload, read and decoded the first time anything asks for a value in it.
4968    ///
4969    /// The bytes handed back are the values of the block laid end to end, which is what the offsets
4970    /// describe, so a caller slices it with the offsets it already has. Where the block sits in the
4971    /// file is the only thing the caller cannot work out for itself, because the stored form is
4972    /// shorter than the decoded one and by a different amount in every block.
4973    fn payload_block(&self, block: usize) -> Result<Option<&[u8]>> {
4974        let Some(slot) = self.blocks.get(block) else { return Ok(None) };
4975        let bytes = slot.get_or_init(|| self.decode_block(block)).as_ref().map_err(Clone::clone)?;
4976        Ok(Some(bytes.as_slice()))
4977    }
4978
4979    /// The character length of every value in one block, counted the first time it is asked for.
4980    ///
4981    /// The block is read out of [`Self::blocks`] where something already kept it and decoded and
4982    /// dropped where nothing did, so counting never adds a block to what this column holds. Two
4983    /// threads asking for the same block at once both count it and one of the two answers is kept,
4984    /// which costs a decode and is cheaper than a lock on every lookup.
4985    fn block_chars(&self, block: usize) -> Result<&[u32]> {
4986        let slot = self
4987            .char_lens
4988            .get(block)
4989            .ok_or_else(|| invalid("a block past the global dictionary"))?;
4990        if let Some(lens) = slot.get() {
4991            return Ok(lens);
4992        }
4993        let decoded;
4994        let bytes: &[u8] = match self.blocks.get(block).and_then(OnceLock::get) {
4995            Some(Ok(kept)) => kept,
4996            _ => {
4997                decoded = self.decode_block(block)?;
4998                &decoded
4999            }
5000        };
5001        let first = block * TEXT_PAYLOAD_VALUES;
5002        let last = (first + TEXT_PAYLOAD_VALUES).min(self.values);
5003        let ends = self.ends_within(first, last)?;
5004        if ends.len() != last - first {
5005            return Err(invalid("global dictionary offsets are short"));
5006        }
5007        let mut lens = Vec::with_capacity(ends.len());
5008        let mut start = u64::from(self.start_within(first)?);
5009        for &end in &ends {
5010            let value = usize::try_from(start)
5011                .ok()
5012                .zip(usize::try_from(end).ok())
5013                .and_then(|(from, to)| bytes.get(from..to))
5014                .ok_or_else(|| invalid("global dictionary value is past its block"))?;
5015            // A continuation byte of UTF-8 is `0b10xx_xxxx` and every other byte starts a
5016            // character, so the bytes that are not continuations are the characters.
5017            let characters = value.iter().filter(|byte| (**byte as i8) >= -0x40).count();
5018            lens.push(u32::try_from(characters).unwrap_or(u32::MAX));
5019            start = end;
5020        }
5021        Ok(slot.get_or_init(|| lens.into_boxed_slice()))
5022    }
5023
5024    /// Reads and decodes one block of the payload, without deciding who keeps it.
5025    ///
5026    /// [`Self::payload_block`] keeps it forever, which is what a point read wants and what a walk
5027    /// of the whole dictionary must not do. Both call this and they differ in nothing else.
5028    fn decode_block(&self, block: usize) -> Result<Vec<u8>> {
5029        let len = self.lengths[block];
5030        let mut stored = vec![
5031            0;
5032            usize::try_from(len).map_err(|_| invalid(
5033                "global dictionary block does not fit in memory"
5034            ))?
5035        ];
5036        read_at(&self.file, self.starts[block], &mut stored)?;
5037        if checksum(&stored) != self.hashes[block] {
5038            return Err(invalid("global dictionary payload checksum differs"));
5039        }
5040        let first = block * TEXT_PAYLOAD_VALUES;
5041        let last = (first + TEXT_PAYLOAD_VALUES).min(self.values);
5042        let want = self.end_within(last - 1)? as usize;
5043        let values = string::decode_flat(&stored)?;
5044        if values.len() != last - first {
5045            return Err(invalid("global dictionary block holds the wrong value count"));
5046        }
5047        let bytes = values.into_bytes();
5048        if bytes.len() != want {
5049            return Err(invalid("global dictionary block decodes to the wrong length"));
5050        }
5051        Ok(bytes)
5052    }
5053
5054    /// The block holding a value that a read hands over on loan, kept or decoded for the call.
5055    ///
5056    /// A block something already kept is read where it is. One nothing kept is kept the second
5057    /// time a loaned read decodes it while the column is holding less than [`Self::keep_budget`],
5058    /// and decoded into `decoded` and dropped with it otherwise, which is the policy
5059    /// [`TextSource::sweep`] explains. `scattered` is a read by code rather than in order, which
5060    /// stops dropping once it has dropped a column's worth of blocks, for the reason
5061    /// [`Self::visit_dropped`] gives.
5062    fn loaned_block<'a>(
5063        &'a self,
5064        block: usize,
5065        decoded: &'a mut Vec<u8>,
5066        scattered: bool,
5067    ) -> Result<&'a [u8]> {
5068        let kept = self.blocks.get(block).and_then(OnceLock::get);
5069        if let Some(Ok(kept)) = kept {
5070            return Ok(kept);
5071        }
5072        let again = kept.is_none()
5073            && self.swept.get(block).is_some_and(|swept| swept.swap(true, Atomic::Relaxed));
5074        let keep = again
5075            && (self.payload_kept.load(Atomic::Relaxed) < self.keep_budget
5076                || (scattered && self.visit_dropped.load(Atomic::Relaxed) >= self.blocks.len()));
5077        if keep {
5078            let kept = self
5079                .payload_block(block)?
5080                .ok_or_else(|| invalid("global dictionary block is past the payload"))?;
5081            self.payload_kept.fetch_add(kept.len(), Atomic::Relaxed);
5082            return Ok(kept);
5083        }
5084        *decoded = self.decode_block(block)?;
5085        if scattered && again {
5086            self.visit_dropped.fetch_add(1, Atomic::Relaxed);
5087        }
5088        Ok(decoded)
5089    }
5090
5091    /// How many single offset reads make [`Self::value_ends`] worth building.
5092    ///
5093    /// As many reads as the dictionary has values. Building the table costs about thirty
5094    /// instructions a value once the fresh pages it lands in are counted, and a read out of it saves
5095    /// about thirty five, so it repays itself after roughly one read per value. The reads so far are
5096    /// the only guess there is at the reads to come, and waiting until they match the size of the
5097    /// dictionary is betting that a column read that much will be read that much again.
5098    ///
5099    /// A sixteenth was the first answer, from counting the unpacking alone at three instructions a
5100    /// value. ClickBench 38 showed what that missed: it reads about twenty thousand titles a
5101    /// statement out of a dictionary of three hundred and fifty thousand, crossed a sixteenth in its
5102    /// second statement and was two percent slower for a table it did not read enough to repay. A
5103    /// scan asking for the length of every row crosses it part way through its first statement on
5104    /// ClickBench, where a string column has about two rows for every value, and a filter that keeps
5105    /// a few thousand rows never does. The floor is there
5106    /// because a short dictionary would otherwise build a table for a handful of reads.
5107    fn ends_worth_unpacking(&self) -> usize {
5108        self.values.max(TEXT_PAYLOAD_VALUES)
5109    }
5110
5111    /// The unpacked ends, if they are built or if this read is the one that makes them worth it.
5112    fn value_ends(&self) -> Option<&[u32]> {
5113        if let Some(built) = self.value_ends.get() {
5114            return built.as_deref();
5115        }
5116        if self.ends_asked.fetch_add(1, Atomic::Relaxed) < self.ends_worth_unpacking() {
5117            return None;
5118        }
5119        self.value_ends.get_or_init(|| self.unpack_ends()).as_deref()
5120    }
5121
5122    /// Every end of the column, a run at a time.
5123    ///
5124    /// `None` rather than an error on anything wrong, because this is a cache in front of a reader
5125    /// that answers the same question. A column whose offsets are short or whose ends do not fit in
5126    /// four bytes gets no table and the same error it would have got, from the read that wanted it.
5127    fn unpack_ends(&self) -> Option<Vec<u32>> {
5128        let mut ends = vec![0u32; self.values];
5129        for (run, into) in ends.chunks_mut(TEXT_OFFSET_RUN).enumerate() {
5130            let bytes = self.packed().get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)?;
5131            bitpack::unpack_tail_into(bytes, self.offset_bits, into, |bits| {
5132                u32::try_from(bits).unwrap_or(u32::MAX)
5133            })
5134            .ok()?;
5135        }
5136        // An end that did not fit was stored as the sentinel, and a real one cannot reach it because
5137        // a payload block is far smaller than four gigabytes. So the column keeps the packed reader.
5138        if ends.contains(&u32::MAX) { None } else { Some(ends) }
5139    }
5140
5141    /// The packed offsets, which is the index past its header.
5142    fn packed(&self) -> &[u8] {
5143        self.offsets.get(DICTIONARY_HEADER..).unwrap_or_default()
5144    }
5145
5146    /// Where the value at `index` ends inside its payload block.
5147    fn end_within(&self, index: usize) -> Result<u32> {
5148        if let Some(ends) = self.value_ends() {
5149            return ends
5150                .get(index)
5151                .copied()
5152                .ok_or_else(|| invalid("global dictionary offsets are short"));
5153        }
5154        let run = index / TEXT_OFFSET_RUN;
5155        let bytes = self
5156            .packed()
5157            .get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)
5158            .ok_or_else(|| invalid("global dictionary offsets are short"))?;
5159        let end = bitpack::tail_at(bytes, self.offset_bits, index % TEXT_OFFSET_RUN)
5160            .map_err(|_| invalid("global dictionary offsets are short"))?;
5161        u32::try_from(end).map_err(|_| invalid("global dictionary offset is past the payload"))
5162    }
5163
5164    /// Where every value in `first..last` ends inside its payload block, in one pass over the runs.
5165    ///
5166    /// [`Self::end_within`] answers for one value and pays for it twice over: it shifts a window to
5167    /// the bit the value starts at, and the copy that fills that window is a length the compiler does
5168    /// not know, so it is a call to `memcpy` rather than a load. A sweep asked for two of those per
5169    /// value, one for the end and one for the start that is the end before it, and on the ClickBench
5170    /// `URL` dictionary of eighteen million that was most of the half second a `LIKE` over it took.
5171    ///
5172    /// [`bitpack::unpack_tail_into`] walks the run instead, which makes the window a fixed width and
5173    /// so an unaligned load, and reads the bit position off a counter. A run is five hundred and
5174    /// twelve values and a block is two of them, so a block of a thousand and twenty four values
5175    /// costs two calls here and nothing per value.
5176    ///
5177    /// The answer is written straight into the result. A run that is wanted from its first value,
5178    /// which is every run but the one the sweep starts in, unpacks into its own window of the result
5179    /// and is never copied. Only a run joined part way through needs the scratch buffer, and there is
5180    /// at most one of those per sweep, so the buffer is allocated the first time one turns up.
5181    fn ends_within(&self, first: usize, last: usize) -> Result<Vec<u64>> {
5182        let mut ends = vec![0u64; last.saturating_sub(first)];
5183        let mut scratch = Vec::new();
5184        let mut at = first;
5185        while at < last {
5186            let run = at / TEXT_OFFSET_RUN;
5187            let stop = ((run + 1) * TEXT_OFFSET_RUN).min(last);
5188            let held = self.values.saturating_sub(run * TEXT_OFFSET_RUN).min(TEXT_OFFSET_RUN);
5189            let bytes = self
5190                .packed()
5191                .get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)
5192                .ok_or_else(|| invalid("global dictionary offsets are short"))?;
5193            let from = at % TEXT_OFFSET_RUN;
5194            let upto = stop - run * TEXT_OFFSET_RUN;
5195            if upto > held || bytes.len() < bitpack::tail_len(held, self.offset_bits) {
5196                return Err(invalid("global dictionary offsets are short"));
5197            }
5198            let into = &mut ends[at - first..stop - first];
5199            if from == 0 {
5200                bitpack::unpack_tail_into(bytes, self.offset_bits, into, |bits| bits)
5201                    .map_err(|_| invalid("global dictionary offsets are short"))?;
5202            } else {
5203                scratch.resize(held, 0);
5204                bitpack::unpack_tail_into(bytes, self.offset_bits, &mut scratch, |bits| bits)
5205                    .map_err(|_| invalid("global dictionary offsets are short"))?;
5206                into.copy_from_slice(&scratch[from..upto]);
5207            }
5208            at = stop;
5209        }
5210        Ok(ends)
5211    }
5212
5213    /// Where the value at `index` starts inside its payload block, which is where the value before
5214    /// it ended unless it is the first of the block.
5215    fn start_within(&self, index: usize) -> Result<u32> {
5216        if index.is_multiple_of(TEXT_PAYLOAD_VALUES) { Ok(0) } else { self.end_within(index - 1) }
5217    }
5218
5219    /// Where the value at `index` starts and ends inside its payload block.
5220    ///
5221    /// The two offsets sit next to each other in the same run unless the value opens one, and a run
5222    /// of seventeen bit offsets, which is what a block of a thousand strings needs, puts a pair of
5223    /// them inside one eight byte load. So the common case reads the packed bytes once rather than
5224    /// twice and does the bounds arithmetic once. This is asked once per string a text column hands
5225    /// out, and on ClickBench 27 the two reads together were a quarter of the query.
5226    fn span_within(&self, index: usize) -> Result<(u32, u32)> {
5227        if let Some(ends) = self.value_ends() {
5228            let end =
5229                *ends.get(index).ok_or_else(|| invalid("global dictionary offsets are short"))?;
5230            // The value before it in the same block, and zero where there is no value before it.
5231            // `index` is inside the table, so the one under it is too.
5232            let start = if index.is_multiple_of(TEXT_PAYLOAD_VALUES) { 0 } else { ends[index - 1] };
5233            if start > end {
5234                return Err(invalid("global dictionary value ends before it starts"));
5235            }
5236            return Ok((start, end));
5237        }
5238        let within = index % TEXT_OFFSET_RUN;
5239        let (start, end) = if within == 0 {
5240            (self.start_within(index)?, self.end_within(index)?)
5241        } else {
5242            let run = index / TEXT_OFFSET_RUN;
5243            let bytes = self
5244                .packed()
5245                .get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)
5246                .ok_or_else(|| invalid("global dictionary offsets are short"))?;
5247            let (start, end) = bitpack::tail_pair(bytes, self.offset_bits, within)
5248                .map_err(|_| invalid("global dictionary offsets are short"))?;
5249            let ends = u32::try_from(end)
5250                .map_err(|_| invalid("global dictionary offset is past the payload"))?;
5251            let starts = u32::try_from(start)
5252                .map_err(|_| invalid("global dictionary offset is past the payload"))?;
5253            (starts, ends)
5254        };
5255        if start > end {
5256            return Err(invalid("global dictionary value ends before it starts"));
5257        }
5258        Ok((start, end))
5259    }
5260
5261    /// The block of the sorted order that holds `rank`, and where in it that rank sits.
5262    ///
5263    /// The block is read from the file and checked against the hash the index carries for it the
5264    /// first time anything asks, and kept after that, the same way a payload block is. A search
5265    /// makes about as many probes as the order has bits, so the whole search reads a handful of
5266    /// these and never the rest.
5267    fn rank_parts(&self, rank: usize) -> Result<(&[u8], usize)> {
5268        let slot = self
5269            .rank_blocks
5270            .get(rank / TEXT_RANK_BLOCK)
5271            .ok_or_else(|| invalid("global dictionary rank is past the order"))?;
5272        let block = slot
5273            .get_or_init(|| {
5274                let mut bytes = Vec::new();
5275                self.read_rank_block(rank / TEXT_RANK_BLOCK, &mut bytes)?;
5276                Ok(bytes)
5277            })
5278            .as_ref()
5279            .map_err(Clone::clone)?;
5280        Ok((block.as_slice(), rank % TEXT_RANK_BLOCK))
5281    }
5282
5283    /// Reads block `which` of the sorted order into `bytes`, checked against the hash the index
5284    /// carries for it.
5285    fn read_rank_block(&self, which: usize, bytes: &mut Vec<u8>) -> Result<()> {
5286        let start = if which == 0 { 0 } else { self.rank_ends[which - 1] };
5287        let end = self.rank_ends[which];
5288        bytes.clear();
5289        bytes.resize((end - start) as usize, 0);
5290        read_at(&self.file, self.rank_at + start, bytes)?;
5291        let expected = self
5292            .rank_hashes
5293            .get(which)
5294            .ok_or_else(|| invalid("global dictionary rank block has no checksum"))?;
5295        if checksum(bytes) != *expected {
5296            return Err(invalid("global dictionary rank checksum differs"));
5297        }
5298        Ok(())
5299    }
5300
5301    /// The first eight bytes of the value at `rank`, as the integer a comparison reads.
5302    fn head_at(&self, rank: usize) -> Result<u64> {
5303        let (block, within) = self.rank_parts(rank)?;
5304        let (base, width, packed) = rank_heads(block)?;
5305        let above = bitpack::tail_at(packed, width, within)
5306            .map_err(|_| invalid("global dictionary rank block is short of heads"))?;
5307        Ok(base.wrapping_add(above))
5308    }
5309
5310    /// The packed codes of one rank block, which follow the heads on the next byte boundary.
5311    fn rank_codes<'block>(&self, block: &'block [u8], count: usize) -> Result<&'block [u8]> {
5312        let (_, width, packed) = rank_heads(block)?;
5313        packed
5314            .get(bitpack::tail_len(count, width)..)
5315            .ok_or_else(|| invalid("global dictionary rank block is short of codes"))
5316    }
5317
5318    /// How many entries the block holding `rank` has, which is a full block except at the end.
5319    fn rank_block_len(&self, rank: usize) -> usize {
5320        let first = rank / TEXT_RANK_BLOCK * TEXT_RANK_BLOCK;
5321        TEXT_RANK_BLOCK.min(self.ranks - first)
5322    }
5323}
5324
5325/// The base, the width and the packed bytes of one rank block's heads.
5326fn rank_heads(block: &[u8]) -> Result<(u64, usize, &[u8])> {
5327    let header = block
5328        .get(..RANK_BLOCK_HEADER)
5329        .ok_or_else(|| invalid("global dictionary rank block is short"))?;
5330    let base = u64::from_le_bytes(header[..8].try_into().expect("eight bytes"));
5331    let width = header[8] as usize;
5332    if width > 64 {
5333        return Err(invalid("global dictionary rank block packs heads past a word"));
5334    }
5335    Ok((base, width, &block[RANK_BLOCK_HEADER..]))
5336}
5337
5338/// Bits one offset of a dictionary takes, which is what its widest payload block spans.
5339///
5340/// One width for the whole column rather than one a block. A block is 1,024 values of the same
5341/// column, so the blocks of a column are within a factor of two of each other on every ClickBench
5342/// string column, and a width a block would save a fraction of a bit and cost a byte a block plus
5343/// the arithmetic that finds where a block starts.
5344fn offset_width(ends: &[u32]) -> usize {
5345    // The ends are already relative to the block the value is in, so the last end of a block is that
5346    // block's total and the largest end anywhere is the widest block. There is no subtraction left
5347    // to do and no need to walk the blocks to find where one starts.
5348    let span = ends.iter().copied().max().unwrap_or(0);
5349    (u32::BITS - span.leading_zeros()) as usize
5350}
5351
5352/// How many bytes `values` offsets take at `bits`, which is what the reader has to know before it
5353/// has read any of them.
5354fn offset_bytes(values: usize, bits: usize) -> usize {
5355    let full = values / TEXT_OFFSET_RUN;
5356    let rest = values % TEXT_OFFSET_RUN;
5357    full * TEXT_OFFSET_RUN / 8 * bits + bitpack::tail_len(rest, bits)
5358}
5359
5360/// The end of every value within its payload block, packed a run at a time.
5361/// A run never straddles a block, because [`TEXT_OFFSET_RUN`] divides [`TEXT_PAYLOAD_VALUES`], which
5362/// is what lets this be a walk of the ends rather than arithmetic against a per block base.
5363fn encode_offsets(ends: &[u32], bits: usize, out: &mut Vec<u8>) -> Result<()> {
5364    let mut run = Vec::with_capacity(TEXT_OFFSET_RUN);
5365    for chunk in ends.chunks(TEXT_OFFSET_RUN) {
5366        run.clear();
5367        run.extend(chunk.iter().map(|&end| u64::from(end)));
5368        bitpack::pack_tail(&run, bits, out)
5369            .map_err(|_| invalid("global dictionary offsets do not pack"))?;
5370    }
5371    Ok(())
5372}
5373
5374/// How many bits a code of a dictionary of `values` entries takes.
5375fn code_width(values: usize) -> usize {
5376    match u64::try_from(values).unwrap_or(u64::MAX) {
5377        0 | 1 => 0,
5378        last => (u64::BITS - (last - 1).leading_zeros()) as usize,
5379    }
5380}
5381
5382impl TextSource for NativeText {
5383    fn len(&self) -> usize {
5384        self.values
5385    }
5386
5387    fn might_contain(&self, first: usize, literal: &[u8]) -> Result<bool> {
5388        let Some(grams) = &self.grams else { return Ok(true) };
5389        if literal.len() < 4 || first >= self.values {
5390            return Ok(true);
5391        }
5392        let verdict = grams.verdicts(&self.file, literal)?;
5393        Ok(verdict.get(first / TEXT_PAYLOAD_VALUES).copied().unwrap_or(true))
5394    }
5395
5396    fn bytes_at(&self, index: usize) -> Result<Option<&[u8]>> {
5397        if index >= self.values {
5398            return Ok(None);
5399        }
5400        let (start, end) = self.span_within(index)?;
5401        if start == end {
5402            return Ok(Some(&[]));
5403        }
5404        // A block holds a fixed number of values rather than a fixed number of bytes, so the value
5405        // is in one block and the offsets already say where in it.
5406        let block = index / TEXT_PAYLOAD_VALUES;
5407        let Some(bytes) = self.payload_block(block)? else { return Ok(None) };
5408        Ok(bytes.get(start as usize..end as usize))
5409    }
5410
5411    fn bytes_len_at(&self, index: usize) -> Result<Option<usize>> {
5412        if index >= self.values {
5413            return Ok(None);
5414        }
5415        let (start, end) = self.span_within(index)?;
5416        Ok(Some((end - start) as usize))
5417    }
5418
5419    /// Every length out of the unpacked ends in one loop, which is the point of having them.
5420    ///
5421    /// The whole run of positions counts towards [`Self::ends_worth_unpacking`] at once, because a
5422    /// caller asking for a vector of lengths has said how many it wants, and a vector of them is
5423    /// usually enough on its own. Until the table is worth building this is the row at a time read,
5424    /// the same as the default.
5425    fn bytes_lens_at(&self, indices: &[u32], into: &mut Vec<i64>) -> Result<()> {
5426        into.reserve(indices.len());
5427        // Once the table is built the count has nothing left to decide, and every thread of a scan
5428        // adding to the one counter moves its cache line from core to core on every chunk.
5429        if let Some(Some(lens)) = self.value_lens.get() {
5430            lens.extend_at(indices, into);
5431            return Ok(());
5432        }
5433        self.ends_asked.fetch_add(indices.len(), Atomic::Relaxed);
5434        let Some(ends) = self.value_ends() else {
5435            for &index in indices {
5436                into.push(
5437                    self.bytes_len_at(index as usize)?
5438                        .map_or(0, |len| i64::try_from(len).unwrap_or(i64::MAX)),
5439                );
5440            }
5441            return Ok(());
5442        };
5443        if let Some(lens) = self.value_lens.get_or_init(|| lengths_of(ends)) {
5444            lens.extend_at(indices, into);
5445            return Ok(());
5446        }
5447        for &index in indices {
5448            let index = index as usize;
5449            // Past the end is no value and so no length, which is what a row at a time read says.
5450            let Some(&end) = ends.get(index) else {
5451                into.push(0);
5452                continue;
5453            };
5454            let start = if index.is_multiple_of(TEXT_PAYLOAD_VALUES) { 0 } else { ends[index - 1] };
5455            if start > end {
5456                return Err(invalid("global dictionary value ends before it starts"));
5457            }
5458            into.push(i64::from(end - start));
5459        }
5460        Ok(())
5461    }
5462
5463    /// Every length in characters out of the counts kept a block at a time, which is what keeps a
5464    /// scan of `length` from holding the column decoded. See [`NativeText::char_lens`].
5465    fn chars_lens_at(&self, indices: &[u32], into: &mut Vec<i64>) -> Result<()> {
5466        into.reserve(indices.len());
5467        for &index in indices {
5468            let index = index as usize;
5469            // Past the end is no value and so no length, which is what a row at a time read says.
5470            if index >= self.values {
5471                into.push(0);
5472                continue;
5473            }
5474            let lens = self.block_chars(index / TEXT_PAYLOAD_VALUES)?;
5475            let len = lens
5476                .get(index % TEXT_PAYLOAD_VALUES)
5477                .ok_or_else(|| invalid("global dictionary block holds the wrong value count"))?;
5478            into.push(i64::from(*len));
5479        }
5480        Ok(())
5481    }
5482
5483    /// The rest of the block holding `first`, decoded into a buffer that may die with the call.
5484    ///
5485    /// A block is the unit this format decodes, so a walk that wants every value is going to decode
5486    /// every block whatever it does. The question is whether it keeps them, and both answers are
5487    /// wrong on their own. [`Self::payload_block`] keeps every block it is asked for, so a reader
5488    /// that walked the whole dictionary through `bytes_at` ended up holding the whole dictionary
5489    /// decoded, 4.2 GB on ClickBench `URL`. Keeping none of them makes the next statement asking
5490    /// the same question decode all of it again, which on the same column at a million rows is a
5491    /// `LIKE` going from 2.7 ms to 16.2 ms.
5492    ///
5493    /// So a sweep keeps what it decodes for the second time while the column is under
5494    /// [`TEXT_KEEP_BUDGET`] and drops it after that. A block already in hand is used where it is there and costs nothing either way.
5495    fn sweep(
5496        &self,
5497        first: usize,
5498        limit: usize,
5499        body: &mut dyn FnMut(usize, &[u8]) -> Result<()>,
5500    ) -> Result<usize> {
5501        let limit = limit.min(self.values);
5502        if first >= limit {
5503            return Ok(first);
5504        }
5505        let block = first / TEXT_PAYLOAD_VALUES;
5506        let last = ((block + 1) * TEXT_PAYLOAD_VALUES).min(limit);
5507        let mut decoded = Vec::new();
5508        let bytes = self.loaned_block(block, &mut decoded, false)?;
5509        let ends = self.ends_within(first, last)?;
5510        if ends.len() != last - first {
5511            return Err(invalid("global dictionary offsets are short"));
5512        }
5513        let mut start = u64::from(self.start_within(first)?);
5514        // row at a time: the caller is handed one value after another, and what it does with one is
5515        // its own business, so there is no shape here for anything but a walk.
5516        for (index, &end) in (first..last).zip(&ends) {
5517            let value = usize::try_from(start)
5518                .ok()
5519                .zip(usize::try_from(end).ok())
5520                .and_then(|(from, to)| bytes.get(from..to))
5521                .ok_or_else(|| invalid("global dictionary value is past its block"))?;
5522            body(index, value)?;
5523            start = end;
5524        }
5525        Ok(last)
5526    }
5527
5528    /// The values at `indices` a block at a time, each block read once for the call.
5529    ///
5530    /// The positions are put in code order first, because the codes of a vector are in row order
5531    /// and land all over the dictionary, and read in that order each block a vector touches would
5532    /// be looked up once for every row in it. Whether a block is kept is
5533    /// [`NativeText::loaned_block`]'s decision, which keeps at most the budget of this column
5534    /// until the reads have shown they come back to the same blocks too often for dropping them to
5535    /// be cheap.
5536    fn visit_at(
5537        &self,
5538        indices: &[u32],
5539        body: &mut dyn FnMut(usize, &[u8]) -> Result<()>,
5540    ) -> Result<()> {
5541        let mut order = (0..indices.len()).collect::<Vec<_>>();
5542        order.sort_unstable_by_key(|&at| indices[at]);
5543        let block_of = |at: usize| {
5544            let index = indices[at] as usize;
5545            (index < self.values).then_some(index / TEXT_PAYLOAD_VALUES)
5546        };
5547        let mut decoded = Vec::new();
5548        let mut run = 0;
5549        while run < order.len() {
5550            let Some(block) = block_of(order[run]) else {
5551                // Past the end is no value, and every position after this one is past it too.
5552                for &at in &order[run..] {
5553                    body(at, &[])?;
5554                }
5555                break;
5556            };
5557            let upto = run + order[run..].partition_point(|&at| block_of(at) == Some(block));
5558            let bytes = self.loaned_block(block, &mut decoded, true)?;
5559            for &at in &order[run..upto] {
5560                let (start, end) = self.span_within(indices[at] as usize)?;
5561                let value = bytes
5562                    .get(start as usize..end as usize)
5563                    .ok_or_else(|| invalid("global dictionary value is past its block"))?;
5564                body(at, value)?;
5565            }
5566            run = upto;
5567        }
5568        Ok(())
5569    }
5570
5571    /// Each block the indices land in, decoded once and dropped, or read where it is already kept.
5572    ///
5573    /// Never kept, unlike [`Self::sweep`] under its budget, because a scattered read is a one off:
5574    /// a synopsis turned into values is turned once and remembered by the reader as values, a few
5575    /// kilobytes, where the blocks it went through are megabytes nobody asks for again.
5576    fn visit(
5577        &self,
5578        indices: &[usize],
5579        body: &mut dyn FnMut(usize, &[u8]) -> Result<()>,
5580    ) -> Result<()> {
5581        let mut at = 0;
5582        while at < indices.len() {
5583            let block = indices[at] / TEXT_PAYLOAD_VALUES;
5584            let upto =
5585                at + indices[at..].partition_point(|&index| index / TEXT_PAYLOAD_VALUES == block);
5586            let wanted = &indices[at..upto];
5587            if wanted.iter().any(|&index| index >= self.values) {
5588                return Err(invalid("a visited value is past the global dictionary"));
5589            }
5590            let decoded;
5591            let bytes: &[u8] = match self.blocks.get(block).and_then(OnceLock::get) {
5592                Some(Ok(kept)) => kept,
5593                _ => {
5594                    decoded = self.decode_block(block)?;
5595                    &decoded
5596                }
5597            };
5598            for (offset, &index) in wanted.iter().enumerate() {
5599                let (start, end) = self.span_within(index)?;
5600                let value = bytes
5601                    .get(start as usize..end as usize)
5602                    .ok_or_else(|| invalid("global dictionary value is past its block"))?;
5603                body(at + offset, value)?;
5604            }
5605            at = upto;
5606        }
5607        Ok(())
5608    }
5609
5610    fn ranks(&self) -> Option<usize> {
5611        (self.ranks > 0).then_some(self.ranks)
5612    }
5613
5614    /// The boundary for `wanted`, out of [`Self::searched`] where it is there and put there where
5615    /// it is not.
5616    ///
5617    /// The lock is held over the search rather than dropped and taken again, so that two threads
5618    /// asking for the same value at the same time do the work once between them. That is the shape
5619    /// the scan actually arrives in: sixteen instances of a top N, all reading the same column, all
5620    /// improving their bound over the same early chunks.
5621    fn below(&self, ranks: usize, wanted: &[u8]) -> Result<(usize, bool)> {
5622        let mut memo = self.searched.lock().map_err(|_| invalid("a poisoned dictionary search"))?;
5623        if let Some(&answer) = memo.get(wanted) {
5624            return Ok(answer);
5625        }
5626        let answer = search_below(self, ranks, wanted)?;
5627        if memo.len() >= TEXT_SEARCH_MEMO {
5628            memo.clear();
5629        }
5630        memo.insert(wanted.to_vec(), answer);
5631        Ok(answer)
5632    }
5633
5634    fn compare_rank(&self, rank: usize, wanted: &[u8]) -> Result<Ordering> {
5635        // The head settles the probe unless the two values start with the same eight bytes, and
5636        // only then is a value read. On a column of URLs that is the difference between a search
5637        // that touches one block of the payload and a search that touches nineteen of them.
5638        let settled = self.head_at(rank)?.cmp(&head(wanted));
5639        if settled != Ordering::Equal {
5640            return Ok(settled);
5641        }
5642        let code = self.code_at_rank(rank)?;
5643        let bytes = self
5644            .bytes_at(code as usize)?
5645            .ok_or_else(|| invalid("global dictionary order names a code it does not have"))?;
5646        Ok(bytes.cmp(wanted))
5647    }
5648
5649    fn code_at_rank(&self, rank: usize) -> Result<u32> {
5650        let (block, within) = self.rank_parts(rank)?;
5651        let codes = self.rank_codes(block, self.rank_block_len(rank))?;
5652        let code = bitpack::tail_at(codes, self.code_bits, within)
5653            .map_err(|_| invalid("global dictionary rank block is short of codes"))?;
5654        let code = u32::try_from(code)
5655            .map_err(|_| invalid("global dictionary order names a code it does not have"))?;
5656        if code as usize >= self.len() {
5657            return Err(invalid("global dictionary order names a code it does not have"));
5658        }
5659        Ok(code)
5660    }
5661
5662    fn code_ranks(&self) -> Option<&[u32]> {
5663        // The order is a permutation of the positions, so inverting it needs every position to be
5664        // named exactly once. Anything else and the slice would have holes, and a caller indexing
5665        // it by a code would read a rank that belongs to nothing.
5666        if self.ranks == 0 || self.ranks != self.len() {
5667            return None;
5668        }
5669        self.code_ranks
5670            .get_or_init(|| {
5671                let mut ranks = vec![u32::MAX; self.ranks];
5672                // A block at a time rather than a rank at a time, because reading it per rank pays
5673                // for the bounds check, the division and the lock on every one of them.
5674                //
5675                // A block nothing has read yet is read into one buffer that is reused, rather than
5676                // through `rank_parts`, which would keep every block of the order once this is
5677                // done with it. The inverse is all anything wants after this, and on the `Referer`
5678                // column of the ClickBench file the blocks are tens of megabytes held for nothing.
5679                let mut scratch = Vec::new();
5680                let mut codes = vec![0u64; TEXT_RANK_BLOCK];
5681                for first in (0..self.ranks).step_by(TEXT_RANK_BLOCK) {
5682                    let which = first / TEXT_RANK_BLOCK;
5683                    let block = match self.rank_blocks.get(which)?.get() {
5684                        Some(kept) => kept.as_ref().ok()?.as_slice(),
5685                        None => {
5686                            self.read_rank_block(which, &mut scratch).ok()?;
5687                            scratch.as_slice()
5688                        }
5689                    };
5690                    let count = self.rank_block_len(first);
5691                    let packed = self.rank_codes(block, count).ok()?;
5692                    let codes = codes.get_mut(..count)?;
5693                    bitpack::unpack_tail_into(packed, self.code_bits, codes, |bits| bits).ok()?;
5694                    for (within, &code) in codes.iter().enumerate() {
5695                        let code = usize::try_from(code).ok()?;
5696                        *ranks.get_mut(code)? = u32::try_from(first + within).ok()?;
5697                    }
5698                }
5699                if ranks.contains(&u32::MAX) {
5700                    return None;
5701                }
5702                Some(ranks)
5703            })
5704            .as_deref()
5705    }
5706
5707    fn footprint(&self) -> usize {
5708        self.offsets.capacity()
5709            + self
5710                .value_ends
5711                .get()
5712                .and_then(Option::as_ref)
5713                .map_or(0, |ends| ends.capacity() * size_of::<u32>())
5714            + self.value_lens.get().and_then(Option::as_ref).map_or(0, Lengths::footprint)
5715            + self
5716                .code_ranks
5717                .get()
5718                .and_then(Option::as_ref)
5719                .map_or(0, |ranks| ranks.capacity() * size_of::<u32>())
5720            + self.rank_hashes.capacity() * size_of::<u64>()
5721            + self.rank_ends.capacity() * size_of::<u64>()
5722            + self.rank_blocks.capacity() * size_of::<OnceLock<Result<Vec<u8>>>>()
5723            + self
5724                .rank_blocks
5725                .iter()
5726                .filter_map(OnceLock::get)
5727                .filter_map(|result| result.as_ref().ok())
5728                .map(Vec::capacity)
5729                .sum::<usize>()
5730            + self.blocks.capacity() * size_of::<OnceLock<Result<Vec<u8>>>>()
5731            + self.char_lens.capacity() * size_of::<OnceLock<Box<[u32]>>>()
5732            + self
5733                .char_lens
5734                .iter()
5735                .filter_map(OnceLock::get)
5736                .map(|lens| lens.len() * size_of::<u32>())
5737                .sum::<usize>()
5738            + self.hashes.capacity() * size_of::<u64>()
5739            + self.starts.capacity() * size_of::<u64>()
5740            + self.lengths.capacity() * size_of::<u64>()
5741            + self.grams.as_ref().map_or(0, NativeGrams::footprint)
5742            + self
5743                .blocks
5744                .iter()
5745                .filter_map(OnceLock::get)
5746                .filter_map(|result| result.as_ref().ok())
5747                .map(Vec::capacity)
5748                .sum::<usize>()
5749    }
5750}
5751
5752/// Every table wide part number in order, with the stripe it belongs to.
5753fn places(table: &Table) -> Result<Vec<Place>> {
5754    let mut places = Vec::with_capacity(table.stripes.len().saturating_mul(STRIPE_PARTS));
5755    for (at, stripe) in table.stripes.iter().enumerate() {
5756        let index = u32::try_from(at).map_err(|_| invalid("too many stripes"))?;
5757        for (part, &rows) in stripe.parts.iter().enumerate() {
5758            places.push(Place {
5759                stripe: index,
5760                part: u32::try_from(part).map_err(|_| invalid("too many parts in a stripe"))?,
5761                rows,
5762            });
5763        }
5764    }
5765    Ok(places)
5766}
5767
5768/// Reads one column's section of a stripe's index page.
5769///
5770/// The section carries its own checksum, so a reader that wants one column out of a hundred and
5771/// five preads a few hundred bytes and still knows that what it got is what was written.
5772fn read_index<F: Positional + ?Sized>(
5773    file: &F,
5774    stripe: &Stripe,
5775    column: usize,
5776) -> Result<Vec<PartSpan>> {
5777    let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
5778    read_index_span(file, stripe.index, *page, stripe.parts.len(), column)
5779}
5780
5781fn read_index_span<F: Positional + ?Sized>(
5782    file: &F,
5783    index: Span,
5784    page: Span,
5785    parts: usize,
5786    column: usize,
5787) -> Result<Vec<PartSpan>> {
5788    let section = index_section(parts)?;
5789    let at = column.checked_mul(section).ok_or_else(|| invalid("index page offset overflow"))?;
5790    let end = at.checked_add(section).ok_or_else(|| invalid("index page offset overflow"))?;
5791    if end > index.length as usize {
5792        return Err(invalid("index page is shorter than its columns"));
5793    }
5794    let mut bytes = vec![0; section];
5795    let offset =
5796        index.offset.checked_add(at as u64).ok_or_else(|| invalid("index page offset overflow"))?;
5797    read_at(file, offset, &mut bytes)?;
5798    let entries = section - size_of::<u64>();
5799    let stored = u64::from_le_bytes(bytes[entries..].try_into().expect("eight bytes"));
5800    if checksum(&bytes[..entries]) != stored {
5801        // With where it was read from, because the two ways this fires look identical from the
5802        // message alone: a file somebody damaged, and a file we wrote to the wrong offset.
5803        return Err(invalid(&format!(
5804            "index page section checksum differs, column {column} of {parts} parts at {offset}, \
5805             wanted {stored:016x} and got {:016x}",
5806            checksum(&bytes[..entries]),
5807        )));
5808    }
5809    let mut spans = Vec::with_capacity(parts);
5810    let mut start = 0_usize;
5811    for part in 0..parts {
5812        let at = part * INDEX_ENTRY;
5813        let length = u32::from_le_bytes(bytes[at..at + 4].try_into().expect("four bytes")) as usize;
5814        let hash = u64::from_le_bytes(bytes[at + 4..at + 12].try_into().expect("eight bytes"));
5815        spans.push(PartSpan { start, length, hash });
5816        start = start.checked_add(length).ok_or_else(|| invalid("column page length overflow"))?;
5817    }
5818    if start != page.length as usize {
5819        return Err(invalid("column page length differs from its index"));
5820    }
5821    Ok(spans)
5822}
5823
5824/// One part's bytes out of a whole column page.
5825fn part_bytes(page: &[u8], span: PartSpan) -> Result<&[u8]> {
5826    let end = span.start.checked_add(span.length).ok_or_else(|| invalid("part range overflow"))?;
5827    page.get(span.start..end).ok_or_else(|| invalid("part exceeds its column page"))
5828}
5829
5830/// Marks part `part` of a stripe of `parts` parts read, and says whether it had been read before and
5831/// whether every part of the stripe had been before this one was asked for again.
5832fn touch(bits: &mut Vec<u64>, part: usize, parts: usize) -> (bool, bool) {
5833    if bits.is_empty() {
5834        bits.resize(parts.div_ceil(64).max(1), 0);
5835    }
5836    let (word, bit) = (part / 64, 1_u64 << (part % 64));
5837    let Some(held) = bits.get_mut(word) else { return (false, false) };
5838    let again = *held & bit != 0;
5839    *held |= bit;
5840    let through = bits.iter().map(|word| word.count_ones() as usize).sum::<usize>() >= parts;
5841    (again, again && through)
5842}
5843
5844/// Puts one stripe of one column in the cache, and hands back the page for the pool to count when
5845/// it is a page the column did not already hold.
5846///
5847/// The index goes in its own slot and stays. Only the page is under the budget, and the pool is
5848/// what enforces it, once the caller has let go of the column's lock.
5849fn remember(cached: &mut Cached, held: &CachedColumn) -> Option<(usize, Arc<AtomicBool>)> {
5850    if let Some(slot) = cached.index.get_mut(held.stripe)
5851        && slot.is_none()
5852    {
5853        *slot = Some(Arc::clone(&held.index));
5854    }
5855    let page = held.page.clone()?;
5856    let slot = cached.pages.get_mut(held.stripe)?;
5857    if slot.is_some() {
5858        return None;
5859    }
5860    let bytes = page.bytes.len();
5861    // Set, so that the page a worker has just paid to read is not the one the pass it pays for
5862    // lets go of before the worker has read a part out of it.
5863    let used = Arc::new(AtomicBool::new(true));
5864    *slot = Some(Resident { page, used: Arc::clone(&used) });
5865    Some((bytes, used))
5866}
5867
5868/// Every table a native file holds, without the directory of any of them.
5869///
5870/// This is what opening a database reads. It is the small level of the directory, so the cost is
5871/// proportional to how many tables there are rather than to how much data they hold, and a session
5872/// that touches two tables of eight decodes two table directories.
5873///
5874/// The file handle is shared with every reader this hands out. Eight tables in one file is one open
5875/// file descriptor, not eight, which is the other thing one file buys over a file per table.
5876#[derive(Debug, Clone)]
5877pub struct Catalog {
5878    file: Arc<File>,
5879    size: u64,
5880    entries: Arc<Vec<Entry>>,
5881    /// The views the file holds, whole, since a view has no second level to read later.
5882    views: Arc<Vec<ViewEntry>>,
5883    opening: Opening,
5884    /// Where every reader this hands out counts its pages.
5885    pool: PagePool,
5886}
5887
5888/// Signed integer sums and non-null counts for selected columns, plus total table rows.
5889#[derive(Debug, Clone, PartialEq, Eq)]
5890pub struct CertifiedSums {
5891    pub columns: Vec<(i128, u64)>,
5892    pub rows: u64,
5893}
5894
5895/// Exact ends of an integer or date column, including a certified all-null column.
5896#[derive(Debug, Clone, Copy, PartialEq, Eq)]
5897pub enum IntegerExtremes {
5898    Null,
5899    Values { low: i128, high: i128 },
5900}
5901
5902/// A complete numeric value-to-row-count synopsis; `None` represents SQL NULL.
5903pub type NumericFrequencies = Vec<(Option<i128>, u64)>;
5904
5905impl Catalog {
5906    /// Reads the highest valid catalog slot and nothing under it.
5907    ///
5908    /// The readers it hands out keep pages in a pool of their own with no budget, so each column
5909    /// holds its floor of four stripes and no more. A database opens with [`Catalog::open_in`].
5910    ///
5911    /// # Errors
5912    ///
5913    /// If the file has no valid committed catalog or a catalog pointer is out of bounds.
5914    pub fn open(path: impl AsRef<Path>) -> Result<Self> {
5915        Self::open_in(path, &PagePool::default())
5916    }
5917
5918    /// The same, with every reader it hands out keeping its pages in `pool`.
5919    ///
5920    /// # Errors
5921    ///
5922    /// If the file has no valid committed catalog or a catalog pointer is out of bounds.
5923    pub fn open_in(path: impl AsRef<Path>, pool: &PagePool) -> Result<Self> {
5924        let path = path.as_ref();
5925        let (file, size, _, bytes, opening) = slot_bytes(path)?;
5926        let (entries, views, card) = decode_catalog(&bytes, size)?;
5927        remember_card(path, card.as_ref());
5928        Ok(Self {
5929            file: Arc::new(file),
5930            size,
5931            entries: Arc::new(entries),
5932            views: Arc::new(views),
5933            opening,
5934            pool: pool.clone(),
5935        })
5936    }
5937
5938    /// The tables in the file, in the order they were written.
5939    pub fn names(&self) -> impl ExactSizeIterator<Item = &str> {
5940        self.entries.iter().map(|entry| entry.name.as_str())
5941    }
5942
5943    /// The same tables with how many rows each of them holds.
5944    ///
5945    /// The names alone answer which tables the file has, which is what a checkpoint needs to know.
5946    /// A load asks a second question: whether a table already in the file is really in the way of
5947    /// the one it wants to write. A table with no rows is not, because it has no pages the next
5948    /// generation would have to carry, so the count has to come out of the catalog beside the name.
5949    pub fn rows(&self) -> impl ExactSizeIterator<Item = (&str, usize)> {
5950        self.entries.iter().map(|entry| (entry.name.as_str(), entry.rows))
5951    }
5952
5953    /// The views in the file, in the order they were written.
5954    ///
5955    /// Whole, unlike [`Catalog::names`], which hands back names and makes the caller ask for a table
5956    /// by one. A view is a few strings and a column list and it was all read at open, so there is
5957    /// nothing left to go and fetch and no reason to make the caller ask twice.
5958    pub fn views(&self) -> impl ExactSizeIterator<Item = &ViewEntry> {
5959        self.views.iter()
5960    }
5961
5962    /// How many tables the file holds.
5963    #[must_use]
5964    pub fn len(&self) -> usize {
5965        self.entries.len()
5966    }
5967
5968    /// Whether the file holds no table at all, which is what [`Writer::empty`] writes and what a
5969    /// database somebody dropped the last table out of comes back as.
5970    #[must_use]
5971    pub fn is_empty(&self) -> bool {
5972        self.entries.is_empty()
5973    }
5974
5975    /// Opens one table by name, decoding its directory now.
5976    ///
5977    /// # Errors
5978    ///
5979    /// If there is no table by that name, or its directory is torn or points outside the file.
5980    pub fn table(&self, name: &str) -> Result<Reader> {
5981        let entry = self
5982            .entries
5983            .iter()
5984            .find(|entry| entry.name == name)
5985            .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
5986        // Checked and then decoded a window at a time, so that the directory's own bytes are never
5987        // all in memory beside the table they decode into. It is read twice, and the second read
5988        // comes out of the page cache the first one filled.
5989        let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
5990        if file_checksum(&self.file, offset, length)? != entry.directory.hash {
5991            return Err(invalid(&format!("the directory of table {name} does not checksum")));
5992        }
5993        let mut opening = self.opening;
5994        opening.reads += 1;
5995        opening.bytes += u64::from(entry.directory.length);
5996        Reader::build(
5997            Arc::clone(&self.file),
5998            self.size,
5999            read_directory(Cursor::over(&self.file, offset, length), self.size, Some(offset))?,
6000            u64::from(entry.directory.length),
6001            opening,
6002            self.pool.clone(),
6003        )
6004    }
6005
6006    /// Counts one signed integer column from its encoded parts without building metadata for
6007    /// unrelated columns. The counts are computed from row encodings when this is called.
6008    /// Nullable and non-cascade parts use the ordinary decoder for that part.
6009    ///
6010    /// # Errors
6011    ///
6012    /// If the directory, selected page index, checksum, or encoded integer is invalid.
6013    pub fn integer_tally(&self, name: &str, column: usize) -> Result<Option<Vec<(i64, u64)>>> {
6014        let mut counts = BTreeMap::<i64, u64>::new();
6015        let Some(()) = self.integer_fold(name, column, |value, count| {
6016            let held = counts.entry(value).or_default();
6017            *held = held.checked_add(count).ok_or_else(|| invalid("integer count overflow"))?;
6018            Ok(())
6019        })?
6020        else {
6021            return Ok(None);
6022        };
6023        Ok(Some(counts.into_iter().collect()))
6024    }
6025
6026    /// Visits a signed integer column's row values without building per-part or table-wide count
6027    /// maps. The caller combines the emitted counts for its query at runtime.
6028    ///
6029    /// # Errors
6030    ///
6031    /// If the selected file data is invalid or the callback rejects a count.
6032    pub fn integer_fold(
6033        &self,
6034        name: &str,
6035        column: usize,
6036        mut emit: impl FnMut(i64, u64) -> Result<()>,
6037    ) -> Result<Option<()>> {
6038        let entry = self
6039            .entries
6040            .iter()
6041            .find(|entry| entry.name == name)
6042            .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
6043        let field =
6044            entry.fields.get(column).ok_or_else(|| invalid("integer column index out of range"))?;
6045        if !signed_integer(&field.ty) {
6046            return Ok(None);
6047        }
6048        let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
6049        if file_checksum(&self.file, offset, length)? != entry.directory.hash {
6050            return Err(invalid(&format!("the directory of table {name} does not checksum")));
6051        }
6052        quick_integer_fold(
6053            &self.file,
6054            Cursor::over(&self.file, offset, length),
6055            entry,
6056            self.size,
6057            column,
6058            &mut emit,
6059        )?;
6060        Ok(Some(()))
6061    }
6062
6063    /// Counts non-null, nonzero values from generic column frequencies when complete. For an
6064    /// older file or a partial catalog synopsis, reads the validated native directory without
6065    /// building a reader for every stripe. Returns `None` when the bounded frequency synopsis
6066    /// cannot prove the count, so callers can use the ordinary query path.
6067    pub fn nonzero_count(&self, name: &str, column: usize) -> Result<Option<u64>> {
6068        let entry = self
6069            .entries
6070            .iter()
6071            .find(|entry| entry.name == name)
6072            .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
6073        let Some(field) = entry.fields.get(column) else {
6074            return Err(invalid("frequency column index out of range"));
6075        };
6076        if !matches!(
6077            field.ty,
6078            LogicalType::TinyInt
6079                | LogicalType::SmallInt
6080                | LogicalType::Integer
6081                | LogicalType::BigInt
6082                | LogicalType::UTinyInt
6083                | LogicalType::USmallInt
6084                | LogicalType::UInteger
6085                | LogicalType::UBigInt
6086        ) {
6087            return Ok(None);
6088        }
6089        let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
6090        if file_checksum(&self.file, offset, length)? != entry.directory.hash {
6091            return Err(invalid(&format!("the directory of table {name} does not checksum")));
6092        }
6093        if let Some(Some(frequencies)) = entry.frequencies.get(column) {
6094            return frequencies
6095                .iter()
6096                .filter(|(value, _)| value.is_some_and(|value| value != 0))
6097                .try_fold(0_u64, |total, (_, count)| total.checked_add(*count))
6098                .map(Some)
6099                .ok_or_else(|| invalid("numeric frequency count overflow"));
6100        }
6101        quick_nonzero(
6102            Cursor::over(&self.file, offset, length),
6103            &entry.name,
6104            &entry.fields,
6105            entry.rows,
6106            column,
6107        )
6108    }
6109
6110    /// Exact signed-integer sums and non-null counts from the small catalog. The table directory
6111    /// checksum is still checked once before any certificate can answer a query.
6112    pub fn aggregate_sums(&self, name: &str, columns: &[usize]) -> Result<Option<CertifiedSums>> {
6113        let entry = self
6114            .entries
6115            .iter()
6116            .find(|entry| entry.name == name)
6117            .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
6118        let mut sums = Vec::with_capacity(columns.len());
6119        for &column in columns {
6120            let Some(field) = entry.fields.get(column) else {
6121                return Err(invalid("aggregate column index out of range"));
6122            };
6123            if !signed_integer(&field.ty) {
6124                return Ok(None);
6125            }
6126            let Some(sum) = entry.aggregates[column] else {
6127                return Ok(None);
6128            };
6129            sums.push(sum);
6130        }
6131        let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
6132        if file_checksum(&self.file, offset, length)? != entry.directory.hash {
6133            return Err(invalid(&format!("the directory of table {name} does not checksum")));
6134        }
6135        Ok(Some(CertifiedSums { columns: sums, rows: entry.rows as u64 }))
6136    }
6137
6138    /// Exact non-null distinct count from the small catalog, after checking the table directory.
6139    pub fn distinct_count(&self, name: &str, column: usize) -> Result<Option<u64>> {
6140        let entry = self
6141            .entries
6142            .iter()
6143            .find(|entry| entry.name == name)
6144            .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
6145        let Some(count) = entry.distincts.get(column).copied() else {
6146            return Err(invalid("distinct column index out of range"));
6147        };
6148        let Some(count) = count else { return Ok(None) };
6149        let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
6150        if file_checksum(&self.file, offset, length)? != entry.directory.hash {
6151            return Err(invalid(&format!("the directory of table {name} does not checksum")));
6152        }
6153        Ok(Some(count))
6154    }
6155
6156    /// Exact integer or date ends from the small catalog after checking the table directory.
6157    pub fn integer_extremes(&self, name: &str, column: usize) -> Result<Option<IntegerExtremes>> {
6158        let entry = self
6159            .entries
6160            .iter()
6161            .find(|entry| entry.name == name)
6162            .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
6163        let Some(extremes) = entry.extremes.get(column).copied() else {
6164            return Err(invalid("extremes column index out of range"));
6165        };
6166        let Some(extremes) = extremes else { return Ok(None) };
6167        let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
6168        if file_checksum(&self.file, offset, length)? != entry.directory.hash {
6169            return Err(invalid(&format!("the directory of table {name} does not checksum")));
6170        }
6171        Ok(Some(match extremes {
6172            None => IntegerExtremes::Null,
6173            Some((low, high)) => IntegerExtremes::Values { low, high },
6174        }))
6175    }
6176
6177    /// Complete numeric frequencies from the small catalog, after checking the table directory.
6178    pub fn exact_numeric_frequencies(
6179        &self,
6180        name: &str,
6181        column: usize,
6182    ) -> Result<Option<NumericFrequencies>> {
6183        let entry = self
6184            .entries
6185            .iter()
6186            .find(|entry| entry.name == name)
6187            .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
6188        let Some(frequencies) = entry.frequencies.get(column).cloned() else {
6189            return Err(invalid("numeric frequency column index out of range"));
6190        };
6191        let Some(frequencies) = frequencies else { return Ok(None) };
6192        let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
6193        if file_checksum(&self.file, offset, length)? != entry.directory.hash {
6194            return Err(invalid(&format!("the directory of table {name} does not checksum")));
6195        }
6196        Ok(Some(frequencies))
6197    }
6198
6199    /// The schema copied into the small file catalog, available without opening the table directory.
6200    pub fn table_fields(&self, name: &str) -> Option<&[Field]> {
6201        self.entries.iter().find(|entry| entry.name == name).map(|entry| entry.fields.as_slice())
6202    }
6203}
6204
6205/// Where the slot naming `generation` goes, which is the one the generation before it did not use.
6206///
6207/// Generation 1 takes the slot at 16, so a file written once is byte for byte the file this wrote
6208/// before there was a second generation to write.
6209fn slot_offset(generation: u64) -> u64 {
6210    16 + (generation - 1) % 2 * SLOT_BYTES as u64
6211}
6212
6213/// The header and the bytes the highest valid slot points at.
6214///
6215/// Both levels of the directory are reached this way, so the magic check, the version check and the
6216/// choice between the two slots live here rather than being written out twice.
6217fn slot_bytes(path: impl AsRef<Path>) -> Result<(File, u64, Slot, Vec<u8>, Opening)> {
6218    let file = File::open(path).map_err(io)?;
6219    let size = file.metadata().map_err(io)?.len();
6220    let (slot, bytes, opening) = committed_slot(&file, size)?;
6221    Ok((file, size, slot, bytes, opening))
6222}
6223
6224/// The committed slot of a file that is `size` bytes long, and the catalog it points at.
6225///
6226/// The half of [`slot_bytes`] that does not care how the file was opened. A reader comes here with
6227/// the `std::fs::File` it goes on to share between its threads, and a writer with the `rudb_io`
6228/// file it is about to append to.
6229fn committed_slot<F: Positional + ?Sized>(file: &F, size: u64) -> Result<(Slot, Vec<u8>, Opening)> {
6230    if size < HEADER {
6231        return Err(invalid("file is shorter than its header"));
6232    }
6233    let mut header = [0; HEADER as usize];
6234    read_at(file, 0, &mut header)?;
6235    let mut opening = Opening { reads: 1, bytes: HEADER };
6236    let version = u32::from_le_bytes([header[8], header[9], header[10], header[11]]);
6237    // The two halves are worth telling apart. A wrong magic is a file that was never ours and
6238    // the answer is to look at the path. A wrong version is our own file from another build,
6239    // and the number this build wants is the only thing that tells the reader whether to
6240    // rebuild the file or to go back to the binary that wrote it.
6241    if &header[..8] != MAGIC {
6242        return Err(invalid("the header does not begin with a rudb native magic"));
6243    }
6244    if !READABLE.contains(&version) {
6245        return Err(invalid(&format!(
6246            "the file is format {version} and this build reads format {FORMAT}, so it has to \
6247                 be written again"
6248        )));
6249    }
6250    let mut selected = None;
6251    for start in [16, 16 + SLOT_BYTES] {
6252        let slot = Slot::read(&header[start..start + SLOT_BYTES]);
6253        if slot.generation == 0 || slot.length == 0 || slot.length as usize > MAX_DIRECTORY {
6254            continue;
6255        }
6256        let Some(end) = slot.offset.checked_add(u64::from(slot.length)) else { continue };
6257        if slot.offset < HEADER || end > size {
6258            continue;
6259        }
6260        let mut bytes = vec![0; slot.length as usize];
6261        read_at(file, slot.offset, &mut bytes)?;
6262        opening.reads += 1;
6263        opening.bytes += u64::from(slot.length);
6264        if checksum(&bytes) == slot.hash
6265            && selected
6266                .as_ref()
6267                .is_none_or(|(old, _): &(Slot, Vec<u8>)| old.generation < slot.generation)
6268        {
6269            selected = Some((slot, bytes));
6270        }
6271    }
6272    let (slot, bytes) = selected.ok_or_else(|| invalid("no committed directory slot is valid"))?;
6273    Ok((slot, bytes, opening))
6274}
6275
6276impl Reader {
6277    /// Opens a file that holds exactly one table.
6278    ///
6279    /// # Errors
6280    ///
6281    /// If the file has no valid committed directory, a directory pointer is out of bounds, or the
6282    /// file holds more than one table, which is a file that has to be opened by name.
6283    pub fn open(path: impl AsRef<Path>) -> Result<Self> {
6284        let catalog = Catalog::open(path)?;
6285        let mut names = catalog.names();
6286        let name = names.next().ok_or_else(|| invalid("the file holds no table"))?.to_string();
6287        if names.next().is_some() {
6288            return Err(invalid(
6289                "the file holds more than one table, so it has to be opened by name",
6290            ));
6291        }
6292        catalog.table(&name)
6293    }
6294
6295    /// Builds a reader over one decoded table directory.
6296    fn build(
6297        file: Arc<File>,
6298        size: u64,
6299        table: Table,
6300        directory: u64,
6301        opening: Opening,
6302        pool: PagePool,
6303    ) -> Result<Self> {
6304        let places = places(&table)?;
6305        let dictionaries = (0..table.fields.len()).map(|_| OnceLock::new()).collect();
6306        let table_fields = table.fields.len();
6307        let stripes = table.stripes.len();
6308        let columns = (0..table.fields.len())
6309            .map(|_| {
6310                Mutex::new(Cached {
6311                    pages: (0..stripes).map(|_| None).collect(),
6312                    index: (0..stripes).map(|_| None).collect(),
6313                    touched: vec![Vec::new(); stripes],
6314                    ..Cached::default()
6315                })
6316            })
6317            .collect::<Vec<_>>();
6318        let cache = Shelf {
6319            columns,
6320            held: (0..table_fields).map(|_| AtomicUsize::new(0)).collect(),
6321            kept: AtomicUsize::new(CACHED_STRIPES_PER_COLUMN),
6322        };
6323        let sieves: Vec<Vec<SieveSlot>> = (0..table.fields.len())
6324            .map(|_| table.stripes.iter().map(|_| OnceLock::new()).collect())
6325            .collect();
6326        let part_ranges: Vec<Vec<RangeSlot>> = (0..table.fields.len())
6327            .map(|_| table.stripes.iter().map(|_| OnceLock::new()).collect())
6328            .collect();
6329        let verified = (places.len() * table_fields).div_ceil(64);
6330        let firsts = places
6331            .iter()
6332            .scan(0, |first, place| {
6333                let at = *first;
6334                *first += place.rows as usize;
6335                Some(at)
6336            })
6337            .collect();
6338        Ok(Self {
6339            file,
6340            table: Arc::new(table),
6341            dictionaries: Arc::new(dictionaries),
6342            loading: Arc::new((0..table_fields).map(|_| Mutex::new(())).collect()),
6343            frequency_values: Arc::new((0..table_fields).map(|_| OnceLock::new()).collect()),
6344            frequency_summaries: Arc::new((0..table_fields).map(|_| OnceLock::new()).collect()),
6345            summaries: Arc::new((0..table_fields).map(|_| OnceLock::new()).collect()),
6346            opened: Arc::new(AtomicUsize::new(0)),
6347            sieves: Arc::new(sieves),
6348            part_ranges: Arc::new(part_ranges),
6349            places: Arc::new(places),
6350            cache: Arc::new(cache),
6351            pool,
6352            pages: Arc::new(AtomicUsize::new(0)),
6353            indexes: Arc::new(AtomicUsize::new(0)),
6354            verified: Arc::new((0..verified).map(|_| AtomicU64::new(0)).collect()),
6355            text_grams: Arc::new((0..table_fields).map(|_| OnceLock::new()).collect()),
6356            firsts: Arc::new(firsts),
6357            size,
6358            directory,
6359            opening,
6360        })
6361    }
6362
6363    /// What this reader has read so far, and what opening it cost.
6364    ///
6365    /// Public because the claim of `spec/stats/04-in-memory.md` section 4.2 is about this number
6366    /// and a claim nobody can check is a comment. A caller that wants to know whether opening a
6367    /// file touched the data asks here, and gets an answer that does not depend on what the page
6368    /// cache happened to hold.
6369    #[must_use]
6370    pub fn reads(&self) -> Reads {
6371        Reads {
6372            opening: self.opening,
6373            pages: self.pages.load(Atomic::Relaxed),
6374            indexes: self.indexes.load(Atomic::Relaxed),
6375            dictionaries: self.opened.load(Atomic::Relaxed),
6376        }
6377    }
6378
6379    /// Where the file's bytes went, from the directory alone.
6380    ///
6381    /// No page is read, so this costs the same on a 45 GB table as on an empty one. See [`Layout`]
6382    /// for what is charged where and for why the three things that are not columns stay separate.
6383    #[must_use]
6384    pub fn layout(&self) -> Layout {
6385        let table = &self.table;
6386        let stripes = table.stripes.as_slice();
6387        let columns = table
6388            .fields
6389            .iter()
6390            .enumerate()
6391            .map(|(at, field)| ColumnLayout {
6392                name: field.name.clone(),
6393                kind: field.ty.to_string(),
6394                pages: sum(stripes.iter().map(|stripe| span_bytes(&stripe.pages, at))),
6395                memberships: sum(stripes.iter().map(|stripe| stripe.memberships.bytes(at))),
6396                sieves: sum(stripes.iter().map(|stripe| stripe.sieves.bytes(at))),
6397                part_ranges: sum(stripes.iter().map(|stripe| stripe.part_ranges.bytes(at))),
6398                dictionary: dictionary_bytes(table, at),
6399            })
6400            .collect();
6401        Layout {
6402            file: self.size,
6403            rows: table.rows,
6404            stripes: stripes.len(),
6405            parts: self.places.len(),
6406            columns,
6407            indexes: sum(stripes.iter().map(|stripe| u64::from(stripe.index.length))),
6408            directory: self.directory,
6409            header: HEADER,
6410        }
6411    }
6412
6413    /// What every part of one column is stored as, which is what `pragma_storage_info` reports.
6414    ///
6415    /// Unlike [`Self::layout`] this reads the data, because the encoder's choice is in the page and
6416    /// nowhere else. The directory says how many bytes a column took and says nothing about what
6417    /// shape they are in, and the shape is the question worth asking: the same rows in a different
6418    /// order come back bit packed on one file and plain on another, and that is the difference a
6419    /// clustered load makes to a scan.
6420    ///
6421    /// One read per stripe rather than one per part. A part is a few kilobytes out of a page that
6422    /// is a quarter of a megabyte, so asking part by part would read the same page sixty four
6423    /// times. Nothing is put in the page cache, because a caller asking what a file looks like is
6424    /// not about to scan it and evicting the pages a real query wants would be a poor trade.
6425    ///
6426    /// # Errors
6427    ///
6428    /// If the column is outside the schema, or a page, index section or checksum is invalid.
6429    pub fn stored(&self, column: usize) -> Result<Vec<StoredPart>> {
6430        let field = self
6431            .table
6432            .fields
6433            .get(column)
6434            .ok_or_else(|| invalid("stored column index out of range"))?;
6435        let mut stored = Vec::with_capacity(self.places.len());
6436        let mut row = 0;
6437        for (at, stripe) in self.table.stripes.iter().enumerate() {
6438            let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
6439            let index = read_index(&self.file, stripe, column)?;
6440            let mut bytes = vec![0; page.length as usize];
6441            read_at(&self.file, page.offset, &mut bytes)?;
6442            let ranges = self.stripe_part_ranges(at, column);
6443            for (part, &rows) in stripe.parts.iter().enumerate() {
6444                let span = *index.get(part).ok_or_else(|| invalid("part index out of range"))?;
6445                let held = part_bytes(&bytes, span)?;
6446                let range = ranges.and_then(|held| held.get(part));
6447                stored.push(StoredPart {
6448                    stripe: at,
6449                    part,
6450                    row,
6451                    rows: rows as usize,
6452                    encoding: page_encoding(&field.ty, rows as usize, held),
6453                    bytes: span.length as u64,
6454                    page: page.offset,
6455                    offset: span.start as u64,
6456                    low: range
6457                        .and_then(|range| range.low.clone())
6458                        .and_then(|bound| bound.into_value(&field.ty)),
6459                    high: range
6460                        .and_then(|range| range.high.clone())
6461                        .and_then(|bound| bound.into_value(&field.ty)),
6462                    nulls: range.map(|range| range.nulls),
6463                });
6464                row += rows as usize;
6465            }
6466        }
6467        Ok(stored)
6468    }
6469
6470    /// How many parts the table has, which is how many chunks a scan of it reads.
6471    #[must_use]
6472    pub fn parts(&self) -> usize {
6473        self.places.len()
6474    }
6475
6476    /// The parts of each stripe, in table wide part numbers.
6477    ///
6478    /// A scan that wants one worker to own the page it reads hands work out in these runs. The
6479    /// stripes are contiguous in part numbering and all but the last hold sixty four parts, but a
6480    /// stripe can be flushed early when rows arrive out of order, so the runs are read off the
6481    /// directory rather than worked out from a constant.
6482    #[must_use]
6483    pub fn stripe_parts(&self) -> Vec<std::ops::Range<usize>> {
6484        let mut runs = Vec::with_capacity(self.table.stripes.len());
6485        let mut start = 0;
6486        for stripe in &self.table.stripes {
6487            let end = start + stripe.parts.len();
6488            runs.push(start..end);
6489            start = end;
6490        }
6491        runs
6492    }
6493
6494    /// How many rows one stripe holds, in the numbering [`Self::stripe_parts`] hands back.
6495    ///
6496    /// Off the directory, which is already in memory, rather than by the caller asking for each
6497    /// part in turn through the catalog. Nothing past the end holds any rows.
6498    #[must_use]
6499    pub fn stripe_rows(&self, stripe: usize) -> usize {
6500        self.table.stripes.get(stripe).map_or(0, |held| held.rows)
6501    }
6502
6503    /// Asks the page cache to keep `stripes` stripes of every column instead of the default.
6504    ///
6505    /// This only ever raises the number. A scan that gives each worker a whole stripe has one page
6506    /// per column per worker open at once, and a cache smaller than that is worse than no cache at
6507    /// all: every worker's page is evicted by the others before it has finished its stripe, so it
6508    /// reads a quarter of a megabyte for every part it takes out of it.
6509    pub fn keep_stripes(&self, stripes: usize) {
6510        self.cache.kept.fetch_max(stripes, Atomic::Relaxed);
6511    }
6512
6513    /// Rows in one part, or zero when the part number is past the table.
6514    #[must_use]
6515    pub fn part_rows(&self, at: usize) -> usize {
6516        self.places.get(at).map_or(0, |place| place.rows as usize)
6517    }
6518
6519    /// The committed table directory.
6520    #[must_use]
6521    pub fn table(&self) -> &Table {
6522        &self.table
6523    }
6524
6525    /// Exact leading frequencies when the stored synopsis proves a count-descending prefix.
6526    ///
6527    /// The returned list can be longer than `top`. Keeping the stored tail lets a later TopN apply
6528    /// additional ordering keys without losing a value tied with the requested boundary.
6529    ///
6530    /// # Errors
6531    ///
6532    /// If the column is outside the schema or a stored value does not fit its declared type.
6533    pub fn top_frequencies(&self, column: usize, top: usize) -> Result<Option<Vec<(Value, u64)>>> {
6534        let field = self
6535            .table
6536            .fields
6537            .get(column)
6538            .ok_or_else(|| invalid("frequency column index out of range"))?;
6539        let Some(summary) = self.frequency_summary(column)? else {
6540            return Ok(None);
6541        };
6542        if top == 0 || summary.entries.len() < top {
6543            return Ok(None);
6544        }
6545        let boundary = summary.entries[top - 1].count;
6546        if boundary <= summary.omitted_max {
6547            return Ok(None);
6548        }
6549        self.decode_frequencies(column, &field.ty, &summary.entries)
6550            .map(|values| Some(Vec::clone(&values)))
6551    }
6552
6553    /// Exact leading counts for a numeric key paired with a stable-dictionary string key.
6554    ///
6555    /// Legacy pair summaries are parsed for file compatibility but never used as query output.
6556    ///
6557    /// # Errors
6558    ///
6559    /// If either column is outside the schema.
6560    pub fn top_pair_frequencies(
6561        &self,
6562        first: usize,
6563        second: usize,
6564        _top: usize,
6565    ) -> Result<Option<PairFrequencyCounts>> {
6566        if first >= self.table.fields.len() || second >= self.table.fields.len() {
6567            return Err(invalid("pair frequency column index out of range"));
6568        }
6569        Ok(None)
6570    }
6571
6572    /// Every value of one column with the number of rows holding it, when the synopsis is complete.
6573    ///
6574    /// The heavy hitter pass keeps a bounded set of candidates and decrements them all when it runs
6575    /// out of room, so what it usually ends with is the leading values and a bound on everything it
6576    /// dropped. `omitted_max` of zero says that never happened: no candidate was ever decremented and
6577    /// the entries did not overflow the stored budget, so the list is every distinct value of the
6578    /// column with an exact count, and a null counts as a value of its own rather than being skipped.
6579    ///
6580    /// That makes a whole class of question answerable without reading a row. How many rows hold a
6581    /// value, how many do not, and what a `GROUP BY` of that column with a count over it produces are
6582    /// all in here. It is only ever true of a column with few enough distinct values, which is the
6583    /// case worth having, because that is exactly the column a grouping or an equality filter would
6584    /// otherwise walk every row to answer.
6585    ///
6586    /// `None` when the column has no synopsis, or has one that dropped anything.
6587    ///
6588    /// # Errors
6589    ///
6590    /// If the column is outside the schema or a stored value does not fit its declared type.
6591    pub fn exact_frequencies(&self, column: usize) -> Result<Option<Vec<(Value, u64)>>> {
6592        let Some(prefix) = self.frequency_prefix(column)? else {
6593            return Ok(None);
6594        };
6595        Ok((prefix.omitted_max == 0).then_some(prefix.entries))
6596    }
6597
6598    /// Every value the synopsis lists with the number of rows holding it, and a bound on the rest.
6599    ///
6600    /// The counts are exact whether or not the list is complete. The heavy hitter pass keeps a
6601    /// bounded candidate set and then recounts only the candidates that survived it, so a value that
6602    /// made it into the list carries the number of rows that really hold it rather than whatever the
6603    /// pass had left over. What the pass loses is values, not counts.
6604    ///
6605    /// `omitted_max` is how many rows the most common value left out can hold, and zero says nothing
6606    /// was left out at all, which is what [`exact_frequencies`] asks for. Above zero the list is the
6607    /// leading values of the column and everything else is somewhere between no rows and that bound.
6608    ///
6609    /// That prefix is worth reading on its own. A column with a value in half its rows and a long
6610    /// tail behind it has no complete synopsis and never will, and it is the column where dividing
6611    /// the rows by the distinct count is furthest from the truth.
6612    ///
6613    /// `None` when the column has no synopsis.
6614    ///
6615    /// # Errors
6616    ///
6617    /// If the column is outside the schema or a stored value does not fit its declared type.
6618    ///
6619    /// [`exact_frequencies`]: Self::exact_frequencies
6620    pub fn frequency_prefix(&self, column: usize) -> Result<Option<FrequencyPrefix>> {
6621        Ok(self.held_prefix(column)?.map(|(entries, omitted_max)| FrequencyPrefix {
6622            entries: Vec::clone(&entries),
6623            omitted_max,
6624        }))
6625    }
6626
6627    /// [`Self::frequency_prefix`] as the reader holds it, shared rather than copied.
6628    ///
6629    /// The planner asks for a column's synopsis for every estimate that touches it, on every
6630    /// statement, and the list of a string column is a few hundred strings, so copying it each time
6631    /// was a hundred allocations for an answer nothing changes.
6632    pub(crate) fn held_prefix(&self, column: usize) -> Result<Option<(Synopsis, u64)>> {
6633        let field = self
6634            .table
6635            .fields
6636            .get(column)
6637            .ok_or_else(|| invalid("frequency column index out of range"))?;
6638        let Some(summary) = self.frequency_summary(column)? else {
6639            return Ok(None);
6640        };
6641        let entries = self.decode_frequencies(column, &field.ty, &summary.entries)?;
6642        Ok(Some((entries, summary.omitted_max)))
6643    }
6644
6645    /// One column's synopsis, read back from the file when the directory left it there.
6646    fn frequency_summary(&self, column: usize) -> Result<Option<Cow<'_, FrequencySummary>>> {
6647        Ok(match self.table.frequencies.get(column) {
6648            None | Some(None) => None,
6649            Some(Some(Frequencies::Held(summary))) => Some(Cow::Borrowed(summary)),
6650            Some(Some(Frequencies::Stored { span, values, entries })) => {
6651                let slot = self
6652                    .frequency_summaries
6653                    .get(column)
6654                    .ok_or_else(|| invalid("frequency column index out of range"))?;
6655                if let Some(summary) = slot.get() {
6656                    return Ok(Some(Cow::Borrowed(summary.as_ref())));
6657                }
6658                let field = self
6659                    .table
6660                    .fields
6661                    .get(column)
6662                    .ok_or_else(|| invalid("frequency column index out of range"))?;
6663                let mut bytes = vec![0; span.length as usize];
6664                read_at(&self.file, span.offset, &mut bytes)?;
6665                let mut cur = Cursor::new(&bytes);
6666                let summary = decode_summary(&mut cur, field, self.table.rows, *values)?;
6667                let summary = summary.ok_or_else(|| invalid("a stored synopsis is missing"))?;
6668                if !cur.done() || summary.entries.len() != *entries {
6669                    return Err(invalid("a stored synopsis differs from its directory span"));
6670                }
6671                let _ = slot.set(Arc::new(summary));
6672                Some(Cow::Borrowed(slot.get().expect("the decoded summary was stored").as_ref()))
6673            }
6674        })
6675    }
6676
6677    /// Turns stored frequency entries into values of the column's own type.
6678    ///
6679    /// Remembered per column, because the planner asks once for every estimate that touches the
6680    /// column and the executor asks again, and the answer is a few hundred values. The codes of a
6681    /// string column are read through [`Vector::try_values_visited`], which does not keep the blocks
6682    /// it decodes, so what a query answered out of the synopsis holds is those values and not the
6683    /// hundred or so dictionary blocks they are scattered over.
6684    fn decode_frequencies(
6685        &self,
6686        column: usize,
6687        ty: &LogicalType,
6688        entries: &[FrequencyEntry],
6689    ) -> Result<Synopsis> {
6690        if let Some(values) = self.frequency_values.get(column).and_then(OnceLock::get) {
6691            return Ok(Arc::clone(values));
6692        }
6693        let values = Arc::new(self.decode_frequencies_once(column, ty, entries)?);
6694        if let Some(slot) = self.frequency_values.get(column) {
6695            let _ = slot.set(Arc::clone(&values));
6696        }
6697        Ok(values)
6698    }
6699
6700    fn decode_frequencies_once(
6701        &self,
6702        column: usize,
6703        ty: &LogicalType,
6704        entries: &[FrequencyEntry],
6705    ) -> Result<Vec<(Value, u64)>> {
6706        let stored_texts = self.table.frequency_texts.get(column).filter(|texts| !texts.is_empty());
6707        if stored_texts.is_some_and(|texts| texts.len() != entries.len()) {
6708            return Err(invalid("frequency text count differs from its synopsis"));
6709        }
6710        let dictionary =
6711            if coded_type(ty) && stored_texts.is_none() { self.dictionary(column)? } else { None };
6712        let mut codes = entries
6713            .iter()
6714            .filter_map(|entry| match entry.value {
6715                FrequencyValue::Code(code) => Some(code as usize),
6716                _ => None,
6717            })
6718            .collect::<Vec<_>>();
6719        codes.sort_unstable();
6720        codes.dedup();
6721        let texts = match &dictionary {
6722            Some(dictionary) if !codes.is_empty() => dictionary.try_values_visited(&codes)?,
6723            _ => Vec::new(),
6724        };
6725        let mut out = Vec::with_capacity(entries.len());
6726        for (entry_at, entry) in entries.iter().enumerate() {
6727            let value = match entry.value {
6728                FrequencyValue::Null => {
6729                    if stored_texts.and_then(|texts| texts[entry_at].as_ref()).is_some() {
6730                        return Err(invalid("a null frequency entry has text"));
6731                    }
6732                    Value::Null
6733                }
6734                FrequencyValue::Integer(value) => match *ty {
6735                    LogicalType::TinyInt => Value::TinyInt(
6736                        i8::try_from(value)
6737                            .map_err(|_| invalid("frequency TINYINT is out of range"))?,
6738                    ),
6739                    LogicalType::UTinyInt => Value::UTinyInt(
6740                        u8::try_from(value)
6741                            .map_err(|_| invalid("frequency UTINYINT is out of range"))?,
6742                    ),
6743                    LogicalType::USmallInt => Value::USmallInt(
6744                        u16::try_from(value)
6745                            .map_err(|_| invalid("frequency USMALLINT is out of range"))?,
6746                    ),
6747                    LogicalType::UInteger => Value::UInteger(
6748                        u32::try_from(value)
6749                            .map_err(|_| invalid("frequency UINTEGER is out of range"))?,
6750                    ),
6751                    LogicalType::UBigInt => Value::UBigInt(
6752                        u64::try_from(value)
6753                            .map_err(|_| invalid("frequency UBIGINT is out of range"))?,
6754                    ),
6755                    LogicalType::SmallInt => Value::SmallInt(
6756                        i16::try_from(value)
6757                            .map_err(|_| invalid("frequency SMALLINT is out of range"))?,
6758                    ),
6759                    LogicalType::Integer => Value::Integer(
6760                        i32::try_from(value)
6761                            .map_err(|_| invalid("frequency INTEGER is out of range"))?,
6762                    ),
6763                    LogicalType::BigInt => Value::BigInt(
6764                        i64::try_from(value)
6765                            .map_err(|_| invalid("frequency BIGINT is out of range"))?,
6766                    ),
6767                    LogicalType::Date => Value::Date(
6768                        i32::try_from(value)
6769                            .map_err(|_| invalid("frequency DATE is out of range"))?,
6770                    ),
6771                    LogicalType::Timestamp => Value::Timestamp(
6772                        i64::try_from(value)
6773                            .map_err(|_| invalid("frequency TIMESTAMP is out of range"))?,
6774                    ),
6775                    _ => return Err(invalid("integer frequency belongs to another type")),
6776                },
6777                FrequencyValue::Code(code) => {
6778                    if let Some(text) = stored_texts.and_then(|texts| texts[entry_at].as_ref()) {
6779                        if *ty == LogicalType::Blob {
6780                            Value::Blob(text.clone())
6781                        } else {
6782                            Value::Varchar(
6783                                String::from_utf8(text.clone())
6784                                    .map_err(|_| invalid("frequency text is not UTF-8"))?,
6785                            )
6786                        }
6787                    } else {
6788                        if dictionary.is_none() {
6789                            return Err(invalid("frequency code has no dictionary or stored text"));
6790                        }
6791                        let at = codes
6792                            .binary_search(&(code as usize))
6793                            .map_err(|_| invalid("frequency code was not among the codes read"))?;
6794                        texts[at].clone()
6795                    }
6796                }
6797            };
6798            out.push((value, entry.count));
6799        }
6800        Ok(out)
6801    }
6802
6803    /// Sparse rows belonging to the bounded numeric frequency candidate set.
6804    ///
6805    /// The list is omitted when collecting it would exceed the fixed storage budget. A composite
6806    /// aggregate may accept a result over these rows only when its requested boundary is strictly
6807    /// greater than `omitted_max`.
6808    ///
6809    /// # Errors
6810    ///
6811    /// If the column is outside the schema.
6812    pub fn frequency_occurrences(&self, column: usize) -> Result<Option<FrequencyOccurrences>> {
6813        let field = self
6814            .table
6815            .fields
6816            .get(column)
6817            .ok_or_else(|| invalid("frequency column index out of range"))?;
6818        let Some(summary) = self.frequency_summary(column)? else {
6819            return Ok(None);
6820        };
6821        if summary.ordinals.is_empty() {
6822            return Ok(None);
6823        }
6824        let (anchors, anchor_indices) = if summary.ordinal_entries.len() == summary.ordinals.len() {
6825            let entries = self.decode_frequencies(column, &field.ty, &summary.entries)?;
6826            (
6827                entries.iter().map(|(value, _)| value.clone()).collect(),
6828                summary.ordinal_entries.clone(),
6829            )
6830        } else {
6831            (Vec::new(), Vec::new())
6832        };
6833        Ok(Some(FrequencyOccurrences {
6834            omitted_max: summary.omitted_max,
6835            ordinals: summary.ordinals.clone(),
6836            anchors,
6837            anchor_indices,
6838        }))
6839    }
6840
6841    /// How many distinct values one column holds, counting a null as no value.
6842    ///
6843    /// A string column of this format is written against one dictionary that covers the whole table.
6844    /// A code is handed out the first time a value is seen and nothing ever removes one, so the
6845    /// number of codes is the number of distinct values exactly rather than an estimate. That makes
6846    /// `COUNT(DISTINCT column)` over a whole table a question the directory already knows the answer
6847    /// to, and the alternative is a hash table with a row per distinct value built from a pass over
6848    /// every row.
6849    ///
6850    /// A null in the column used to make this `None` and no longer does. A null row is written as
6851    /// the code for the empty string, so a nullable column's dictionary can hold an empty string
6852    /// that no row of it actually has, and the dictionary on its own does not say which case it is.
6853    /// The writer does know, because it counts the non-null rows that use each code on its way to
6854    /// the frequency summary, so it records how many codes any row holds and the directory carries
6855    /// that number. This reads it rather than the size of the dictionary, which also means the
6856    /// dictionary page is not opened to answer.
6857    ///
6858    /// An integer column has no dictionary, and its count comes from the set the writer keeps on its
6859    /// numeric frequency pass instead, which is exact up to a cap. `None` for a column past that cap
6860    /// and for every column that is neither, where a sketch would answer approximately and SQL asked
6861    /// for the exact number.
6862    ///
6863    /// # Errors
6864    ///
6865    /// If the column is outside the schema.
6866    pub fn distinct_values(&self, column: usize) -> Result<Option<u64>> {
6867        self.table
6868            .distincts
6869            .get(column)
6870            .copied()
6871            .ok_or_else(|| invalid("distinct column index out of range"))
6872    }
6873
6874    /// How many rows of one column are null, added up over the stripes.
6875    ///
6876    /// Every stripe records this exactly when it is written, because a null count is not a bound
6877    /// that is allowed to be wide the way a minimum and a maximum are: a filter that reads one too
6878    /// many is slow and a `COUNT` that reads one too many is wrong. Adding up a few hundred numbers
6879    /// already in memory is what makes `COUNT(column)` over a whole table free.
6880    ///
6881    /// # Errors
6882    ///
6883    /// If the column is outside the schema.
6884    pub fn null_count(&self, column: usize) -> Result<u64> {
6885        if column >= self.table.fields.len() {
6886            return Err(invalid("null count column index out of range"));
6887        }
6888        let mut nulls = 0_u64;
6889        for stripe in &self.table.stripes {
6890            let range = stripe
6891                .zone
6892                .column(column)
6893                .ok_or_else(|| invalid("stripe zone is narrower than the schema"))?;
6894            nulls = nulls
6895                .checked_add(range.nulls as u64)
6896                .ok_or_else(|| invalid("null count overflow"))?;
6897        }
6898        Ok(nulls)
6899    }
6900
6901    /// The smallest and the largest value of one string column, from the order beside its values.
6902    ///
6903    /// The dictionary holds exactly the values the column holds, so the first and the last of them
6904    /// in sorted order are the column's minimum and maximum. Two reads of a rank block settle what
6905    /// otherwise walks a million rows.
6906    ///
6907    /// `None` when the column is not a string, when the file was written before version 9 and so has
6908    /// no order, when the column has no values at all, or when it has a null in it, which is the
6909    /// placeholder again: the empty string a null is written as would sort ahead of every real
6910    /// value and be reported as the minimum.
6911    ///
6912    /// # Errors
6913    ///
6914    /// If the column is outside the schema, or a rank names a code the dictionary does not have.
6915    pub fn text_extremes(&self, column: usize) -> Result<Option<(Value, Value)>> {
6916        if self.null_count(column)? > 0 || self.demoted(column) {
6917            return Ok(None);
6918        }
6919        let Some(dictionary) = self.dictionary(column)? else { return Ok(None) };
6920        let Some(ranks) = dictionary.ranks() else { return Ok(None) };
6921        if ranks == 0 {
6922            return Ok(None);
6923        }
6924        let low = text_at_rank(&dictionary, 0)?;
6925        let high = text_at_rank(&dictionary, ranks - 1)?;
6926        Ok(Some((low, high)))
6927    }
6928
6929    /// The smallest and the largest value of one column, when every stripe wrote exact ends.
6930    ///
6931    /// A stripe's ends are allowed to be wider than the truth, because a bound that rules out a
6932    /// chunk that could not match is still correct when it rules out nothing. That is what makes
6933    /// them cheap to write for a bit packed or a dictionary column, and it is also what stops them
6934    /// answering a `MIN`. So each stripe says which of the two it wrote, and this answers only when
6935    /// all of them walked their rows.
6936    ///
6937    /// `None` for a column with no ends, for an empty table, and for a column any stripe of which
6938    /// guessed. Nulls need no special case, because the ends skip them the same way `MIN` does.
6939    ///
6940    /// One case is given up on that did not have to be. A stripe merges the ends of its sixty four
6941    /// parts, and a part with no ends at all erases the merged ones, because a part whose rows are
6942    /// not covered by the stripe's ends is a stripe that would skip rows it should keep. A part of
6943    /// nothing but nulls has no rows to cover and so did not need to erase anything, but the merge
6944    /// cannot tell that part from a part whose layout it could not read. So a column with a chunk
6945    /// of nothing but nulls in the middle of it goes and reads the rows. That is slow and right,
6946    /// and the fix is a row count per part rather than anything here.
6947    ///
6948    /// # Errors
6949    ///
6950    /// If the column is outside the schema.
6951    pub fn exact_extremes(&self, column: usize) -> Result<Option<(Bound, Bound)>> {
6952        if column >= self.table.fields.len() {
6953            return Err(invalid("extremes column index out of range"));
6954        }
6955        let mut low: Option<Bound> = None;
6956        let mut high: Option<Bound> = None;
6957        for stripe in &self.table.stripes {
6958            let range = stripe
6959                .zone
6960                .column(column)
6961                .ok_or_else(|| invalid("stripe zone is narrower than the schema"))?;
6962            if !range.exact {
6963                return Ok(None);
6964            }
6965            // A stripe of nothing but nulls has no ends and says nothing about the column's, which
6966            // is why this skips it rather than giving up on the whole column. A stripe that has
6967            // rows and still has no end is a layout whose values this cannot see, and skipping that
6968            // one would answer with an end taken from the other stripes, so it gives up instead.
6969            let (Some(small), Some(large)) = (range.low.as_ref(), range.high.as_ref()) else {
6970                if stripe.rows > range.nulls {
6971                    return Ok(None);
6972                }
6973                continue;
6974            };
6975            low = Some(low.map_or_else(|| small.clone(), |held| held.smaller(small.clone())));
6976            high = Some(high.map_or_else(|| large.clone(), |held| held.larger(large.clone())));
6977        }
6978        Ok(low.zip(high))
6979    }
6980
6981    /// The sum of one integer column and how many rows went into it, when every stripe wrote one.
6982    ///
6983    /// The count beside the sum is the non-null rows, because that is what a `SUM` adds up and what
6984    /// an `AVG` divides by, and a caller that had to work it out from the row count and the null
6985    /// count would be doing the same walk twice.
6986    ///
6987    /// `None` for anything that is not an integer column, for a file written by something that did
6988    /// not record it, and when adding the stripes together would overflow.
6989    ///
6990    /// # Errors
6991    ///
6992    /// If the column is outside the schema.
6993    pub fn exact_sum(&self, column: usize) -> Result<Option<(i128, u64)>> {
6994        if column >= self.table.fields.len() {
6995            return Err(invalid("sum column index out of range"));
6996        }
6997        let mut total = 0_i128;
6998        let mut rows = 0_u64;
6999        for stripe in &self.table.stripes {
7000            let range = stripe
7001                .zone
7002                .column(column)
7003                .ok_or_else(|| invalid("stripe zone is narrower than the schema"))?;
7004            let Some(part) = range.sum else { return Ok(None) };
7005            let Some(sum) = total.checked_add(part) else { return Ok(None) };
7006            total = sum;
7007            rows = rows.saturating_add(stripe.rows as u64 - range.nulls as u64);
7008        }
7009        Ok(Some((total, rows)))
7010    }
7011
7012    /// Legacy derived host groups are parsed for file compatibility but never used as query output.
7013    pub fn host_groups(
7014        &self,
7015        column: usize,
7016        _minimum_count: u64,
7017    ) -> Result<Option<Vec<host::HostEntry>>> {
7018        if column >= self.table.fields.len() {
7019            return Err(invalid("host group column index out of range"));
7020        }
7021        Ok(None)
7022    }
7023
7024    /// Whether the column's dictionary stopped taking values partway through the load, and so
7025    /// decodes the stripes written before that and says nothing about the column as a whole. See
7026    /// `DEMOTED`.
7027    #[must_use]
7028    pub fn demoted(&self, column: usize) -> bool {
7029        self.table.demoted.get(column).copied().unwrap_or(false)
7030    }
7031
7032    /// The global dictionary of a column, opened once however many workers ask for it at once.
7033    ///
7034    /// The unlocked look is first because it is the answer every time after the first and it costs a
7035    /// load. Everybody who misses it queues on [`Self::loading`] and looks again on the way in, so
7036    /// the one who arrived first does the reading and the rest take what it left. Waiting is the
7037    /// cheaper thing to do: the work behind the lock is a page read, a checksum and the decode of a
7038    /// dictionary that can hold half a million entries, and the alternative is every worker of the
7039    /// scan doing all of it and all but one dropping the result on the floor.
7040    fn dictionary(&self, column: usize) -> Result<Option<Arc<Vector>>> {
7041        let Some(page) = self.table.dictionaries[column] else { return Ok(None) };
7042        if let Some(dictionary) = self.dictionaries[column].get() {
7043            return Ok(Some(Arc::clone(dictionary)));
7044        }
7045        let _queued = self.loading[column].lock().map_err(|_| invalid("a poisoned dictionary"))?;
7046        if let Some(dictionary) = self.dictionaries[column].get() {
7047            return Ok(Some(Arc::clone(dictionary)));
7048        }
7049        self.opened.fetch_add(1, Atomic::Relaxed);
7050        let dictionary = Arc::new(open_global_dictionary(
7051            Arc::clone(&self.file),
7052            page,
7053            &self.table.fields[column].ty,
7054            TEXT_KEEP_BUDGET,
7055        )?);
7056        let _ = self.dictionaries[column].set(Arc::clone(&dictionary));
7057        Ok(Some(dictionary))
7058    }
7059
7060    /// Reads one section's extent table and checks it against the entry that names it.
7061    ///
7062    /// # Errors
7063    ///
7064    /// If the entry points outside the file, the table does not checksum, or it does not decode as
7065    /// a run of extents in element order.
7066    pub fn extents(&self, of: &Section) -> Result<Vec<section::Extent>> {
7067        if of.extent_bytes == 0 {
7068            return Ok(Vec::new());
7069        }
7070        let mut bytes = vec![0; of.extent_bytes as usize];
7071        read_at(&self.file, of.extent_page, &mut bytes)?;
7072        if checksum(&bytes) != of.hash {
7073            return Err(invalid("a section's extent table does not checksum"));
7074        }
7075        let extents = section::decode_extents(&bytes)?;
7076        if extents.len() != of.extents as usize {
7077            return Err(invalid("a section's extent table is not the length the entry says"));
7078        }
7079        Ok(extents)
7080    }
7081
7082    /// Reads and verifies one extent of a section.
7083    ///
7084    /// This is what section 3.2's second rule is for. A reduction that needs one extent of a two
7085    /// gigabyte forward link reads and checksums that extent and nothing else, which is the whole
7086    /// difference between a structure that works at SF100 and issue #745.
7087    ///
7088    /// # Errors
7089    ///
7090    /// If the extent points outside the file, or its bytes do not checksum.
7091    pub fn extent(&self, of: &section::Extent) -> Result<Vec<u8>> {
7092        let mut bytes = Vec::new();
7093        self.extent_into(of, &mut bytes)?;
7094        Ok(bytes)
7095    }
7096
7097    /// Read a verified extent into a caller-owned buffer so repeated extents can reuse its pages.
7098    fn extent_into(&self, of: &section::Extent, bytes: &mut Vec<u8>) -> Result<()> {
7099        let end = of
7100            .offset
7101            .checked_add(u64::from(of.length))
7102            .ok_or_else(|| invalid("an extent overflows the file"))?;
7103        if of.offset < HEADER || end > self.size {
7104            return Err(invalid("an extent is outside the file"));
7105        }
7106        bytes.resize(of.length as usize, 0);
7107        read_at(&self.file, of.offset, bytes)?;
7108        if checksum(bytes) != of.hash {
7109            return Err(invalid("an extent does not checksum"));
7110        }
7111        Ok(())
7112    }
7113
7114    /// Reads the first `len` bytes of a section's payload, or all of it when it is shorter, without
7115    /// checking them.
7116    ///
7117    /// Only the extent table is checked, because an extent's checksum is over the whole extent and
7118    /// checking it is reading the whole of it, which is what this is here to avoid. It is for a
7119    /// kind-specific header that a planner reads to decide what to plan, and never for bytes a
7120    /// query's answer is made of: a reader that goes on to use the structure reads it again through
7121    /// [`Self::payload`], and a header that was torn is found there.
7122    ///
7123    /// # Errors
7124    ///
7125    /// If the extent table fails its check or the first extent points outside the file.
7126    pub fn payload_head(&self, of: &Section, len: usize) -> Result<Vec<u8>> {
7127        let extents = self.extents(of)?;
7128        let Some(first) = extents.first() else { return Ok(Vec::new()) };
7129        let end = first
7130            .offset
7131            .checked_add(u64::from(first.length))
7132            .ok_or_else(|| invalid("an extent overflows the file"))?;
7133        if first.offset < HEADER || end > self.size {
7134            return Err(invalid("an extent is outside the file"));
7135        }
7136        let mut bytes = vec![0; len.min(first.length as usize)];
7137        read_at(&self.file, first.offset, &mut bytes)?;
7138        Ok(bytes)
7139    }
7140
7141    /// Reads a whole section's payload, every extent of it, in order.
7142    ///
7143    /// For a structure that is resident anyway, which a key map is. Anything large enough that the
7144    /// split matters should be walking [`Reader::extents`] and taking the one it needs.
7145    ///
7146    /// # Errors
7147    ///
7148    /// If the extent table or any extent fails its check.
7149    pub fn payload(&self, of: &Section) -> Result<Vec<u8>> {
7150        let extents = self.extents(of)?;
7151        let mut bytes =
7152            Vec::with_capacity(sum(extents.iter().map(|one| u64::from(one.length))) as usize);
7153        for one in &extents {
7154            if one.first != bytes.len() as u64 {
7155                return Err(invalid("a section's extents do not join up"));
7156            }
7157            bytes.extend_from_slice(&self.extent(one)?);
7158        }
7159        // The same exception `write_section` makes: a budget record has no bytes, so its
7160        // `header_bytes` is a size rather than a header and there is nothing for it to run past.
7161        if !bytes.is_empty() && of.header_bytes as usize > bytes.len() {
7162            return Err(invalid("a section's header is longer than its payload"));
7163        }
7164        Ok(bytes)
7165    }
7166
7167    /// Reads only the named columns from one part.
7168    ///
7169    /// The whole stripe page each column lives in is read and kept once a scan has been through the
7170    /// stripe before, because a session that scans a table again asks for the parts of a stripe one
7171    /// after another and this is what turns sixty four reads into one. The first time through, the
7172    /// part is read alone. See `Cached`.
7173    ///
7174    /// # Errors
7175    ///
7176    /// If a part, column, page, or checksum is invalid.
7177    pub fn read(&self, part: usize, columns: &[usize]) -> Result<Chunk> {
7178        self.read_impl(part, columns, true, None)
7179    }
7180
7181    /// Reads named columns from one part without keeping the stripe page it came out of.
7182    ///
7183    /// This is for sparse row fetches after a selective TopN or filter, which reach a few parts of
7184    /// a stripe rather than all of them. A caller that will read most of a stripe should use
7185    /// [`Self::read`] instead, because this reads and discards the page index every time.
7186    ///
7187    /// # Errors
7188    ///
7189    /// If a part, column, page, or checksum is invalid.
7190    pub fn read_sparse(&self, part: usize, columns: &[usize]) -> Result<Chunk> {
7191        self.read_impl(part, columns, false, None)
7192    }
7193
7194    /// Counts one signed integer part from its encoded row values when it uses an all-valid
7195    /// cascade. Sparse and run-length cascades are folded without expanding their rows. Other
7196    /// page forms return `None` so the caller can use the ordinary reader.
7197    ///
7198    /// # Errors
7199    ///
7200    /// If a part, column, page checksum, or encoded integer is invalid.
7201    pub fn integer_tally(&self, part: usize, column: usize) -> Result<Option<Vec<(i64, u64)>>> {
7202        let place = *self.places.get(part).ok_or_else(|| invalid("part index out of range"))?;
7203        let field =
7204            self.table.fields.get(column).ok_or_else(|| invalid("column index out of range"))?;
7205        if !matches!(
7206            field.ty,
7207            LogicalType::TinyInt
7208                | LogicalType::SmallInt
7209                | LogicalType::Integer
7210                | LogicalType::BigInt
7211        ) {
7212            return Ok(None);
7213        }
7214        let (rows, counts) = match self.with_part(place, column, |bytes| {
7215            if bytes.first() != Some(&5) || bytes.get(1) != Some(&0) {
7216                return Ok(None);
7217            }
7218            integer::tally(&bytes[2..]).map(Some)
7219        })? {
7220            Some(tallied) => tallied,
7221            None => return Ok(None),
7222        };
7223        if rows != place.rows as usize {
7224            return Err(invalid("encoded integer part holds the wrong number of rows"));
7225        }
7226        for &(value, _) in &counts {
7227            let fits = match field.ty {
7228                LogicalType::TinyInt => i8::try_from(value).is_ok(),
7229                LogicalType::SmallInt => i16::try_from(value).is_ok(),
7230                LogicalType::Integer => i32::try_from(value).is_ok(),
7231                LogicalType::BigInt => true,
7232                _ => false,
7233            };
7234            if !fits {
7235                return Err(invalid("encoded integer value is outside its column type"));
7236            }
7237        }
7238        Ok(Some(counts))
7239    }
7240
7241    /// The rows of one text part that hold `sequence`'s pieces in order, or with `negated` the rows
7242    /// that do not, answered on the compressed page without decompressing it. Nulls are in neither.
7243    /// `None` for a part that is not compressed text, which the caller reads the usual way.
7244    ///
7245    /// For a scan whose filter is the only thing that reads the column, which then never has the
7246    /// strings at all. In TPC-H q13 that is `o_comment NOT LIKE '%special%requests%'`, and
7247    /// decompressing the comments and searching them was most of the orders scan.
7248    ///
7249    /// # Errors
7250    ///
7251    /// If a part, column, page, or checksum is invalid.
7252    pub fn rows_holding(
7253        &self,
7254        part: usize,
7255        column: usize,
7256        sequence: &Sequence,
7257        negated: bool,
7258    ) -> Result<Option<Vec<u32>>> {
7259        let place = *self.places.get(part).ok_or_else(|| invalid("part index out of range"))?;
7260        let field =
7261            self.table.fields.get(column).ok_or_else(|| invalid("column index out of range"))?;
7262        if field.ty != LogicalType::Varchar {
7263            return Ok(None);
7264        }
7265        let rows = place.rows as usize;
7266        self.with_part(place, column, |bytes| {
7267            if bytes.first() != Some(&6) {
7268                return Ok(None);
7269            }
7270            let mut cur = Cursor::new(bytes);
7271            cur.u8()?;
7272            let mask = match cur.u8()? {
7273                0 => None,
7274                1 => return Ok(Some(Vec::new())),
7275                2 => {
7276                    let from = cur.at;
7277                    cur.take(rows.div_ceil(8))?;
7278                    Some(&bytes[from..cur.at])
7279                }
7280                _ => return Err(invalid("page validity tag differs")),
7281            };
7282            // A row whose sketch lacks a bit the pieces need cannot hold them, so only the rest
7283            // are walked. See `grams`.
7284            let needs = sequence.needs();
7285            let first = self.firsts.get(part).copied().unwrap_or_default();
7286            let sketch = self
7287                .text_grams
7288                .get(column)
7289                .and_then(|slot| slot.get_or_init(|| grams::text_grams(self, column)).as_deref())
7290                .and_then(|words| words.get(first..first + rows));
7291            let maybe = |row: usize| sketch.is_none_or(|words| words[row] & needs == needs);
7292            let Some(held) = string::holds_in_where(&bytes[cur.at..], sequence, maybe)? else {
7293                return Ok(None);
7294            };
7295            if held.len() != rows {
7296                return Err(invalid("compressed text page holds the wrong number of rows"));
7297            }
7298            let valid = |row: usize| mask.is_none_or(|mask| mask[row / 8] >> (row % 8) & 1 == 1);
7299            Ok(Some(
7300                (0..rows)
7301                    .filter(|&row| held[row] != negated && valid(row))
7302                    .map(|row| row as u32)
7303                    .collect(),
7304            ))
7305        })
7306    }
7307
7308    /// Whether part and column `bit`, numbered as [`Reader::verified`] numbers them, has matched its
7309    /// checksum since this reader was opened.
7310    fn is_verified(&self, bit: usize) -> bool {
7311        self.verified
7312            .get(bit / 64)
7313            .is_some_and(|word| word.load(Atomic::Relaxed) >> (bit % 64) & 1 == 1)
7314    }
7315
7316    /// Remembers that part and column `bit` matched its checksum.
7317    fn set_verified(&self, bit: usize) {
7318        if let Some(word) = self.verified.get(bit / 64) {
7319            word.fetch_or(1 << (bit % 64), Atomic::Relaxed);
7320        }
7321    }
7322
7323    /// Runs `read` over the stored bytes of one column of one part, out of the stripe's page when
7324    /// it is held and read off the file on their own when it is not.
7325    fn with_part<T>(
7326        &self,
7327        place: Place,
7328        column: usize,
7329        read: impl FnOnce(&[u8]) -> Result<T>,
7330    ) -> Result<T> {
7331        let stripe_index = place.stripe as usize;
7332        let stripe = self
7333            .table
7334            .stripes
7335            .get(stripe_index)
7336            .ok_or_else(|| invalid("stripe index out of range"))?;
7337        let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
7338        let held = self.held(stripe_index, place.part as usize, stripe, column, true)?;
7339        let span = *held
7340            .index
7341            .get(place.part as usize)
7342            .ok_or_else(|| invalid("part index out of range"))?;
7343        match &held.page {
7344            Some(page) => read(page.part(place.part as usize, span)?),
7345            None => {
7346                let offset = page
7347                    .offset
7348                    .checked_add(span.start as u64)
7349                    .ok_or_else(|| invalid("part range overflow"))?;
7350                let mut bytes = vec![0; span.length];
7351                read_at(&self.file, offset, &mut bytes)?;
7352                verify_part(&bytes, span)?;
7353                read(&bytes)
7354            }
7355        }
7356    }
7357
7358    /// Reads named columns from one part, only at the rows `positions` names.
7359    ///
7360    /// For a scan that already knows which rows of the part it keeps, from the columns it read
7361    /// first. A compressed string page decompresses only those rows, and every other page is
7362    /// decoded whole and gathered, which is what reading it and narrowing it costs anyway. With
7363    /// `whole` the stripe's pages are kept the way [`Self::read`] keeps them, and without it they
7364    /// are not, the way [`Self::read_sparse`] does.
7365    ///
7366    /// # Errors
7367    ///
7368    /// If a part, column, page, or checksum is invalid, or the positions do not rise or run past
7369    /// the end of the part.
7370    pub fn read_rows(
7371        &self,
7372        part: usize,
7373        columns: &[usize],
7374        positions: &[u32],
7375        whole: bool,
7376    ) -> Result<Chunk> {
7377        self.read_impl(part, columns, whole, Some(positions))
7378    }
7379
7380    /// Whether an exact global-code membership index proves that the stripe holding a part cannot
7381    /// contain any of the sorted candidate codes.
7382    ///
7383    /// # Errors
7384    ///
7385    /// If the part, column, index page, checksum, or delta stream is invalid.
7386    pub fn skips_codes(&self, part: usize, column: usize, candidates: &[u32]) -> Result<bool> {
7387        // A demoted column's later stripes hold values the dictionary never coded, so no list of
7388        // codes can prove a stripe of it holds none of a value.
7389        if self.demoted(column) {
7390            return Ok(false);
7391        }
7392        if candidates.is_empty() {
7393            return Ok(true);
7394        }
7395        if candidates.windows(2).any(|pair| pair[0] >= pair[1]) {
7396            return Err(Error::internal("native code candidates are not sorted and unique"));
7397        }
7398        let stripe = self.stripe_of(part)?;
7399        let Some(page) = stripe.memberships.get(column) else {
7400            return Ok(false);
7401        };
7402        let mut bytes = vec![0; page.length as usize];
7403        read_at(&self.file, page.offset, &mut bytes)?;
7404        if checksum(&bytes) != page.hash {
7405            return Err(invalid("membership page checksum differs"));
7406        }
7407        let codes = decode_membership(&bytes)?;
7408        let mut left = 0;
7409        let mut right = 0;
7410        while left < codes.len() && right < candidates.len() {
7411            match codes[left].cmp(&candidates[right]) {
7412                Ordering::Less => left += 1,
7413                Ordering::Greater => right += 1,
7414                Ordering::Equal => return Ok(false),
7415            }
7416        }
7417        Ok(true)
7418    }
7419
7420    fn stripe_of(&self, part: usize) -> Result<&Stripe> {
7421        let place = self.places.get(part).ok_or_else(|| invalid("part index out of range"))?;
7422        self.table
7423            .stripes
7424            .get(place.stripe as usize)
7425            .ok_or_else(|| invalid("stripe index out of range"))
7426    }
7427
7428    /// The page index of one column of one stripe, and its page when the caller wants all of it.
7429    ///
7430    /// A scan hands parts out in order, so every worker on a column crosses into a new stripe within
7431    /// a few parts of the others and they all want the same page at the same moment. This used to
7432    /// let all of them read it, which cost the scan as many copies of every page as it had workers.
7433    /// On the full ClickBench file a `MIN(EventDate), MAX(EventDate)` moved 3.2 GB off the disk to
7434    /// look at 400 MB of column.
7435    ///
7436    /// A worker that finds the page it wants already being read neither waits for it nor reads it
7437    /// again. It comes back with the index alone, which sends [`Reader::read_impl`] down the path
7438    /// that reads the one part it came for, a few kilobytes against a quarter of a megabyte, and it
7439    /// picks the page up from the cache on its next part. Waiting would be the other way to avoid
7440    /// the duplicate read and it is worse: the pages that matter are the wide string ones, they take
7441    /// milliseconds to copy even warm, and every other worker would be stopped for all of it.
7442    ///
7443    /// The file is never read under the lock.
7444    fn held(
7445        &self,
7446        at: usize,
7447        part: usize,
7448        stripe: &Stripe,
7449        column: usize,
7450        whole: bool,
7451    ) -> Result<CachedColumn> {
7452        let cache =
7453            self.cache.columns.get(column).ok_or_else(|| invalid("column index out of range"))?;
7454        let mut cached = cache.lock().map_err(|_| invalid("column page cache is poisoned"))?;
7455        let known = cached.index.get(at).and_then(Clone::clone);
7456        let page = cached.pages.get(at).and_then(Option::as_ref).map(|slot| {
7457            slot.used.store(true, Atomic::Relaxed);
7458            Arc::clone(&slot.page)
7459        });
7460        // Whole only for a part asked for before, see [`Cached`].
7461        let (again, through) = match cached.touched.get_mut(at) {
7462            Some(bits) if whole && page.is_none() => touch(bits, part, stripe.parts.len()),
7463            _ => (false, false),
7464        };
7465        let whole = whole && again;
7466        if let Some(index) = known.clone()
7467            && (!whole || page.is_some())
7468        {
7469            return Ok(CachedColumn { stripe: at, index, page });
7470        }
7471        if cached.loading.contains(&at) {
7472            drop(cached);
7473            // The index is almost always already here, because somebody read this stripe to get
7474            // into the loading list in the first place, so this branch usually costs no read at
7475            // all and the one part read in `read_impl` is all the losing worker pays for.
7476            if let Some(index) = known {
7477                return Ok(CachedColumn { stripe: at, index, page: None });
7478            }
7479            let held = self.page_of(stripe, column, at, false, None)?;
7480            let mut cached = cache.lock().map_err(|_| invalid("column page cache is poisoned"))?;
7481            remember(&mut cached, &held);
7482            return Ok(held);
7483        }
7484        cached.loading.push(at);
7485        drop(cached);
7486
7487        let read = self.page_of(stripe, column, at, whole, known);
7488
7489        // The stripe leaves the loading list and its page enters the cache under one lock. Doing
7490        // them separately would leave a moment where another worker sees neither and reads the
7491        // page a second time, which is the whole thing this is here to stop.
7492        let mut cached = cache.lock().map_err(|_| invalid("column page cache is poisoned"))?;
7493        if let Some(position) = cached.loading.iter().position(|loading| *loading == at) {
7494            cached.loading.remove(position);
7495        }
7496        let held = read?;
7497        let taken = remember(&mut cached, &held);
7498        if taken.is_some() && !through {
7499            let floor = self.cache.kept.load(Atomic::Relaxed).max(1);
7500            cached.passing.push_back(at);
7501            while cached.passing.len() > floor {
7502                let Some(old) = cached.passing.pop_front() else { break };
7503                if let Some(slot) = cached.pages.get_mut(old) {
7504                    *slot = None;
7505                }
7506            }
7507            return Ok(held);
7508        }
7509        drop(cached);
7510        if let Some((bytes, used)) = taken {
7511            self.cache.held[column].fetch_add(1, Atomic::Relaxed);
7512            self.pool.admit(Held {
7513                shelf: Arc::downgrade(&self.cache),
7514                column,
7515                stripe: at,
7516                bytes,
7517                used,
7518            });
7519        }
7520        Ok(held)
7521    }
7522
7523    /// Reads one stripe's index for a column, and its page when the caller wants all of it.
7524    ///
7525    /// `known` is the index when the reader has already read it, which after the first worker
7526    /// through a stripe it always has, because [`remember`] keeps every index for the life of the
7527    /// reader. Without that a scan reads the index again on every part that misses the page cache.
7528    fn page_of(
7529        &self,
7530        stripe: &Stripe,
7531        column: usize,
7532        at: usize,
7533        whole: bool,
7534        known: Option<Arc<Vec<PartSpan>>>,
7535    ) -> Result<CachedColumn> {
7536        let index = match known {
7537            Some(index) => index,
7538            None => {
7539                self.indexes.fetch_add(1, Atomic::Relaxed);
7540                Arc::new(read_index(&self.file, stripe, column)?)
7541            }
7542        };
7543        let page = if whole {
7544            self.pages.fetch_add(1, Atomic::Relaxed);
7545            let span = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
7546            let mut bytes = vec![0; span.length as usize];
7547            read_at(&self.file, span.offset, &mut bytes)?;
7548            let checked = index.iter().map(|_| AtomicBool::new(false)).collect();
7549            Some(Arc::new(HeldPage { bytes, checked }))
7550        } else {
7551            None
7552        };
7553        Ok(CachedColumn { stripe: at, index, page })
7554    }
7555
7556    fn read_impl(
7557        &self,
7558        at: usize,
7559        columns: &[usize],
7560        whole: bool,
7561        positions: Option<&[u32]>,
7562    ) -> Result<Chunk> {
7563        let place = *self.places.get(at).ok_or_else(|| invalid("part index out of range"))?;
7564        let index = place.stripe as usize;
7565        let stripe =
7566            self.table.stripes.get(index).ok_or_else(|| invalid("stripe index out of range"))?;
7567        let rows = place.rows as usize;
7568        let mut picked = Vec::with_capacity(columns.len());
7569        for &column in columns {
7570            let field = self
7571                .table
7572                .fields
7573                .get(column)
7574                .ok_or_else(|| invalid("column index out of range"))?;
7575            let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
7576            let held = self.held(index, place.part as usize, stripe, column, whole)?;
7577            let span = *held
7578                .index
7579                .get(place.part as usize)
7580                .ok_or_else(|| invalid("part index out of range"))?;
7581            let owned;
7582            let bit = at * self.table.fields.len() + column;
7583            let bytes = match &held.page {
7584                Some(held) if self.is_verified(bit) => part_bytes(&held.bytes, span),
7585                Some(held) => {
7586                    held.part(place.part as usize, span).inspect(|_| self.set_verified(bit))
7587                }
7588                None => {
7589                    let offset = page
7590                        .offset
7591                        .checked_add(span.start as u64)
7592                        .ok_or_else(|| invalid("part range overflow"))?;
7593                    let mut bytes = vec![0; span.length];
7594                    read_at(&self.file, offset, &mut bytes)?;
7595                    owned = bytes;
7596                    if self.is_verified(bit) {
7597                        Ok(owned.as_slice())
7598                    } else {
7599                        verify_part(&owned, span).map(|()| {
7600                            self.set_verified(bit);
7601                            owned.as_slice()
7602                        })
7603                    }
7604                }
7605            }
7606            .map_err(|error| {
7607                invalid(&format!(
7608                    "{}, column {column} part {} of the page at {}",
7609                    error.message(),
7610                    place.part,
7611                    page.offset,
7612                ))
7613            })?;
7614            let dictionary = self.dictionary(column)?;
7615            // Held as a page, because a column that came out of a file is handed out more than
7616            // once. A group by clones its key columns out of the chunk so the keys outlive it, a
7617            // projection of a bare column name does the same, and a cut of a flat run copies unless
7618            // the run is a page. One `Arc` per column per part buys all of those, and it moves the
7619            // run into the `Arc` without touching a value.
7620            let mut vector = match positions {
7621                None => decode(&field.ty, rows, bytes, dictionary)?,
7622                Some(positions) => decode_at(&field.ty, rows, bytes, dictionary, positions)?,
7623            };
7624            // A demoted column's codes are not the column's codes, only the codes of the stripes
7625            // written before the demotion, so they are not handed out as if they were. See
7626            // [`DEMOTED`].
7627            if self.demoted(column) && vector.stable_dictionary_parts().is_some() {
7628                vector = vector.flatten()?;
7629            }
7630            picked.push(vector.into_pages());
7631        }
7632        Chunk::with_rows(picked, positions.map_or(rows, <[u32]>::len))
7633    }
7634
7635    /// Whether persisted statistics prove that a part cannot match the predicates.
7636    ///
7637    /// Three of them, asked cheapest first.
7638    ///
7639    /// The stripe's bounds are in memory already, so they are free, and they are also the coarsest:
7640    /// every part of a stripe gets the same answer and a scan that skips one part that way skips all
7641    /// sixty four. Then the part's own bounds, which are a read of one page per column per stripe
7642    /// and are sixty four times finer. Then the sieves, which are per part and answer equality, the
7643    /// test bounds are worst at: a column of identifiers has every stripe and nearly every part
7644    /// covering the whole of its type, so bounds keep them all and the sieve keeps the ones that
7645    /// really hold the value.
7646    ///
7647    /// The middle one is what an ordered comparison on a column the rows are not sorted by needs. On
7648    /// ClickBench 24 the stripe bounds leave eight stripes of sixteen alive, which is half the file,
7649    /// and the part bounds leave thirty parts of nine hundred and seventy four.
7650    #[must_use]
7651    pub fn skips(&self, part: usize, probes: &[Probe]) -> bool {
7652        let Some(place) = self.places.get(part).copied() else { return false };
7653        let Some(stripe) = self.table.stripes.get(place.stripe as usize) else { return false };
7654        if stripe.zone.skips(probes) {
7655            return true;
7656        }
7657        probes.iter().any(|probe| self.outside(place, probe) || self.sifted(place, probe))
7658    }
7659
7660    /// Whether `rule` rules out a part from what the stored range of one column says about it.
7661    ///
7662    /// The same two steps as [`Self::skips`] without the sieve, for a test no [`Probe`] can write.
7663    /// A probe is one comparison against one constant, and the keys a join's build side holds are a
7664    /// set, which rules a part out when none of them falls inside the part's two ends. Handing the
7665    /// range to the caller is what lets the set stay with the join that knows what it is.
7666    #[must_use]
7667    pub fn ruled_by(&self, part: usize, column: usize, rule: impl Fn(&Range) -> bool) -> bool {
7668        let Some(place) = self.places.get(part).copied() else { return false };
7669        let Some(stripe) = self.table.stripes.get(place.stripe as usize) else { return false };
7670        if stripe.zone.column(column).is_some_and(&rule) {
7671            return true;
7672        }
7673        self.stripe_part_ranges(place.stripe as usize, column)
7674            .and_then(|ranges| ranges.get(place.part as usize))
7675            .is_some_and(rule)
7676    }
7677
7678    /// The stored range of one column over one part, the part's own where its stripe kept one and
7679    /// the stripe's where it did not, which is wider but still holds every row of the part.
7680    #[must_use]
7681    pub fn part_range(&self, part: usize, column: usize) -> Option<Range> {
7682        let place = self.places.get(part).copied()?;
7683        let own = self
7684            .stripe_part_ranges(place.stripe as usize, column)
7685            .and_then(|ranges| ranges.get(place.part as usize));
7686        own.or_else(|| self.table.stripes.get(place.stripe as usize)?.zone.column(column)).cloned()
7687    }
7688
7689    /// The half of [`Self::ruled_by`] that reads nothing, asked about a whole stripe.
7690    #[must_use]
7691    pub fn stripe_ruled_by(
7692        &self,
7693        stripe: usize,
7694        column: usize,
7695        rule: impl Fn(&Range) -> bool,
7696    ) -> bool {
7697        self.table.stripes.get(stripe).and_then(|held| held.zone.column(column)).is_some_and(rule)
7698    }
7699
7700    /// Whether the bounds of one part rule out one probe.
7701    ///
7702    /// The part's own two ends, which are narrower than the stripe's and cost a page read the first
7703    /// time this is asked about a column. A column with no page here answers `false`, which is the
7704    /// answer a caller got before there were any.
7705    fn outside(&self, place: Place, probe: &Probe) -> bool {
7706        match self.stripe_part_ranges(place.stripe as usize, probe.column) {
7707            Some(ranges) => ranges
7708                .get(place.part as usize)
7709                .is_some_and(|range| range.excludes(probe.op, &probe.value)),
7710            None => false,
7711        }
7712    }
7713
7714    /// The per part ranges of one stripe of one column, read once and kept.
7715    ///
7716    /// `None` when the column has no page in that stripe and when the page is damaged, on the same
7717    /// reasoning as the sieves: this is an index over data that is still there, so a caller that
7718    /// cannot read one reads the rows and gets the right answer slowly.
7719    fn stripe_part_ranges(&self, stripe: usize, column: usize) -> Option<&[Range]> {
7720        let slot = self.part_ranges.get(column)?.get(stripe)?;
7721        if let Some(held) = slot.get() {
7722            return Some(held);
7723        }
7724        let page = self.table.stripes.get(stripe)?.part_ranges.get(column)?;
7725        let mut bytes = vec![0; page.length as usize];
7726        read_at(&self.file, page.offset, &mut bytes).ok()?;
7727        if checksum(&bytes) != page.hash {
7728            return None;
7729        }
7730        let ranges = Arc::new(decode_part_ranges(&bytes).ok()?);
7731        let _ = slot.set(ranges);
7732        slot.get().map(|held| held.as_slice())
7733    }
7734
7735    /// Whether persisted statistics prove that every row of a part matches the predicates.
7736    ///
7737    /// Only the bounds. The sieves say nothing here, because a sieve that holds a value is a sieve
7738    /// that may be holding somebody else's hash, so it can rule a part out and can never wave one
7739    /// through.
7740    ///
7741    /// The stripe first and the part after it, the same two steps and in the same order as
7742    /// [`Self::skips`]. The stripe's bounds are in memory already and its null count covers sixty
7743    /// four parts rather than one, so a stripe that answers is an answer for nothing, and the part's
7744    /// own bounds are only read for the probes it could not settle. Both directions are safe: a
7745    /// stretch where everything passes contains no narrower stretch where something fails, and a
7746    /// stripe with no nulls has no nulls in any of its parts.
7747    ///
7748    /// A string end a part recorded is cut down to its first few bytes, so a part's stretch can be
7749    /// wider than its rows really are as well. That is the same safe direction for the same reason,
7750    /// and it is why this asks the two ends rather than anything `exact` says.
7751    #[must_use]
7752    pub fn certain(&self, part: usize, probes: &[Probe]) -> bool {
7753        let Some(place) = self.places.get(part).copied() else { return false };
7754        let Some(stripe) = self.table.stripes.get(place.stripe as usize) else { return false };
7755        if stripe.zone.certain(probes) {
7756            return true;
7757        }
7758        probes
7759            .iter()
7760            .all(|probe| stripe.zone.certain(slice::from_ref(probe)) || self.inside(place, probe))
7761    }
7762
7763    /// Whether one part's own two ends prove that every row of it passes `probe`.
7764    ///
7765    /// The mirror of [`Self::outside`], reading the same page. `false` for a part whose stripe wrote
7766    /// no range page, which is a stripe of one part, because there the stripe's own bounds are the
7767    /// part's and the caller has already asked them.
7768    fn inside(&self, place: Place, probe: &Probe) -> bool {
7769        match self.stripe_part_ranges(place.stripe as usize, probe.column) {
7770            Some(ranges) => ranges
7771                .get(place.part as usize)
7772                .is_some_and(|range| range.certain(probe.op, &probe.value)),
7773            None => false,
7774        }
7775    }
7776
7777    /// Whether the bounds of one stripe prove that none of its parts can match the predicates.
7778    ///
7779    /// The cheap half of [`Self::skips`], asked about a whole stripe at once. The bounds live in the
7780    /// directory and are already in memory, so this answers without touching the file, and that is
7781    /// the reason it is worth having on its own: a caller that wants to know roughly where the work
7782    /// is before it starts any workers can ask this about sixteen stripes for nothing, where asking
7783    /// [`Self::skips`] about nine hundred parts would read and decode a sieve page per stripe first.
7784    ///
7785    /// It keeps stripes that [`Self::skips`] would rule out part by part, which is the right way for
7786    /// it to be wrong: the parts are still checked when they are read.
7787    #[must_use]
7788    pub fn stripe_skips(&self, stripe: usize, probes: &[Probe]) -> bool {
7789        self.table.stripes.get(stripe).is_some_and(|held| held.zone.skips(probes))
7790    }
7791
7792    /// Whether the sieve of one part rules out one probe.
7793    ///
7794    /// Only equality. An ordered comparison is what the bounds are for and a sieve says nothing
7795    /// about it, and a read that cannot answer keeps the part, which is the answer a caller with no
7796    /// sieve gets anyway.
7797    fn sifted(&self, place: Place, probe: &Probe) -> bool {
7798        if probe.op != Op::Equal {
7799            return false;
7800        }
7801        match self.stripe_sieves(place.stripe as usize, probe.column) {
7802            Some(sieves) => sieves
7803                .get(place.part as usize)
7804                .and_then(Option::as_ref)
7805                .is_some_and(|sieve| sieve.excludes(&probe.value)),
7806            None => false,
7807        }
7808    }
7809
7810    /// The sieves of one stripe of one column, read once and kept.
7811    ///
7812    /// `None` when the column has no sieves in that stripe, when the page is damaged, and when the
7813    /// bytes are not a page this version can read. A sieve is an index over data that is still there
7814    /// and a caller that cannot read one reads the rows, so this is the one place in the file where
7815    /// a bad checksum is a slow query rather than an error.
7816    fn stripe_sieves(&self, stripe: usize, column: usize) -> Option<&[Option<Sieve>]> {
7817        let slot = self.sieves.get(column)?.get(stripe)?;
7818        if let Some(held) = slot.get() {
7819            return Some(held);
7820        }
7821        let page = self.table.stripes.get(stripe)?.sieves.get(column)?;
7822        let mut bytes = vec![0; page.length as usize];
7823        read_at(&self.file, page.offset, &mut bytes).ok()?;
7824        if checksum(&bytes) != page.hash {
7825            return None;
7826        }
7827        let sieves = Arc::new(decode_sieves(&bytes).ok()?);
7828        let _ = slot.set(sieves);
7829        slot.get().map(|held| held.as_slice())
7830    }
7831}
7832
7833/// The value sitting at one position of a dictionary's sorted order.
7834fn text_at_rank(dictionary: &Vector, rank: usize) -> Result<Value> {
7835    let code = dictionary.code_at_rank(rank)? as usize;
7836    if dictionary.logical_type() == &LogicalType::Blob {
7837        let bytes = dictionary
7838            .try_bytes_at(code)?
7839            .ok_or_else(|| invalid("global dictionary order names a code it does not have"))?;
7840        return Ok(Value::Blob(bytes.to_vec()));
7841    }
7842    let text = dictionary
7843        .try_text_at(code)?
7844        .ok_or_else(|| invalid("global dictionary order names a code it does not have"))?;
7845    Ok(Value::Varchar(text.into()))
7846}
7847
7848/// Reads one span of a file at an offset, without moving a cursor anybody else can see.
7849///
7850/// Every reader of a table shares one [`File`] behind an [`Arc`], and a grouped aggregate reads its
7851/// pages from several threads at once, so this has to be positional. Seeking and then reading is
7852/// two calls with a gap in the middle, and in that gap another thread's seek lands and the read
7853/// comes back with somebody else's bytes.
7854///
7855/// The writer reads back through here too, out of the `rudb_io` file it writes through, which is
7856/// why this takes anything [`Positional`] rather than a [`File`].
7857fn read_at<F: Positional + ?Sized>(file: &F, offset: u64, bytes: &mut [u8]) -> Result<()> {
7858    file.fill_at(offset, bytes)
7859}
7860
7861/// Something a span of bytes can be read out of by offset.
7862///
7863/// There are two of these. The reader holds a `std::fs::File`, because it shares it between its
7864/// threads behind an [`Arc`] and every read it makes is on the hot path of a scan. The writer holds
7865/// an `rudb_io::File`, because everything it does to the file has to be something the simulated
7866/// filesystem can stop and crash. The few helpers both of them use, [`read_index`] and the choice
7867/// of committed slot, are written once over this rather than once for each.
7868trait Positional {
7869    /// Fills `bytes` from `offset`, or fails if the file ends first.
7870    ///
7871    /// Both kinds can come back short, so both loop. A read of zero bytes before the span is filled
7872    /// means the file stops earlier than the directory said it does.
7873    fn fill_at(&self, offset: u64, bytes: &mut [u8]) -> Result<()>;
7874}
7875
7876impl<T: Positional + ?Sized> Positional for &T {
7877    fn fill_at(&self, offset: u64, bytes: &mut [u8]) -> Result<()> {
7878        (**self).fill_at(offset, bytes)
7879    }
7880}
7881
7882impl<T: Positional + ?Sized> Positional for Arc<T> {
7883    fn fill_at(&self, offset: u64, bytes: &mut [u8]) -> Result<()> {
7884        (**self).fill_at(offset, bytes)
7885    }
7886}
7887
7888impl<T: Positional + ?Sized> Positional for Box<T> {
7889    fn fill_at(&self, offset: u64, bytes: &mut [u8]) -> Result<()> {
7890        (**self).fill_at(offset, bytes)
7891    }
7892}
7893
7894impl Positional for dyn rudb_io::File + '_ {
7895    fn fill_at(&self, mut offset: u64, mut bytes: &mut [u8]) -> Result<()> {
7896        while !bytes.is_empty() {
7897            let read = self.read_at(offset, bytes)?;
7898            if read == 0 {
7899                return Err(invalid("column page ends before its declared length"));
7900            }
7901            offset += read as u64;
7902            bytes = &mut bytes[read..];
7903        }
7904        Ok(())
7905    }
7906}
7907
7908impl Positional for File {
7909    #[cfg(unix)]
7910    fn fill_at(&self, mut offset: u64, mut bytes: &mut [u8]) -> Result<()> {
7911        use std::os::unix::fs::FileExt;
7912        while !bytes.is_empty() {
7913            let read = self.read_at(bytes, offset).map_err(io)?;
7914            if read == 0 {
7915                return Err(invalid("column page ends before its declared length"));
7916            }
7917            offset += read as u64;
7918            bytes = &mut bytes[read..];
7919        }
7920        Ok(())
7921    }
7922
7923    /// The same read, on the call Windows spells differently.
7924    ///
7925    /// `seek_read` is one `ReadFile` carrying the offset with it, so two of them cannot interleave
7926    /// the way a seek and a read can. It does leave the shared cursor somewhere afterwards, which is
7927    /// why nothing in this file may read that cursor.
7928    #[cfg(windows)]
7929    fn fill_at(&self, mut offset: u64, mut bytes: &mut [u8]) -> Result<()> {
7930        use std::os::windows::fs::FileExt;
7931        while !bytes.is_empty() {
7932            let read = self.seek_read(bytes, offset).map_err(io)?;
7933            if read == 0 {
7934                return Err(invalid("column page ends before its declared length"));
7935            }
7936            offset += read as u64;
7937            bytes = &mut bytes[read..];
7938        }
7939        Ok(())
7940    }
7941
7942    /// Somewhere that is neither, where the cursor is all there is.
7943    ///
7944    /// This one does race, and there is no way to write it so it does not. Nothing we build for
7945    /// runs here, so it exists to keep the crate compiling rather than to be correct under threads.
7946    #[cfg(not(any(unix, windows)))]
7947    fn fill_at(&self, offset: u64, bytes: &mut [u8]) -> Result<()> {
7948        use std::io::{Read, Seek, SeekFrom};
7949        let mut file = self.try_clone().map_err(io)?;
7950        file.seek(SeekFrom::Start(offset)).map_err(io)?;
7951        file.read_exact(bytes).map_err(io)
7952    }
7953}
7954
7955/// Overwrites one span of a file in place, which is how the tests damage a file on purpose.
7956///
7957/// The writer does not come through here. It writes through `rudb_io`, and this is a
7958/// `std::fs::File` opened by a test beside it.
7959#[cfg(test)]
7960fn write_at(file: &File, offset: u64, bytes: &[u8]) -> Result<()> {
7961    use std::io::{Seek, SeekFrom, Write};
7962    let mut file = file;
7963    file.seek(SeekFrom::Start(offset)).map_err(io)?;
7964    file.write_all(bytes).map_err(io)
7965}
7966
7967/// What a column type is called in the directory.
7968///
7969/// A tag is a number in a file somebody else wrote, so a tag that has been used is used forever and
7970/// the only thing that may happen to this list is that it grows. 1 to 13 are the tags the format
7971/// had when it could store thirteen types, and 14 to 27 are the rest, in the order they were added
7972/// rather than in an order that means anything.
7973fn type_tag(ty: &LogicalType) -> Result<u8> {
7974    match ty {
7975        LogicalType::SmallInt => Ok(1),
7976        LogicalType::Integer => Ok(2),
7977        LogicalType::BigInt => Ok(3),
7978        LogicalType::Varchar => Ok(4),
7979        LogicalType::Date => Ok(5),
7980        LogicalType::Timestamp => Ok(6),
7981        LogicalType::Boolean => Ok(7),
7982        LogicalType::TinyInt => Ok(8),
7983        LogicalType::UTinyInt => Ok(9),
7984        LogicalType::USmallInt => Ok(10),
7985        LogicalType::UInteger => Ok(11),
7986        LogicalType::UBigInt => Ok(12),
7987        LogicalType::Decimal { .. } => Ok(13),
7988        LogicalType::Float => Ok(14),
7989        LogicalType::Double => Ok(15),
7990        LogicalType::HugeInt => Ok(16),
7991        LogicalType::UHugeInt => Ok(17),
7992        LogicalType::Time => Ok(18),
7993        LogicalType::TimeTz => Ok(19),
7994        LogicalType::TimestampTz => Ok(20),
7995        LogicalType::Interval => Ok(21),
7996        LogicalType::Uuid => Ok(22),
7997        LogicalType::Blob => Ok(23),
7998        LogicalType::Bit => Ok(24),
7999        LogicalType::TimestampS => Ok(25),
8000        LogicalType::TimestampMs => Ok(26),
8001        LogicalType::TimestampNs => Ok(27),
8002        _ => Err(Error::not_implemented(format!("native storage for {ty}"))),
8003    }
8004}
8005
8006/// The tag of a column type, and the parameters of the ones that have any.
8007///
8008/// Only `DECIMAL` has parameters today. Width and scale go after the tag rather than into it
8009/// because they are what says how wide a value is on disk, and a reader that guessed would read the
8010/// wrong number of bytes per row rather than the wrong number of digits.
8011fn put_type(out: &mut Vec<u8>, ty: &LogicalType) -> Result<()> {
8012    out.push(type_tag(ty)?);
8013    if let LogicalType::Decimal { width, scale } = ty {
8014        out.push(*width);
8015        out.push(*scale);
8016    }
8017    Ok(())
8018}
8019
8020/// The other half of [`put_type`], reading the parameters the tag says are there.
8021fn read_type(cur: &mut Cursor<'_>) -> Result<LogicalType> {
8022    let tag = cur.u8()?;
8023    if tag == 13 {
8024        let width = cur.u8()?;
8025        let scale = cur.u8()?;
8026        return LogicalType::decimal(width, scale)
8027            .map_err(|_| invalid("decimal column width and scale are not a decimal"));
8028    }
8029    tag_type(tag)
8030}
8031
8032fn tag_type(tag: u8) -> Result<LogicalType> {
8033    match tag {
8034        1 => Ok(LogicalType::SmallInt),
8035        2 => Ok(LogicalType::Integer),
8036        3 => Ok(LogicalType::BigInt),
8037        4 => Ok(LogicalType::Varchar),
8038        5 => Ok(LogicalType::Date),
8039        6 => Ok(LogicalType::Timestamp),
8040        7 => Ok(LogicalType::Boolean),
8041        8 => Ok(LogicalType::TinyInt),
8042        9 => Ok(LogicalType::UTinyInt),
8043        10 => Ok(LogicalType::USmallInt),
8044        11 => Ok(LogicalType::UInteger),
8045        12 => Ok(LogicalType::UBigInt),
8046        14 => Ok(LogicalType::Float),
8047        15 => Ok(LogicalType::Double),
8048        16 => Ok(LogicalType::HugeInt),
8049        17 => Ok(LogicalType::UHugeInt),
8050        18 => Ok(LogicalType::Time),
8051        19 => Ok(LogicalType::TimeTz),
8052        20 => Ok(LogicalType::TimestampTz),
8053        21 => Ok(LogicalType::Interval),
8054        22 => Ok(LogicalType::Uuid),
8055        23 => Ok(LogicalType::Blob),
8056        24 => Ok(LogicalType::Bit),
8057        25 => Ok(LogicalType::TimestampS),
8058        26 => Ok(LogicalType::TimestampMs),
8059        27 => Ok(LogicalType::TimestampNs),
8060        _ => Err(invalid("column type tag is unknown")),
8061    }
8062}
8063
8064fn put_u16(out: &mut Vec<u8>, value: u16) {
8065    out.extend_from_slice(&value.to_le_bytes());
8066}
8067fn put_u32(out: &mut Vec<u8>, value: u32) {
8068    out.extend_from_slice(&value.to_le_bytes());
8069}
8070fn put_u64(out: &mut Vec<u8>, value: u64) {
8071    out.extend_from_slice(&value.to_le_bytes());
8072}
8073fn put_var_u64(out: &mut Vec<u8>, mut value: u64) {
8074    while value >= 0x80 {
8075        out.push((value as u8 & 0x7f) | 0x80);
8076        value >>= 7;
8077    }
8078    out.push(value as u8);
8079}
8080
8081fn frequency_order(left: FrequencyValue, right: FrequencyValue) -> Ordering {
8082    match (left, right) {
8083        (FrequencyValue::Null, FrequencyValue::Null) => Ordering::Equal,
8084        (FrequencyValue::Null, _) => Ordering::Less,
8085        (_, FrequencyValue::Null) => Ordering::Greater,
8086        (FrequencyValue::Integer(left), FrequencyValue::Integer(right)) => left.cmp(&right),
8087        (FrequencyValue::Code(left), FrequencyValue::Code(right)) => left.cmp(&right),
8088        (FrequencyValue::Integer(_), FrequencyValue::Code(_)) => Ordering::Less,
8089        (FrequencyValue::Code(_), FrequencyValue::Integer(_)) => Ordering::Greater,
8090    }
8091}
8092
8093/// Leaves the [`FREQUENCY_ENTRIES`] commonest entries in order and says what the next one counted.
8094///
8095/// There is one entry a distinct value, so on `URL` this is handed two and a quarter million of
8096/// them and keeps five hundred and twelve. Sorting all of them to throw almost all of them away is
8097/// the whole of what counting a dictionary column used to cost, 2.13 seconds of it on `URL` at eight
8098/// million rows against 11.93 for compressing the same column's values.
8099///
8100/// Partitioning answers both questions instead. It puts the five hundred and thirteenth entry where
8101/// it belongs and everything commoner in front of it, which is the entries to keep and the count to
8102/// report as the largest one omitted, and then only the part that survives is sorted. The order that
8103/// comes out is the order the sort gave, because the tie break makes the comparison total: two
8104/// entries never hold the same value.
8105fn keep_most_frequent(entries: &mut Vec<FrequencyEntry>) -> u64 {
8106    let order = |left: &FrequencyEntry, right: &FrequencyEntry| {
8107        right.count.cmp(&left.count).then_with(|| frequency_order(left.value, right.value))
8108    };
8109    let omitted_max = if entries.len() > FREQUENCY_ENTRIES {
8110        let (_, next, _) = entries.select_nth_unstable_by(FREQUENCY_ENTRIES, order);
8111        let omitted_max = next.count;
8112        entries.truncate(FREQUENCY_ENTRIES);
8113        omitted_max
8114    } else {
8115        0
8116    };
8117    entries.sort_unstable_by(order);
8118    omitted_max
8119}
8120
8121fn code_frequency(
8122    dictionary: &GlobalDictionary,
8123    flat: &[u8],
8124    bases: &[u64],
8125) -> Result<(FrequencySummary, Vec<Option<Vec<u8>>>)> {
8126    let mut entries = dictionary
8127        .counts
8128        .iter()
8129        .enumerate()
8130        .filter(|(_, count)| **count != 0)
8131        .map(|(code, &count)| FrequencyEntry { value: FrequencyValue::Code(code as u32), count })
8132        .collect::<Vec<_>>();
8133    if dictionary.nulls != 0 {
8134        entries.push(FrequencyEntry { value: FrequencyValue::Null, count: dictionary.nulls });
8135    }
8136    let omitted_max = keep_most_frequent(&mut entries);
8137    let mut spans = Vec::with_capacity(entries.len());
8138    let mut text_bytes = 0_usize;
8139    for entry in &entries {
8140        let span = match entry.value {
8141            FrequencyValue::Code(code) => {
8142                let span = GlobalDictionary::value_span(&dictionary.ends, bases, code as usize);
8143                let bytes = flat
8144                    .get(span.0..span.1)
8145                    .ok_or_else(|| invalid("a frequency code is outside its dictionary"))?;
8146                text_bytes = text_bytes.saturating_add(bytes.len());
8147                Some(span)
8148            }
8149            FrequencyValue::Null | FrequencyValue::Integer(_) => None,
8150        };
8151        spans.push(span);
8152    }
8153    let texts = if text_bytes > FREQUENCY_TEXT_BUDGET {
8154        Vec::new()
8155    } else {
8156        spans.into_iter().map(|span| span.map(|(from, to)| flat[from..to].to_vec())).collect()
8157    };
8158    Ok((
8159        FrequencySummary {
8160            entries,
8161            omitted_max,
8162            ordinals: Vec::new(),
8163            ordinal_entries: Vec::new(),
8164        },
8165        texts,
8166    ))
8167}
8168
8169fn encode_directory(table: &Table) -> Result<Vec<u8>> {
8170    let mut out = DIRECTORY.to_vec();
8171    let name = table.name.as_bytes();
8172    put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("table name too long"))?);
8173    out.extend_from_slice(name);
8174    put_u16(&mut out, u16::try_from(table.fields.len()).map_err(|_| invalid("too many columns"))?);
8175    for field in &table.fields {
8176        let name = field.name.as_bytes();
8177        put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("column name too long"))?);
8178        out.extend_from_slice(name);
8179        put_type(&mut out, &field.ty)?;
8180        out.push(u8::from(field.not_null));
8181    }
8182    for (field, dictionary) in table.fields.iter().zip(&table.dictionaries) {
8183        match dictionary {
8184            None => out.push(0),
8185            Some(page) => {
8186                out.push(dictionary_tag(&field.ty));
8187                put_u64(&mut out, page.offset);
8188                put_u32(&mut out, page.length);
8189                put_u64(&mut out, page.hash);
8190            }
8191        }
8192    }
8193    for distinct in &table.distincts {
8194        match distinct {
8195            None => out.push(0),
8196            Some(count) => {
8197                out.push(1);
8198                put_u64(&mut out, *count);
8199            }
8200        }
8201    }
8202    put_u64(&mut out, u64::try_from(table.rows).map_err(|_| invalid("row count overflow"))?);
8203    put_u32(&mut out, u32::try_from(table.stripes.len()).map_err(|_| invalid("too many stripes"))?);
8204    for stripe in &table.stripes {
8205        put_u32(
8206            &mut out,
8207            u32::try_from(stripe.parts.len()).map_err(|_| invalid("too many parts in a stripe"))?,
8208        );
8209        for &rows in &stripe.parts {
8210            put_u32(&mut out, rows);
8211        }
8212        put_u64(&mut out, stripe.index.offset);
8213        put_u32(&mut out, stripe.index.length);
8214        for page in &stripe.pages {
8215            put_u64(&mut out, page.offset);
8216            put_u32(&mut out, page.length);
8217        }
8218        // A membership index says which of a dictionary's codes a part holds, so a column the writer
8219        // decided against giving a dictionary has nothing for it to be about and writes none. Every
8220        // file written before that decision existed has a dictionary on every varchar column, so
8221        // this reads those files byte for byte the way it always did.
8222        for (column, ((field, dictionary), membership)) in
8223            table.fields.iter().zip(&table.dictionaries).zip(stripe.memberships.slots()).enumerate()
8224        {
8225            if !coded_type(&field.ty) || dictionary.is_none() {
8226                continue;
8227            }
8228            let page = match membership {
8229                Some(page) => page,
8230                None if table.demoted.get(column).copied().unwrap_or(false) => {
8231                    Page { offset: HEADER, length: 0, hash: 0 }
8232                }
8233                None => return Err(invalid("string page has no code membership index")),
8234            };
8235            put_u64(&mut out, page.offset);
8236            put_u32(&mut out, page.length);
8237            put_u64(&mut out, page.hash);
8238        }
8239        for sieve in stripe.sieves.slots() {
8240            match sieve {
8241                None => out.push(0),
8242                Some(page) => {
8243                    out.push(1);
8244                    put_u64(&mut out, page.offset);
8245                    put_u32(&mut out, page.length);
8246                    put_u64(&mut out, page.hash);
8247                }
8248            }
8249        }
8250        for held in stripe.part_ranges.slots() {
8251            match held {
8252                None => out.push(0),
8253                Some(page) => {
8254                    out.push(1);
8255                    put_u64(&mut out, page.offset);
8256                    put_u32(&mut out, page.length);
8257                    put_u64(&mut out, page.hash);
8258                }
8259            }
8260        }
8261        for range in stripe.zone.columns() {
8262            put_bound(&mut out, range.low.as_ref())?;
8263            put_bound(&mut out, range.high.as_ref())?;
8264            put_u32(
8265                &mut out,
8266                u32::try_from(range.nulls).map_err(|_| invalid("null count overflow"))?,
8267            );
8268            out.push(u8::from(range.exact));
8269            match range.sum {
8270                None => out.push(0),
8271                Some(total) => {
8272                    out.push(1);
8273                    out.extend_from_slice(&total.to_le_bytes());
8274                }
8275            }
8276        }
8277    }
8278    out.extend_from_slice(FREQUENCIES_SPANS);
8279    put_u16(
8280        &mut out,
8281        u16::try_from(table.frequencies.len())
8282            .map_err(|_| invalid("too many frequency columns"))?,
8283    );
8284    for summary in &table.frequencies {
8285        let summary = match summary {
8286            None => {
8287                put_u32(&mut out, 0);
8288                put_u32(&mut out, 0);
8289                continue;
8290            }
8291            Some(Frequencies::Held(summary)) => summary,
8292            // Only a reader leaves a synopsis in the file, and nothing writes a reader's table back.
8293            Some(Frequencies::Stored { .. }) => {
8294                return Err(invalid("a synopsis left in the file cannot be written back"));
8295            }
8296        };
8297        let length_at = out.len();
8298        put_u32(&mut out, 0);
8299        put_u32(
8300            &mut out,
8301            u32::try_from(summary.entries.len())
8302                .map_err(|_| invalid("too many frequency entries"))?,
8303        );
8304        let start = out.len();
8305        out.push(1);
8306        put_u64(&mut out, summary.omitted_max);
8307        put_u32(
8308            &mut out,
8309            u32::try_from(summary.entries.len())
8310                .map_err(|_| invalid("too many frequency entries"))?,
8311        );
8312        for entry in &summary.entries {
8313            match entry.value {
8314                FrequencyValue::Null => out.push(0),
8315                FrequencyValue::Integer(value) => {
8316                    out.push(1);
8317                    out.extend_from_slice(&value.to_le_bytes());
8318                }
8319                FrequencyValue::Code(value) => {
8320                    out.push(2);
8321                    put_u32(&mut out, value);
8322                }
8323            }
8324            put_u64(&mut out, entry.count);
8325        }
8326        put_u32(
8327            &mut out,
8328            u32::try_from(summary.ordinals.len())
8329                .map_err(|_| invalid("too many frequency ordinals"))?,
8330        );
8331        let mut previous = 0_u64;
8332        for (at, &ordinal) in summary.ordinals.iter().enumerate() {
8333            let delta = if at == 0 {
8334                ordinal
8335            } else {
8336                ordinal
8337                    .checked_sub(previous)
8338                    .ok_or_else(|| invalid("frequency ordinals are not ordered"))?
8339            };
8340            if at != 0 && delta == 0 {
8341                return Err(invalid("frequency ordinals are not unique"));
8342            }
8343            put_var_u64(&mut out, delta);
8344            previous = ordinal;
8345        }
8346        if summary.ordinal_entries.len() != summary.ordinals.len() {
8347            return Err(invalid("frequency ordinal values have a different length"));
8348        }
8349        for &entry in &summary.ordinal_entries {
8350            if entry as usize >= summary.entries.len() {
8351                return Err(invalid("frequency ordinal value is outside its entries"));
8352            }
8353            put_u16(&mut out, entry);
8354        }
8355        let length = u32::try_from(out.len() - start)
8356            .map_err(|_| invalid("a frequency synopsis is too long"))?;
8357        out[length_at..length_at + 4].copy_from_slice(&length.to_le_bytes());
8358    }
8359    if !table.pair_frequencies.is_empty() {
8360        out.extend_from_slice(PAIR_FREQUENCIES);
8361        put_u16(
8362            &mut out,
8363            u16::try_from(table.pair_frequencies.len())
8364                .map_err(|_| invalid("too many pair frequency summaries"))?,
8365        );
8366        for summary in &table.pair_frequencies {
8367            put_u16(&mut out, summary.first);
8368            put_u16(&mut out, summary.second);
8369            put_u64(&mut out, summary.omitted_max);
8370            put_u16(
8371                &mut out,
8372                u16::try_from(summary.entries.len())
8373                    .map_err(|_| invalid("too many pair frequency entries"))?,
8374            );
8375            for entry in &summary.entries {
8376                put_u16(&mut out, entry.first_entry);
8377                match entry.second {
8378                    None => out.push(0),
8379                    Some(code) => {
8380                        out.push(1);
8381                        put_u32(&mut out, code);
8382                    }
8383                }
8384                put_u64(&mut out, entry.count);
8385            }
8386        }
8387    }
8388    let text_columns = table.frequency_texts.iter().filter(|texts| !texts.is_empty()).count();
8389    if text_columns != 0 {
8390        out.extend_from_slice(FREQUENCY_TEXTS);
8391        put_u16(
8392            &mut out,
8393            u16::try_from(text_columns)
8394                .map_err(|_| invalid("too many string frequency columns"))?,
8395        );
8396        for (column, texts) in table.frequency_texts.iter().enumerate() {
8397            if texts.is_empty() {
8398                continue;
8399            }
8400            put_u16(
8401                &mut out,
8402                u16::try_from(column).map_err(|_| invalid("frequency text column overflows"))?,
8403            );
8404            put_u16(
8405                &mut out,
8406                u16::try_from(texts.len())
8407                    .map_err(|_| invalid("too many frequency text entries"))?,
8408            );
8409            for text in texts {
8410                match text {
8411                    None => out.push(0),
8412                    Some(text) => {
8413                        out.push(1);
8414                        put_u32(
8415                            &mut out,
8416                            u32::try_from(text.len())
8417                                .map_err(|_| invalid("frequency text is too long"))?,
8418                        );
8419                        out.extend_from_slice(text);
8420                    }
8421                }
8422            }
8423        }
8424    }
8425    if let Some(summary) = &table.host_groups {
8426        out.extend_from_slice(HOST_GROUPS);
8427        put_u16(
8428            &mut out,
8429            u16::try_from(summary.column).map_err(|_| invalid("host column overflows"))?,
8430        );
8431        put_u64(&mut out, summary.omitted_max);
8432        put_u16(
8433            &mut out,
8434            u16::try_from(summary.entries.len()).map_err(|_| invalid("too many host groups"))?,
8435        );
8436        for entry in &summary.entries {
8437            put_u32(
8438                &mut out,
8439                u32::try_from(entry.host.len()).map_err(|_| invalid("host name is too long"))?,
8440            );
8441            out.extend_from_slice(entry.host.as_bytes());
8442            put_u64(&mut out, entry.count);
8443            out.extend_from_slice(&entry.bytes_sum.to_le_bytes());
8444            put_u32(
8445                &mut out,
8446                u32::try_from(entry.minimum.len())
8447                    .map_err(|_| invalid("host minimum is too long"))?,
8448            );
8449            out.extend_from_slice(entry.minimum.as_bytes());
8450        }
8451    }
8452    // Written only when there is a declaration, so that the common file is the same bytes it was
8453    // and the section is not a byte of zero on every table in the world that never asked for one.
8454    if let Some(clustering) = &table.clustering {
8455        out.extend_from_slice(CLUSTERING);
8456        out.push(clustering.width().tag());
8457        put_u16(
8458            &mut out,
8459            u16::try_from(clustering.columns().len())
8460                .map_err(|_| invalid("too many clustering columns"))?,
8461        );
8462        for &column in clustering.columns() {
8463            put_u16(
8464                &mut out,
8465                u16::try_from(column).map_err(|_| invalid("clustering column index overflow"))?,
8466            );
8467        }
8468    }
8469    let demoted = (0..table.fields.len())
8470        .filter(|&column| table.demoted.get(column).copied().unwrap_or(false))
8471        .collect::<Vec<_>>();
8472    if !demoted.is_empty() {
8473        out.extend_from_slice(DEMOTED);
8474        put_u16(
8475            &mut out,
8476            u16::try_from(demoted.len()).map_err(|_| invalid("too many demoted columns"))?,
8477        );
8478        for column in demoted {
8479            put_u16(
8480                &mut out,
8481                u16::try_from(column).map_err(|_| invalid("demoted column index overflow"))?,
8482            );
8483        }
8484    }
8485    if !table.constraints.is_empty() {
8486        out.extend_from_slice(KEYS);
8487        put_count(&mut out, table.constraints.keys.len())?;
8488        for (columns, primary) in &table.constraints.keys {
8489            out.push(u8::from(*primary));
8490            put_columns(&mut out, columns)?;
8491        }
8492        put_count(&mut out, table.constraints.foreign.len())?;
8493        for foreign in &table.constraints.foreign {
8494            put_columns(&mut out, &foreign.columns)?;
8495            put_columns(&mut out, &foreign.referenced)?;
8496            put_u32(
8497                &mut out,
8498                u32::try_from(foreign.table.len())
8499                    .map_err(|_| invalid("table name is too long"))?,
8500            );
8501            out.extend_from_slice(foreign.table.as_bytes());
8502        }
8503    }
8504    // The section table, last, behind its own magic, for the same reason the frequency block is
8505    // behind its own: a reader that stops before it gets a table with no sections, and a table with
8506    // no sections is a correct table. The one difference from the blocks before it is that this one
8507    // is written even when it is empty, so that a file written by this build always says which
8508    // sections it has rather than leaving a reader to infer it from where the bytes ran out.
8509    out.extend_from_slice(SECTIONS);
8510    put_u64(&mut out, table.generation);
8511    put_u16(
8512        &mut out,
8513        u16::try_from(table.sections.len()).map_err(|_| invalid("too many sections"))?,
8514    );
8515    for held in &table.sections {
8516        held.encode(&mut out)?;
8517    }
8518    if table.dictionary_payloads.iter().any(|&bytes| bytes != 0) {
8519        out.extend_from_slice(DICTIONARY_PAYLOADS);
8520        put_u16(
8521            &mut out,
8522            u16::try_from(table.fields.len()).map_err(|_| invalid("too many columns"))?,
8523        );
8524        for at in 0..table.fields.len() {
8525            put_u64(&mut out, table.dictionary_payloads.get(at).copied().unwrap_or(0));
8526        }
8527    }
8528    Ok(out)
8529}
8530
8531/// The small level of the directory, naming every table in the file.
8532///
8533/// This is what a footer slot points at. Each entry carries its own checksum over its table
8534/// directory, so a table whose directory is torn is found when that table is first touched rather
8535/// than being trusted because the catalog around it checksummed.
8536///
8537/// The views go after the tables and are whole here, since a view is text and a column list and has
8538/// no pages for a second level to point at.
8539fn signed_integer(ty: &LogicalType) -> bool {
8540    matches!(
8541        ty,
8542        LogicalType::TinyInt | LogicalType::SmallInt | LogicalType::Integer | LogicalType::BigInt
8543    )
8544}
8545
8546fn integer_or_date(ty: &LogicalType) -> bool {
8547    matches!(
8548        ty,
8549        LogicalType::TinyInt
8550            | LogicalType::SmallInt
8551            | LogicalType::Integer
8552            | LogicalType::BigInt
8553            | LogicalType::UTinyInt
8554            | LogicalType::USmallInt
8555            | LogicalType::UInteger
8556            | LogicalType::UBigInt
8557            | LogicalType::Date
8558    )
8559}
8560
8561fn table_integer_extremes(table: &Table) -> Vec<StoredIntegerExtremes> {
8562    table
8563        .fields
8564        .iter()
8565        .enumerate()
8566        .map(|(column, field)| {
8567            if !integer_or_date(&field.ty) {
8568                return None;
8569            }
8570            let mut low: Option<i128> = None;
8571            let mut high: Option<i128> = None;
8572            for stripe in &table.stripes {
8573                let range = stripe.zone.column(column)?;
8574                if !range.exact {
8575                    return None;
8576                }
8577                match (range.low.as_ref(), range.high.as_ref()) {
8578                    (Some(Bound::Int(small)), Some(Bound::Int(large))) => {
8579                        low = Some(low.map_or(*small, |held| held.min(*small)));
8580                        high = Some(high.map_or(*large, |held| held.max(*large)));
8581                    }
8582                    (None, None) if stripe.rows == range.nulls => {}
8583                    _ => return None,
8584                }
8585            }
8586            Some(low.zip(high))
8587        })
8588        .collect()
8589}
8590
8591fn reader_integer_extremes(reader: &Reader) -> Result<Vec<StoredIntegerExtremes>> {
8592    reader
8593        .table
8594        .fields
8595        .iter()
8596        .enumerate()
8597        .map(|(column, field)| {
8598            if !integer_or_date(&field.ty) {
8599                return Ok(None);
8600            }
8601            match reader.exact_extremes(column)? {
8602                Some((Bound::Int(low), Bound::Int(high))) => Ok(Some(Some((low, high)))),
8603                None if reader.null_count(column)? == reader.table.rows as u64 => Ok(Some(None)),
8604                _ => Ok(None),
8605            }
8606        })
8607        .collect()
8608}
8609
8610fn table_complete_numeric_frequencies(table: &Table) -> Vec<StoredNumericFrequencies> {
8611    table
8612        .fields
8613        .iter()
8614        .enumerate()
8615        .map(|(column, field)| {
8616            if !integer_or_date(&field.ty) {
8617                return None;
8618            }
8619            let Some(Frequencies::Held(summary)) = table.frequencies.get(column)?.as_ref() else {
8620                return None;
8621            };
8622            if summary.omitted_max != 0 || summary.entries.len() > MAX_CATALOG_FREQUENCIES {
8623                return None;
8624            }
8625            let entries = summary
8626                .entries
8627                .iter()
8628                .map(|entry| {
8629                    let value = match entry.value {
8630                        FrequencyValue::Null => None,
8631                        FrequencyValue::Integer(value) => Some(value),
8632                        FrequencyValue::Code(_) => return None,
8633                    };
8634                    Some((value, entry.count))
8635                })
8636                .collect::<Option<Vec<_>>>()?;
8637            let rows = entries.iter().try_fold(0_u64, |sum, (_, count)| sum.checked_add(*count))?;
8638            (rows == table.rows as u64).then_some(entries)
8639        })
8640        .collect()
8641}
8642
8643/// The sixty four bits the close keys a numeric column's frequencies by, for a value the writer's
8644/// tally held.
8645///
8646/// The same bits [`Writer::visit_numeric`] hands over: a signed value sign extended to `i64`, and an
8647/// unsigned one as it is.
8648/// The value a column's sixty four bits stand for, read as signed or unsigned the way the column is.
8649fn integer_value(bits: u64, signed: bool) -> FrequencyValue {
8650    if signed {
8651        FrequencyValue::Integer(i128::from(bits as i64))
8652    } else {
8653        FrequencyValue::Integer(i128::from(bits))
8654    }
8655}
8656
8657fn frequency_bits(value: &Value) -> Option<u64> {
8658    Some(match value {
8659        Value::TinyInt(value) => i64::from(*value) as u64,
8660        Value::SmallInt(value) => i64::from(*value) as u64,
8661        Value::Integer(value) | Value::Date(value) => i64::from(*value) as u64,
8662        Value::BigInt(value) | Value::Timestamp(value) => *value as u64,
8663        Value::UTinyInt(value) => u64::from(*value),
8664        Value::USmallInt(value) => u64::from(*value),
8665        Value::UInteger(value) => u64::from(*value),
8666        Value::UBigInt(value) => *value,
8667        _ => return None,
8668    })
8669}
8670
8671fn numeric_frequency_value(value: &Value) -> Option<Option<i128>> {
8672    Some(match value {
8673        Value::Null => None,
8674        Value::TinyInt(value) => Some(i128::from(*value)),
8675        Value::SmallInt(value) => Some(i128::from(*value)),
8676        Value::Integer(value) | Value::Date(value) => Some(i128::from(*value)),
8677        Value::BigInt(value) => Some(i128::from(*value)),
8678        Value::UTinyInt(value) => Some(i128::from(*value)),
8679        Value::USmallInt(value) => Some(i128::from(*value)),
8680        Value::UInteger(value) => Some(i128::from(*value)),
8681        Value::UBigInt(value) => Some(i128::from(*value)),
8682        _ => return None,
8683    })
8684}
8685
8686fn reader_complete_numeric_frequencies(reader: &Reader) -> Result<Vec<StoredNumericFrequencies>> {
8687    reader
8688        .table
8689        .fields
8690        .iter()
8691        .enumerate()
8692        .map(|(column, field)| {
8693            if !integer_or_date(&field.ty) {
8694                return Ok(None);
8695            }
8696            let Some(summary) = reader.frequency_summary(column)? else { return Ok(None) };
8697            if summary.omitted_max != 0 || summary.entries.len() > MAX_CATALOG_FREQUENCIES {
8698                return Ok(None);
8699            }
8700            let entries = reader.decode_frequencies(column, &field.ty, &summary.entries)?;
8701            let Some(entries) = entries
8702                .iter()
8703                .map(|(value, count)| Some((numeric_frequency_value(value)?, *count)))
8704                .collect::<Option<Vec<_>>>()
8705            else {
8706                return Ok(None);
8707            };
8708            let rows = entries.iter().try_fold(0_u64, |sum, (_, count)| sum.checked_add(*count));
8709            Ok((rows == Some(reader.table.rows as u64)).then_some(entries))
8710        })
8711        .collect()
8712}
8713
8714fn table_exact_sum(table: &Table, column: usize) -> Option<(i128, u64)> {
8715    table.stripes.iter().try_fold((0_i128, 0_u64), |(sum, count), stripe| {
8716        let range = stripe.zone.column(column)?;
8717        let sum = sum.checked_add(range.sum?)?;
8718        let nonnull = (stripe.rows as u64).checked_sub(range.nulls as u64)?;
8719        Some((sum, count.checked_add(nonnull)?))
8720    })
8721}
8722
8723fn table_aggregate_sums(table: &Table) -> Vec<Option<(i128, u64)>> {
8724    table
8725        .fields
8726        .iter()
8727        .enumerate()
8728        .map(|(column, field)| {
8729            signed_integer(&field.ty).then(|| table_exact_sum(table, column)).flatten()
8730        })
8731        .collect()
8732}
8733
8734fn reader_aggregate_sums(reader: &Reader) -> Result<Vec<Option<(i128, u64)>>> {
8735    reader
8736        .table
8737        .fields
8738        .iter()
8739        .enumerate()
8740        .map(
8741            |(column, field)| {
8742                if signed_integer(&field.ty) { reader.exact_sum(column) } else { Ok(None) }
8743            },
8744        )
8745        .collect()
8746}
8747
8748fn encode_catalog(
8749    entries: &[Entry],
8750    views: &[ViewEntry],
8751    card: Option<&KeptCard>,
8752) -> Result<Vec<u8>> {
8753    let mut out = CATALOG.to_vec();
8754    put_u32(&mut out, u32::try_from(entries.len()).map_err(|_| invalid("too many tables"))?);
8755    for entry in entries {
8756        let name = entry.name.as_bytes();
8757        put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("table name too long"))?);
8758        out.extend_from_slice(name);
8759        put_u64(&mut out, u64::try_from(entry.rows).map_err(|_| invalid("row count overflow"))?);
8760        put_u16(
8761            &mut out,
8762            u16::try_from(entry.fields.len()).map_err(|_| invalid("too many columns"))?,
8763        );
8764        for field in &entry.fields {
8765            let name = field.name.as_bytes();
8766            put_u16(
8767                &mut out,
8768                u16::try_from(name.len()).map_err(|_| invalid("column name too long"))?,
8769            );
8770            out.extend_from_slice(name);
8771            put_type(&mut out, &field.ty)?;
8772            out.push(u8::from(field.not_null));
8773        }
8774        put_u64(&mut out, entry.directory.offset);
8775        put_u32(&mut out, entry.directory.length);
8776        put_u64(&mut out, entry.directory.hash);
8777    }
8778    put_u32(&mut out, u32::try_from(views.len()).map_err(|_| invalid("too many views"))?);
8779    for view in views {
8780        let name = view.name.as_bytes();
8781        put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("view name too long"))?);
8782        out.extend_from_slice(name);
8783        put_long_text(&mut out, &view.sql, "view body")?;
8784        put_long_text(&mut out, &view.statement, "view statement")?;
8785        put_u16(
8786            &mut out,
8787            u16::try_from(view.aliases.len()).map_err(|_| invalid("too many aliases"))?,
8788        );
8789        for alias in &view.aliases {
8790            let alias = alias.as_bytes();
8791            put_u16(
8792                &mut out,
8793                u16::try_from(alias.len()).map_err(|_| invalid("alias name too long"))?,
8794            );
8795            out.extend_from_slice(alias);
8796        }
8797        put_u16(
8798            &mut out,
8799            u16::try_from(view.columns.len()).map_err(|_| invalid("too many columns"))?,
8800        );
8801        for field in &view.columns {
8802            let name = field.name.as_bytes();
8803            put_u16(
8804                &mut out,
8805                u16::try_from(name.len()).map_err(|_| invalid("column name too long"))?,
8806            );
8807            out.extend_from_slice(name);
8808            put_type(&mut out, &field.ty)?;
8809            out.push(u8::from(field.not_null));
8810        }
8811    }
8812    out.extend_from_slice(NONZERO_COUNTS);
8813    for entry in entries {
8814        if entry.nonzero.len() != entry.fields.len() {
8815            return Err(invalid("nonzero count width differs from schema"));
8816        }
8817        for count in &entry.nonzero {
8818            match count {
8819                None => out.push(0),
8820                Some(count) => {
8821                    out.push(1);
8822                    put_u64(&mut out, *count);
8823                }
8824            }
8825        }
8826    }
8827    out.extend_from_slice(AGGREGATE_SUMS);
8828    for entry in entries {
8829        if entry.aggregates.len() != entry.fields.len() {
8830            return Err(invalid("aggregate sum width differs from schema"));
8831        }
8832        for summary in &entry.aggregates {
8833            match summary {
8834                None => out.push(0),
8835                Some((sum, count)) => {
8836                    out.push(1);
8837                    out.extend_from_slice(&sum.to_le_bytes());
8838                    put_u64(&mut out, *count);
8839                }
8840            }
8841        }
8842    }
8843    out.extend_from_slice(DISTINCT_COUNTS);
8844    for entry in entries {
8845        if entry.distincts.len() != entry.fields.len() {
8846            return Err(invalid("distinct count width differs from schema"));
8847        }
8848        for count in &entry.distincts {
8849            match count {
8850                None => out.push(0),
8851                Some(count) => {
8852                    if *count > entry.rows as u64 {
8853                        return Err(invalid("distinct count exceeds table rows"));
8854                    }
8855                    out.push(1);
8856                    put_u64(&mut out, *count);
8857                }
8858            }
8859        }
8860    }
8861    out.extend_from_slice(INTEGER_EXTREMES);
8862    for entry in entries {
8863        if entry.extremes.len() != entry.fields.len() {
8864            return Err(invalid("integer extremes width differs from schema"));
8865        }
8866        for (field, extremes) in entry.fields.iter().zip(&entry.extremes) {
8867            match extremes {
8868                None => out.push(0),
8869                Some(None) if integer_or_date(&field.ty) => out.push(1),
8870                Some(Some((low, high))) if integer_or_date(&field.ty) && low <= high => {
8871                    out.push(2);
8872                    out.extend_from_slice(&low.to_le_bytes());
8873                    out.extend_from_slice(&high.to_le_bytes());
8874                }
8875                _ => return Err(invalid("integer extremes type or range differs")),
8876            }
8877        }
8878    }
8879    out.extend_from_slice(COMPLETE_FREQUENCIES);
8880    for entry in entries {
8881        if entry.frequencies.len() != entry.fields.len() {
8882            return Err(invalid("numeric frequency width differs from schema"));
8883        }
8884        for (field, frequencies) in entry.fields.iter().zip(&entry.frequencies) {
8885            match frequencies {
8886                None => out.push(0),
8887                Some(entries)
8888                    if integer_or_date(&field.ty) && entries.len() <= MAX_CATALOG_FREQUENCIES =>
8889                {
8890                    let mut total = 0_u64;
8891                    for (at, (value, count)) in entries.iter().enumerate() {
8892                        if entries[..at].iter().any(|(held, _)| held == value) {
8893                            return Err(invalid("numeric frequency value repeats"));
8894                        }
8895                        total = total
8896                            .checked_add(*count)
8897                            .ok_or_else(|| invalid("numeric frequency count overflows"))?;
8898                    }
8899                    if total != entry.rows as u64 {
8900                        return Err(invalid("numeric frequencies do not cover table rows"));
8901                    }
8902                    out.push(1);
8903                    out.push(entries.len() as u8);
8904                    for (value, count) in entries {
8905                        match value {
8906                            None => out.push(0),
8907                            Some(value) => {
8908                                out.push(1);
8909                                out.extend_from_slice(&value.to_le_bytes());
8910                            }
8911                        }
8912                        put_u64(&mut out, *count);
8913                    }
8914                }
8915                _ => return Err(invalid("numeric frequency type or width differs")),
8916            }
8917        }
8918    }
8919    if let Some(card) = card {
8920        out.extend_from_slice(DEVICE_CARD);
8921        let device = card.device.as_bytes();
8922        put_u16(&mut out, u16::try_from(device.len()).map_err(|_| invalid("device id too long"))?);
8923        out.extend_from_slice(device);
8924        put_u32(&mut out, u32::try_from(card.bytes.len()).map_err(|_| invalid("card too long"))?);
8925        out.extend_from_slice(&card.bytes);
8926    }
8927    Ok(out)
8928}
8929
8930/// A length and that many bytes, for text that is allowed to be longer than a name.
8931fn put_long_text(out: &mut Vec<u8>, text: &str, what: &str) -> Result<()> {
8932    let bytes = text.as_bytes();
8933    put_u32(out, u32::try_from(bytes.len()).map_err(|_| invalid(&format!("{what} too long")))?);
8934    out.extend_from_slice(bytes);
8935    Ok(())
8936}
8937
8938/// Reads the catalog directory back, checking every span against the file before anything is
8939/// allocated for it.
8940fn decode_catalog(bytes: &[u8], size: u64) -> Result<Decoded> {
8941    let mut cur = Cursor::new(bytes);
8942    if cur.take(8)? != CATALOG {
8943        return Err(invalid("catalog magic differs"));
8944    }
8945    let count = cur.u32()? as usize;
8946    let mut entries: Vec<Entry> = Vec::with_capacity(count.min(1024));
8947    for _ in 0..count {
8948        let name = cur.text()?;
8949        let rows = usize::try_from(cur.u64()?).map_err(|_| invalid("row count does not fit"))?;
8950        let width = cur.u16()? as usize;
8951        let mut fields = Vec::with_capacity(width);
8952        for _ in 0..width {
8953            let name = cur.text()?;
8954            let ty = read_type(&mut cur)?;
8955            let not_null = match cur.u8()? {
8956                0 => false,
8957                1 => true,
8958                _ => return Err(invalid("nullability flag differs")),
8959            };
8960            fields.push(Field { name, ty, not_null });
8961        }
8962        let directory = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
8963        let end = directory
8964            .offset
8965            .checked_add(u64::from(directory.length))
8966            .ok_or_else(|| invalid("table directory offset overflow"))?;
8967        if directory.offset < HEADER
8968            || end > size
8969            || directory.length as usize > MAX_DIRECTORY
8970            || directory.length == 0
8971        {
8972            return Err(invalid("table directory range is outside the file"));
8973        }
8974        if entries.iter().any(|held| held.name == name) {
8975            return Err(invalid("two tables in the catalog have the same name"));
8976        }
8977        let nonzero = vec![None; fields.len()];
8978        let aggregates = vec![None; fields.len()];
8979        let distincts = vec![None; fields.len()];
8980        let extremes = vec![None; fields.len()];
8981        let frequencies = vec![None; fields.len()];
8982        entries.push(Entry {
8983            name,
8984            fields,
8985            rows,
8986            directory,
8987            nonzero,
8988            aggregates,
8989            distincts,
8990            extremes,
8991            frequencies,
8992        });
8993    }
8994    // A catalog that ends where the tables end is a catalog with no views in it, which is every
8995    // file written before format 25. That is why the count is allowed to be missing rather than
8996    // read as a zero that has to be there: an older file has nothing after the last table entry at
8997    // all, and [`READABLE`] says those files still open.
8998    let count = if cur.done() { 0 } else { cur.u32()? as usize };
8999    let mut views: Vec<ViewEntry> = Vec::with_capacity(count.min(1024));
9000    for _ in 0..count {
9001        let name = cur.text()?;
9002        let sql = cur.long_text()?;
9003        let statement = cur.long_text()?;
9004        let width = cur.u16()? as usize;
9005        let mut aliases = Vec::with_capacity(width);
9006        for _ in 0..width {
9007            aliases.push(cur.text()?);
9008        }
9009        let width = cur.u16()? as usize;
9010        let mut columns = Vec::with_capacity(width);
9011        for _ in 0..width {
9012            let name = cur.text()?;
9013            let ty = read_type(&mut cur)?;
9014            let not_null = match cur.u8()? {
9015                0 => false,
9016                1 => true,
9017                _ => return Err(invalid("nullability flag differs")),
9018            };
9019            columns.push(Field { name, ty, not_null });
9020        }
9021        // The same rule the tables above get, and for the same reason. Two entries under one name
9022        // is a catalog nothing can answer a lookup from, and finding that out here is better than
9023        // finding it out from whichever of the two a search happened to reach first.
9024        if views.iter().any(|held| held.name == name) {
9025            return Err(invalid("two views in the catalog have the same name"));
9026        }
9027        if entries.iter().any(|held| held.name == name) {
9028            return Err(invalid("a table and a view in the catalog have the same name"));
9029        }
9030        views.push(ViewEntry { name, sql, statement, aliases, columns });
9031    }
9032    if !cur.done() {
9033        if cur.take(8)? != NONZERO_COUNTS {
9034            return Err(invalid("catalog extension magic differs"));
9035        }
9036        for entry in &mut entries {
9037            for (field, count) in entry.fields.iter().zip(&mut entry.nonzero) {
9038                *count = match cur.u8()? {
9039                    0 => None,
9040                    1 if matches!(
9041                        field.ty,
9042                        LogicalType::TinyInt
9043                            | LogicalType::SmallInt
9044                            | LogicalType::Integer
9045                            | LogicalType::BigInt
9046                            | LogicalType::UTinyInt
9047                            | LogicalType::USmallInt
9048                            | LogicalType::UInteger
9049                            | LogicalType::UBigInt
9050                    ) =>
9051                    {
9052                        let value = cur.u64()?;
9053                        if value > entry.rows as u64 {
9054                            return Err(invalid("nonzero count exceeds rows"));
9055                        }
9056                        Some(value)
9057                    }
9058                    _ => return Err(invalid("nonzero count tag or column type differs")),
9059                };
9060            }
9061        }
9062    }
9063    if !cur.done() {
9064        if cur.take(8)? != AGGREGATE_SUMS {
9065            return Err(invalid("aggregate catalog extension magic differs"));
9066        }
9067        for entry in &mut entries {
9068            for (field, summary) in entry.fields.iter().zip(&mut entry.aggregates) {
9069                *summary = match cur.u8()? {
9070                    0 => None,
9071                    1 if signed_integer(&field.ty) => {
9072                        let sum = i128::from_le_bytes(
9073                            cur.take(16)?
9074                                .try_into()
9075                                .map_err(|_| invalid("aggregate sum is truncated"))?,
9076                        );
9077                        let count = cur.u64()?;
9078                        if count > entry.rows as u64 {
9079                            return Err(invalid("aggregate count exceeds table rows"));
9080                        }
9081                        Some((sum, count))
9082                    }
9083                    _ => return Err(invalid("aggregate sum tag or column type differs")),
9084                };
9085            }
9086        }
9087    }
9088    if !cur.done() {
9089        if cur.take(8)? != DISTINCT_COUNTS {
9090            return Err(invalid("distinct catalog extension magic differs"));
9091        }
9092        for entry in &mut entries {
9093            for count in &mut entry.distincts {
9094                *count = match cur.u8()? {
9095                    0 => None,
9096                    1 => {
9097                        let value = cur.u64()?;
9098                        if value > entry.rows as u64 {
9099                            return Err(invalid("distinct count exceeds table rows"));
9100                        }
9101                        Some(value)
9102                    }
9103                    _ => return Err(invalid("distinct count tag differs")),
9104                };
9105            }
9106        }
9107    }
9108    if !cur.done() {
9109        if cur.take(8)? != INTEGER_EXTREMES {
9110            return Err(invalid("integer extremes catalog extension magic differs"));
9111        }
9112        for entry in &mut entries {
9113            for (field, extremes) in entry.fields.iter().zip(&mut entry.extremes) {
9114                *extremes = match cur.u8()? {
9115                    0 => None,
9116                    1 if integer_or_date(&field.ty) => Some(None),
9117                    2 if integer_or_date(&field.ty) => {
9118                        let low = i128::from_le_bytes(
9119                            cur.take(16)?
9120                                .try_into()
9121                                .map_err(|_| invalid("minimum is truncated"))?,
9122                        );
9123                        let high = i128::from_le_bytes(
9124                            cur.take(16)?
9125                                .try_into()
9126                                .map_err(|_| invalid("maximum is truncated"))?,
9127                        );
9128                        if low > high {
9129                            return Err(invalid("integer extremes are reversed"));
9130                        }
9131                        Some(Some((low, high)))
9132                    }
9133                    _ => return Err(invalid("integer extremes tag or type differs")),
9134                };
9135            }
9136        }
9137    }
9138    if !cur.done() {
9139        if cur.take(8)? != COMPLETE_FREQUENCIES {
9140            return Err(invalid("numeric frequency catalog extension magic differs"));
9141        }
9142        for entry in &mut entries {
9143            for (field, frequencies) in entry.fields.iter().zip(&mut entry.frequencies) {
9144                *frequencies = match cur.u8()? {
9145                    0 => None,
9146                    1 if integer_or_date(&field.ty) => {
9147                        let len = cur.u8()? as usize;
9148                        if len > MAX_CATALOG_FREQUENCIES {
9149                            return Err(invalid("too many catalog numeric frequencies"));
9150                        }
9151                        let mut values = Vec::with_capacity(len);
9152                        let mut total = 0_u64;
9153                        for _ in 0..len {
9154                            let value = match cur.u8()? {
9155                                0 => None,
9156                                1 => Some(i128::from_le_bytes(cur.take(16)?.try_into().map_err(
9157                                    |_| invalid("numeric frequency value is truncated"),
9158                                )?)),
9159                                _ => return Err(invalid("numeric frequency value tag differs")),
9160                            };
9161                            if values.iter().any(|(held, _)| *held == value) {
9162                                return Err(invalid("numeric frequency value repeats"));
9163                            }
9164                            let count = cur.u64()?;
9165                            total = total
9166                                .checked_add(count)
9167                                .ok_or_else(|| invalid("numeric frequency count overflows"))?;
9168                            values.push((value, count));
9169                        }
9170                        if total != entry.rows as u64 {
9171                            return Err(invalid("numeric frequencies do not cover table rows"));
9172                        }
9173                        Some(values)
9174                    }
9175                    _ => return Err(invalid("numeric frequency tag or type differs")),
9176                };
9177            }
9178        }
9179    }
9180    let mut card = None;
9181    if !cur.done() {
9182        if cur.take(8)? != DEVICE_CARD {
9183            return Err(invalid("device card catalog extension magic differs"));
9184        }
9185        let device = cur.text()?;
9186        let len = cur.u32()? as usize;
9187        if len > MAX_CARD {
9188            return Err(invalid("device card is longer than any card"));
9189        }
9190        card = Some(KeptCard { device, bytes: cur.take(len)?.to_vec() });
9191    }
9192    if !cur.done() {
9193        return Err(invalid("catalog has trailing bytes"));
9194    }
9195    Ok((entries, views, card))
9196}
9197
9198/// What [`decode_catalog`] reads: the tables, the views and the device card.
9199type Decoded = (Vec<Entry>, Vec<ViewEntry>, Option<KeptCard>);
9200
9201/// The most a kept device card can take, which is many times what one holds.
9202const MAX_CARD: usize = 64 << 10;
9203
9204/// The device card a file keeps, as `rudb_io::device` encodes it, and the device it was measured
9205/// on.
9206///
9207/// `16-measurement.md` section 16.3 keeps the card in the file so that a process opening the file
9208/// does not measure the device again. It is only good on that device, so it carries the device id
9209/// and a file copied somewhere else keeps its card but nobody takes it.
9210#[derive(Debug, Clone, PartialEq, Eq)]
9211struct KeptCard {
9212    device: String,
9213    bytes: Vec<u8>,
9214}
9215
9216/// The directory a database file is in, which is the one its device card is about.
9217fn directory_of(path: &Path) -> &Path {
9218    path.parent().filter(|dir| !dir.as_os_str().is_empty()).unwrap_or(Path::new("."))
9219}
9220
9221/// The card the next commit of the file at `path` writes down.
9222///
9223/// The one this process has for the device the file is on when there is one, since it was either
9224/// measured here or read out of a file on the same device, and otherwise whatever the file already
9225/// kept. A file never makes a process measure: the card is measured when something asks for it,
9226/// and this only writes down what is already known.
9227fn card_for(path: &Path, held: Option<KeptCard>) -> Option<KeptCard> {
9228    let Ok(device) = rudb_io::device::device_key(directory_of(path)) else {
9229        return held;
9230    };
9231    match rudb_io::device::kept(&device) {
9232        Some(card) => Some(KeptCard { device, bytes: card.encode() }),
9233        None => held,
9234    }
9235}
9236
9237/// Hands the card a file kept to this process, when the file is still on the device it describes.
9238fn remember_card(path: &Path, card: Option<&KeptCard>) {
9239    let Some(card) = card else { return };
9240    let dir = directory_of(path);
9241    let Ok(device) = rudb_io::device::device_key(dir) else { return };
9242    if device != card.device {
9243        return;
9244    }
9245    if let Ok(decoded) = rudb_io::device::Card::decode(&card.bytes, dir) {
9246        rudb_io::device::remember(&device, decoded);
9247    }
9248}
9249
9250/// Reads the fields of a directory or a catalog in order, off bytes in memory or out of the file.
9251///
9252/// A catalog is small and is read whole. A table directory is not: at ten million rows of `hits` it
9253/// is nearly a megabyte, and holding that buffer while the table it describes is built out of it
9254/// put both at the peak of every query. Out of the file, the cursor holds one window of
9255/// [`DIRECTORY_WINDOW`] bytes and moves it forward as the fields are read, so what a directory
9256/// costs at open is what it decodes into and not that plus its own bytes.
9257struct Cursor<'a> {
9258    bytes: &'a [u8],
9259    at: usize,
9260    window: Option<Window<'a>>,
9261}
9262
9263/// The part of a directory in the file that a [`Cursor`] has read in.
9264struct Window<'a> {
9265    file: &'a File,
9266    offset: u64,
9267    length: usize,
9268    /// Where `held` starts, counted from the start of the directory.
9269    start: usize,
9270    held: Vec<u8>,
9271    /// How much to read at once, which is [`DIRECTORY_WINDOW`] outside the tests.
9272    size: usize,
9273}
9274
9275/// How much of a directory a cursor reading one out of the file holds at once.
9276const DIRECTORY_WINDOW: usize = 64 << 10;
9277
9278impl<'a> Cursor<'a> {
9279    fn new(bytes: &'a [u8]) -> Self {
9280        Self { bytes, at: 0, window: None }
9281    }
9282
9283    /// A cursor over `length` bytes of `file` from `offset`, which it reads a window at a time.
9284    fn over(file: &'a File, offset: u64, length: usize) -> Self {
9285        let window =
9286            Window { file, offset, length, start: 0, held: Vec::new(), size: DIRECTORY_WINDOW };
9287        Self { bytes: &[], at: 0, window: Some(window) }
9288    }
9289
9290    /// How many bytes the cursor walks in all.
9291    fn len(&self) -> usize {
9292        self.window.as_ref().map_or(self.bytes.len(), |window| window.length)
9293    }
9294
9295    /// Makes sure the next `len` bytes are in memory.
9296    fn ensure(&mut self, len: usize) -> Result<()> {
9297        let end = self.at.checked_add(len).ok_or_else(|| invalid("directory offset overflow"))?;
9298        if end > self.len() {
9299            return Err(invalid("directory is truncated"));
9300        }
9301        let Some(window) = &mut self.window else { return Ok(()) };
9302        if self.at < window.start || end > window.start + window.held.len() {
9303            let want = len.max(window.size).min(window.length - self.at);
9304            window.start = self.at;
9305            window.held.resize(want, 0);
9306            read_at(window.file, window.offset + self.at as u64, &mut window.held)?;
9307        }
9308        Ok(())
9309    }
9310
9311    /// `len` bytes from `at`, which [`Self::ensure`] has already brought in.
9312    fn held(&self, at: usize, len: usize) -> &[u8] {
9313        match &self.window {
9314            Some(window) => &window.held[at - window.start..at - window.start + len],
9315            None => &self.bytes[at..at + len],
9316        }
9317    }
9318
9319    /// The next `len` bytes, without moving past them.
9320    #[inline]
9321    fn peek(&mut self, len: usize) -> Result<&[u8]> {
9322        if self.window.is_none() {
9323            let bytes = self.bytes;
9324            return Ok(&bytes[self.at..self.end(len)?]);
9325        }
9326        self.ensure(len)?;
9327        Ok(self.held(self.at, len))
9328    }
9329
9330    /// The next `len` bytes, moving past them.
9331    ///
9332    /// Every data page is decoded through this, a byte or a word at a time, so a cursor over bytes
9333    /// already in memory takes them here and never reaches [`Self::ensure`]. With the window check
9334    /// on every call, q06 on TPC-H spent a seventh of its instructions in it.
9335    #[inline]
9336    fn take(&mut self, len: usize) -> Result<&[u8]> {
9337        if self.window.is_none() {
9338            let bytes = self.bytes;
9339            let (at, end) = (self.at, self.end(len)?);
9340            self.at = end;
9341            return Ok(&bytes[at..end]);
9342        }
9343        self.take_windowed(len)
9344    }
9345
9346    /// Moves over a checked field without reading its payload from a windowed directory.
9347    fn skip(&mut self, len: usize) -> Result<()> {
9348        let end = self.at.checked_add(len).ok_or_else(|| invalid("directory offset overflow"))?;
9349        if end > self.len() {
9350            return Err(invalid("directory is truncated"));
9351        }
9352        self.at = end;
9353        Ok(())
9354    }
9355
9356    fn skip_bound(&mut self) -> Result<()> {
9357        match self.u8()? {
9358            0 => Ok(()),
9359            1 => self.skip(16),
9360            2 => self.skip(8),
9361            3 => {
9362                let length = self.u32()? as usize;
9363                self.skip(length)
9364            }
9365            4 => self.skip(17),
9366            _ => Err(invalid("a stored bound has an unknown tag")),
9367        }
9368    }
9369
9370    /// Where `len` bytes from here end, when they end inside the bytes.
9371    #[inline]
9372    fn end(&self, len: usize) -> Result<usize> {
9373        let end = self.at.checked_add(len).ok_or_else(|| invalid("directory offset overflow"))?;
9374        if end > self.bytes.len() {
9375            return Err(invalid("directory is truncated"));
9376        }
9377        Ok(end)
9378    }
9379
9380    /// [`Self::take`] out of the file, a window at a time.
9381    #[inline(never)]
9382    fn take_windowed(&mut self, len: usize) -> Result<&[u8]> {
9383        self.ensure(len)?;
9384        self.at += len;
9385        Ok(self.held(self.at - len, len))
9386    }
9387    #[inline]
9388    fn u8(&mut self) -> Result<u8> {
9389        Ok(self.take(1)?[0])
9390    }
9391    #[inline]
9392    fn u16(&mut self) -> Result<u16> {
9393        Ok(u16::from_le_bytes(self.take(2)?.try_into().expect("two bytes")))
9394    }
9395    #[inline]
9396    fn u32(&mut self) -> Result<u32> {
9397        Ok(u32::from_le_bytes(self.take(4)?.try_into().expect("four bytes")))
9398    }
9399    #[inline]
9400    fn u64(&mut self) -> Result<u64> {
9401        Ok(u64::from_le_bytes(self.take(8)?.try_into().expect("eight bytes")))
9402    }
9403    fn var_u64(&mut self) -> Result<u64> {
9404        let mut value = 0_u64;
9405        for shift in (0..=63).step_by(7) {
9406            let byte = self.u8()?;
9407            let part = u64::from(byte & 0x7f);
9408            if shift == 63 && part > 1 {
9409                return Err(invalid("frequency ordinal varint overflows"));
9410            }
9411            value |= part << shift;
9412            if byte & 0x80 == 0 {
9413                return Ok(value);
9414            }
9415        }
9416        Err(invalid("frequency ordinal varint is too long"))
9417    }
9418    /// A zone map's end, in the layout `rudb_common::bounds` defines.
9419    ///
9420    /// The bytes are the ones this directory has written since format 10 and the codec moved to
9421    /// rank zero rather than being copied, because a column summary now writes the same two ends
9422    /// and two encodings of one type is how the two quietly stop agreeing.
9423    ///
9424    /// A bound's length is in the bound, so out of the file the cursor offers the codec a few bytes
9425    /// and offers it twice as many whenever it runs out before the directory does.
9426    fn bound(&mut self) -> Result<Option<Bound>> {
9427        let rest = self.len().saturating_sub(self.at);
9428        let mut want = 32;
9429        loop {
9430            let offered = self.peek(want.min(rest))?;
9431            let mut used = 0;
9432            match bounds::get(offered, &mut used) {
9433                Ok(bound) => {
9434                    self.at += used;
9435                    return Ok(bound);
9436                }
9437                Err(_) if want < rest => want *= 2,
9438                Err(error) => return Err(error),
9439            }
9440        }
9441    }
9442    fn text(&mut self) -> Result<String> {
9443        let len = self.u16()? as usize;
9444        String::from_utf8(self.take(len)?.to_vec()).map_err(|_| invalid("name is not UTF-8"))
9445    }
9446    /// Whether everything has been read, which is how a section that an older file does not have at
9447    /// all is told from one that is there and empty.
9448    fn done(&self) -> bool {
9449        self.at >= self.len()
9450    }
9451    /// The same, for text that is a query rather than a name.
9452    ///
9453    /// A name fits in sixteen bits of length and a view body does not have to. Nobody writes a 64
9454    /// kilobyte identifier by accident and people do write generated queries that long, and a view
9455    /// that could not be written down because its body was too big would be a limit invented here
9456    /// rather than one anything else in the engine has.
9457    fn long_text(&mut self) -> Result<String> {
9458        let len = self.u32()? as usize;
9459        String::from_utf8(self.take(len)?.to_vec()).map_err(|_| invalid("text is not UTF-8"))
9460    }
9461}
9462
9463/// One column's frequency synopsis, or `None` for a column that has none, checked against the column.
9464fn decode_summary(
9465    cur: &mut Cursor<'_>,
9466    field: &Field,
9467    rows: usize,
9468    values: bool,
9469) -> Result<Option<FrequencySummary>> {
9470    Ok(match cur.u8()? {
9471        0 => None,
9472        1 => {
9473            let omitted_max = cur.u64()?;
9474            let count = cur.u32()? as usize;
9475            if count > FREQUENCY_ENTRIES {
9476                return Err(invalid("frequency entry count exceeds its bound"));
9477            }
9478            let mut entries = Vec::with_capacity(count);
9479            // row at a time: directory decoding validates each persisted bounded frequency entry.
9480            for _ in 0..count {
9481                let value = match cur.u8()? {
9482                    0 => FrequencyValue::Null,
9483                    1 => FrequencyValue::Integer(i128::from_le_bytes(
9484                        cur.take(16)?.try_into().expect("sixteen bytes"),
9485                    )),
9486                    2 => FrequencyValue::Code(cur.u32()?),
9487                    _ => return Err(invalid("frequency value tag differs")),
9488                };
9489                let valid = matches!(
9490                    (&field.ty, value),
9491                    (_, FrequencyValue::Null)
9492                        | (LogicalType::Varchar | LogicalType::Blob, FrequencyValue::Code(_))
9493                        | (
9494                            LogicalType::TinyInt
9495                                | LogicalType::SmallInt
9496                                | LogicalType::Integer
9497                                | LogicalType::BigInt
9498                                | LogicalType::UTinyInt
9499                                | LogicalType::USmallInt
9500                                | LogicalType::UInteger
9501                                | LogicalType::UBigInt
9502                                | LogicalType::Date
9503                                | LogicalType::Timestamp,
9504                            FrequencyValue::Integer(_),
9505                        )
9506                );
9507                if !valid {
9508                    return Err(invalid("frequency value does not match its column"));
9509                }
9510                let count = cur.u64()?;
9511                if count == 0 || count > rows as u64 {
9512                    return Err(invalid("frequency count is outside the table"));
9513                }
9514                entries.push(FrequencyEntry { value, count });
9515            }
9516            if entries.windows(2).any(|pair| pair[0].count < pair[1].count) {
9517                return Err(invalid("frequency entries are not descending"));
9518            }
9519            let ordinals = {
9520                let ordinal_count = cur.u32()? as usize;
9521                if ordinal_count > FREQUENCY_ORDINALS || ordinal_count > rows {
9522                    return Err(invalid("frequency ordinal count exceeds its bound"));
9523                }
9524                let mut ordinals = Vec::with_capacity(ordinal_count);
9525                let mut previous = 0_u64;
9526                for at in 0..ordinal_count {
9527                    let delta = cur.var_u64()?;
9528                    if at != 0 && delta == 0 {
9529                        return Err(invalid("frequency ordinals are not increasing"));
9530                    }
9531                    let ordinal = if at == 0 {
9532                        delta
9533                    } else {
9534                        previous
9535                            .checked_add(delta)
9536                            .ok_or_else(|| invalid("frequency ordinal overflows"))?
9537                    };
9538                    if ordinal >= rows as u64 {
9539                        return Err(invalid("frequency ordinal is outside the table"));
9540                    }
9541                    ordinals.push(ordinal);
9542                    previous = ordinal;
9543                }
9544                ordinals
9545            };
9546            let ordinal_entries = if values {
9547                let mut ordinal_entries = Vec::with_capacity(ordinals.len());
9548                for _ in 0..ordinals.len() {
9549                    let entry = cur.u16()?;
9550                    if entry as usize >= entries.len() {
9551                        return Err(invalid("frequency ordinal value is outside its entries"));
9552                    }
9553                    ordinal_entries.push(entry);
9554                }
9555                ordinal_entries
9556            } else {
9557                Vec::new()
9558            };
9559            Some(FrequencySummary { entries, omitted_max, ordinals, ordinal_entries })
9560        }
9561        _ => return Err(invalid("frequency summary tag differs")),
9562    })
9563}
9564
9565/// Reads the fixed envelope of a frequency synopsis in a span-based directory.
9566fn summary_span(cur: &mut Cursor<'_>) -> Result<Option<(usize, usize)>> {
9567    let length = cur.u32()? as usize;
9568    let entries = cur.u32()? as usize;
9569    if entries > FREQUENCY_ENTRIES {
9570        return Err(invalid("frequency entry count exceeds its bound"));
9571    }
9572    if length == 0 {
9573        if entries != 0 {
9574            return Err(invalid("missing frequency synopsis has entries"));
9575        }
9576        return Ok(None);
9577    }
9578    if length > MAX_DIRECTORY || length > cur.len().saturating_sub(cur.at) {
9579        return Err(invalid("frequency synopsis span is outside the directory"));
9580    }
9581    Ok(Some((length, entries)))
9582}
9583
9584/// Skips a synopsis whose column the caller does not need. The directory checksum was checked
9585/// before this walk, and the fields still need their lengths and tags checked to find the next one.
9586fn skip_summary(cur: &mut Cursor<'_>, values: bool, rows: usize) -> Result<()> {
9587    match cur.u8()? {
9588        0 => Ok(()),
9589        1 => {
9590            cur.skip(8)?;
9591            let entries = cur.u32()? as usize;
9592            if entries > FREQUENCY_ENTRIES {
9593                return Err(invalid("frequency entry count exceeds its bound"));
9594            }
9595            for _ in 0..entries {
9596                match cur.u8()? {
9597                    0 => {}
9598                    1 => cur.skip(16)?,
9599                    2 => cur.skip(4)?,
9600                    _ => return Err(invalid("frequency value tag differs")),
9601                }
9602                cur.skip(8)?;
9603            }
9604            let ordinals = cur.u32()? as usize;
9605            if ordinals > FREQUENCY_ORDINALS || ordinals > rows {
9606                return Err(invalid("frequency ordinal count exceeds its bound"));
9607            }
9608            for _ in 0..ordinals {
9609                cur.var_u64()?;
9610            }
9611            if values {
9612                cur.skip(ordinals * 2)?;
9613            }
9614            Ok(())
9615        }
9616        _ => Err(invalid("frequency summary tag differs")),
9617    }
9618}
9619
9620/// Reads only the catalog, stripe null counts, and one frequency synopsis. This is the cold path
9621/// for a summary-backed count; constructing every page descriptor and zone map would make it cost
9622/// the size of the table directory even when no row is read.
9623fn quick_nonzero(
9624    mut cur: Cursor<'_>,
9625    name: &str,
9626    fields: &[Field],
9627    rows: usize,
9628    wanted: usize,
9629) -> Result<Option<u64>> {
9630    if cur.take(8)? != DIRECTORY || cur.text()? != name {
9631        return Err(invalid("table directory differs from the catalog"));
9632    }
9633    let width = cur.u16()? as usize;
9634    if width != fields.len() {
9635        return Err(invalid("table directory width differs from the catalog"));
9636    }
9637    for field in fields {
9638        let stored =
9639            Field { name: cur.text()?, ty: read_type(&mut cur)?, not_null: cur.u8()? != 0 };
9640        if &stored != field {
9641            return Err(invalid("table directory schema differs from the catalog"));
9642        }
9643    }
9644    let mut dictionaries = Vec::with_capacity(width);
9645    for field in fields {
9646        let held = match cur.u8()? {
9647            0 => false,
9648            tag if coded_type(&field.ty) && tag == dictionary_tag(&field.ty) => {
9649                cur.skip(20)?;
9650                true
9651            }
9652            _ => return Err(invalid("dictionary page tag differs")),
9653        };
9654        dictionaries.push(held);
9655    }
9656    for _ in 0..width {
9657        match cur.u8()? {
9658            0 => {}
9659            1 => cur.skip(8)?,
9660            _ => return Err(invalid("distinct count tag differs")),
9661        }
9662    }
9663    if cur.u64()? != rows as u64 {
9664        return Err(invalid("table row count differs from the catalog"));
9665    }
9666    let stripes = cur.u32()? as usize;
9667    let mut total = 0_usize;
9668    let mut nulls = 0_u64;
9669    for _ in 0..stripes {
9670        let parts = cur.u32()? as usize;
9671        if parts == 0 || parts > STRIPE_PARTS {
9672            return Err(invalid("stripe part count is outside its bound"));
9673        }
9674        let mut stripe_rows = 0_usize;
9675        for _ in 0..parts {
9676            stripe_rows = stripe_rows
9677                .checked_add(cur.u32()? as usize)
9678                .ok_or_else(|| invalid("stripe row count overflow"))?;
9679        }
9680        total =
9681            total.checked_add(stripe_rows).ok_or_else(|| invalid("stripe row count overflow"))?;
9682        cur.skip(12 + width * 12)?;
9683        for (field, held) in fields.iter().zip(&dictionaries) {
9684            if coded_type(&field.ty) && *held {
9685                cur.skip(20)?;
9686            }
9687        }
9688        for _ in 0..width * 2 {
9689            match cur.u8()? {
9690                0 => {}
9691                1 => cur.skip(20)?,
9692                _ => return Err(invalid("stripe page tag differs")),
9693            }
9694        }
9695        for column in 0..width {
9696            cur.skip_bound()?;
9697            cur.skip_bound()?;
9698            let count = cur.u32()? as u64;
9699            if count > stripe_rows as u64 {
9700                return Err(invalid("null count exceeds stripe rows"));
9701            }
9702            if column == wanted {
9703                nulls = nulls.checked_add(count).ok_or_else(|| invalid("null count overflow"))?;
9704            }
9705            cur.skip(1)?;
9706            match cur.u8()? {
9707                0 => {}
9708                1 => cur.skip(16)?,
9709                _ => return Err(invalid("a stripe sum has an unknown tag")),
9710            }
9711        }
9712    }
9713    if total != rows {
9714        return Err(invalid("table row count differs from stripes"));
9715    }
9716    if cur.done() {
9717        return Ok(None);
9718    }
9719    let magic = cur.take(8)?;
9720    let spanned = magic == FREQUENCIES_SPANS;
9721    let values = magic == FREQUENCIES || spanned;
9722    if !values && magic != FREQUENCIES_V2 {
9723        return Err(invalid("directory extension magic differs"));
9724    }
9725    if cur.u16()? as usize != width {
9726        return Err(invalid("frequency column count differs"));
9727    }
9728    for _ in 0..wanted {
9729        if spanned {
9730            if let Some((length, _)) = summary_span(&mut cur)? {
9731                cur.skip(length)?;
9732            }
9733        } else {
9734            skip_summary(&mut cur, values, rows)?;
9735        }
9736    }
9737    let summary = if spanned {
9738        let Some((length, entries)) = summary_span(&mut cur)? else {
9739            return Ok(None);
9740        };
9741        let start = cur.at;
9742        let summary = decode_summary(&mut cur, &fields[wanted], rows, values)?
9743            .ok_or_else(|| invalid("a stored synopsis is missing"))?;
9744        if cur.at - start != length || summary.entries.len() != entries {
9745            return Err(invalid("a stored synopsis differs from its directory span"));
9746        }
9747        summary
9748    } else {
9749        let Some(summary) = decode_summary(&mut cur, &fields[wanted], rows, values)? else {
9750            return Ok(None);
9751        };
9752        summary
9753    };
9754    let zero = summary
9755        .entries
9756        .iter()
9757        .find(|entry| entry.value == FrequencyValue::Integer(0))
9758        .map(|entry| entry.count)
9759        .or_else(|| (summary.omitted_max == 0).then_some(0));
9760    Ok(zero.and_then(|zero| (rows as u64).checked_sub(nulls)?.checked_sub(zero)))
9761}
9762
9763/// Walks the row-oriented directory while retaining only one column's index and page spans.
9764/// The catalog supplies the schema and the caller checks the complete directory checksum first.
9765fn quick_integer_fold(
9766    file: &File,
9767    mut cur: Cursor<'_>,
9768    entry: &Entry,
9769    size: u64,
9770    wanted: usize,
9771    emit: &mut impl FnMut(i64, u64) -> Result<()>,
9772) -> Result<()> {
9773    let name = &entry.name;
9774    let fields = &entry.fields;
9775    let rows = entry.rows;
9776    if cur.take(8)? != DIRECTORY || cur.text()? != name.as_str() {
9777        return Err(invalid("table directory differs from the catalog"));
9778    }
9779    let width = cur.u16()? as usize;
9780    if width != fields.len() {
9781        return Err(invalid("table directory width differs from the catalog"));
9782    }
9783    for field in fields {
9784        let stored =
9785            Field { name: cur.text()?, ty: read_type(&mut cur)?, not_null: cur.u8()? != 0 };
9786        if &stored != field {
9787            return Err(invalid("table directory schema differs from the catalog"));
9788        }
9789    }
9790    let mut dictionaries = Vec::with_capacity(width);
9791    for field in fields {
9792        dictionaries.push(match cur.u8()? {
9793            0 => false,
9794            tag if coded_type(&field.ty) && tag == dictionary_tag(&field.ty) => {
9795                cur.skip(20)?;
9796                true
9797            }
9798            _ => return Err(invalid("dictionary page tag differs")),
9799        });
9800    }
9801    for _ in 0..width {
9802        match cur.u8()? {
9803            0 => {}
9804            1 => cur.skip(8)?,
9805            _ => return Err(invalid("distinct count tag differs")),
9806        }
9807    }
9808    if cur.u64()? != rows as u64 {
9809        return Err(invalid("table row count differs from the catalog"));
9810    }
9811    let stripes = cur.u32()? as usize;
9812    let mut total = 0_usize;
9813    let mut bytes = Vec::new();
9814    for _ in 0..stripes {
9815        let parts = cur.u32()? as usize;
9816        if parts == 0 || parts > STRIPE_PARTS {
9817            return Err(invalid("stripe part count is outside its bound"));
9818        }
9819        let mut part_rows = Vec::with_capacity(parts);
9820        for _ in 0..parts {
9821            let count = cur.u32()? as usize;
9822            if count == 0 {
9823                return Err(invalid("empty part"));
9824            }
9825            total = total.checked_add(count).ok_or_else(|| invalid("stripe row count overflow"))?;
9826            part_rows.push(count);
9827        }
9828        let index = Span { offset: cur.u64()?, length: cur.u32()? };
9829        let section = index_section(parts)?;
9830        let index_length =
9831            section.checked_mul(width).ok_or_else(|| invalid("index page length overflow"))?;
9832        if index.offset < HEADER
9833            || index.offset.checked_add(u64::from(index.length)).is_none_or(|end| end > size)
9834            || index.length as usize != index_length
9835        {
9836            return Err(invalid("index page range is outside the file"));
9837        }
9838        cur.skip(wanted * 12)?;
9839        let page = Span { offset: cur.u64()?, length: cur.u32()? };
9840        if page.offset < HEADER
9841            || page.offset.checked_add(u64::from(page.length)).is_none_or(|end| end > size)
9842            || page.length as usize > MAX_PAGE
9843        {
9844            return Err(invalid("column page range is outside the file"));
9845        }
9846        cur.skip((width - wanted - 1) * 12)?;
9847        for (field, held) in fields.iter().zip(&dictionaries) {
9848            if coded_type(&field.ty) && *held {
9849                cur.skip(20)?;
9850            }
9851        }
9852        for _ in 0..width * 2 {
9853            match cur.u8()? {
9854                0 => {}
9855                1 => cur.skip(20)?,
9856                _ => return Err(invalid("stripe page tag differs")),
9857            }
9858        }
9859        for _ in 0..width {
9860            cur.skip_bound()?;
9861            cur.skip_bound()?;
9862            cur.skip(5)?;
9863            match cur.u8()? {
9864                0 => {}
9865                1 => cur.skip(16)?,
9866                _ => return Err(invalid("a stripe sum has an unknown tag")),
9867            }
9868        }
9869        let spans = read_index_span(file, index, page, parts, wanted)?;
9870        for (span, expected_rows) in spans.into_iter().zip(part_rows) {
9871            bytes.resize(span.length, 0);
9872            let at = page
9873                .offset
9874                .checked_add(span.start as u64)
9875                .ok_or_else(|| invalid("part range overflow"))?;
9876            read_at(file, at, &mut bytes)?;
9877            if checksum(&bytes) != span.hash {
9878                return Err(invalid("integer part checksum differs"));
9879            }
9880            if bytes.first() == Some(&5) && bytes.get(1) == Some(&0) {
9881                let decoded_rows = integer::fold(&bytes[2..], |value, count| {
9882                    check_integer_tally_value(value, &fields[wanted].ty)?;
9883                    emit(value, count)
9884                })?;
9885                if decoded_rows != expected_rows {
9886                    return Err(invalid("encoded integer part holds the wrong number of rows"));
9887                }
9888            } else {
9889                let column = decode(&fields[wanted].ty, expected_rows, &bytes, None)?;
9890                if let Some(packed) = column.packed_parts() {
9891                    let validity = column.validity();
9892                    let all_valid = column.none_null();
9893                    let base = packed.base();
9894                    let mut codes = [0_u64; 64];
9895                    for from in (0..expected_rows).step_by(codes.len()) {
9896                        let count = (expected_rows - from).min(codes.len());
9897                        packed.unpack(from, &mut codes[..count]);
9898                        for (offset, &code) in codes[..count].iter().enumerate() {
9899                            if all_valid || validity.is_valid(from + offset) {
9900                                // Vector::packed checked that this entire range fits the type.
9901                                emit((base + i128::from(code)) as i64, 1)?;
9902                            }
9903                        }
9904                    }
9905                    continue;
9906                }
9907                let column = column.into_flat()?;
9908                let validity = column.validity();
9909                macro_rules! count_decoded {
9910                    ($values:expr) => {
9911                        for (row, &value) in $values.as_slice().iter().enumerate() {
9912                            if validity.is_valid(row) {
9913                                emit(i64::from(value), 1)?;
9914                            }
9915                        }
9916                    };
9917                }
9918                match column.data() {
9919                    Some(Data::Int8(values)) => count_decoded!(values),
9920                    Some(Data::Int16(values)) => count_decoded!(values),
9921                    Some(Data::Int32(values)) => count_decoded!(values),
9922                    Some(Data::Int64(values)) => count_decoded!(values),
9923                    _ => return Err(invalid("decoded integer part has the wrong type")),
9924                }
9925            }
9926        }
9927    }
9928    if total != rows {
9929        return Err(invalid("table row count differs from stripes"));
9930    }
9931    Ok(())
9932}
9933
9934fn check_integer_tally_value(value: i64, ty: &LogicalType) -> Result<()> {
9935    let fits = match ty {
9936        LogicalType::TinyInt => i8::try_from(value).is_ok(),
9937        LogicalType::SmallInt => i16::try_from(value).is_ok(),
9938        LogicalType::Integer => i32::try_from(value).is_ok(),
9939        LogicalType::BigInt => true,
9940        _ => false,
9941    };
9942    if fits { Ok(()) } else { Err(invalid("encoded integer value is outside its column type")) }
9943}
9944
9945fn decode_directory(bytes: &[u8], size: u64) -> Result<Table> {
9946    read_directory(Cursor::new(bytes), size, None)
9947}
9948
9949/// A directory out of `cur`, which is a whole one in memory or one being read out of the file.
9950///
9951/// `stored_at` is where the directory starts in the file when it is being read out of it, and then
9952/// every frequency synopsis is checked and left there, as [`Frequencies::Stored`].
9953fn read_directory(mut cur: Cursor<'_>, size: u64, stored_at: Option<u64>) -> Result<Table> {
9954    if cur.take(8)? != DIRECTORY {
9955        return Err(invalid("directory magic differs"));
9956    }
9957    let name = cur.text()?;
9958    let width = cur.u16()? as usize;
9959    let mut fields = Vec::with_capacity(width);
9960    for _ in 0..width {
9961        let name = cur.text()?;
9962        let ty = read_type(&mut cur)?;
9963        let not_null = match cur.u8()? {
9964            0 => false,
9965            1 => true,
9966            _ => return Err(invalid("nullability flag differs")),
9967        };
9968        fields.push(Field { name, ty, not_null });
9969    }
9970    let mut dictionaries = Vec::with_capacity(width);
9971    for field in &fields {
9972        dictionaries.push(match cur.u8()? {
9973            0 => None,
9974            tag if tag == dictionary_tag(&field.ty) => {
9975                let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
9976                let end = page
9977                    .offset
9978                    .checked_add(u64::from(page.length))
9979                    .ok_or_else(|| invalid("dictionary page offset overflow"))?;
9980                // A global dictionary covers a whole column, not one bounded stripe. Its lazy
9981                // payload is intentionally allowed to grow past `MAX_PAGE`; only ordinary column
9982                // pages are capped there. `Writer::finish` has already bounded this length by the
9983                // on-disk `u32`, and the range check below keeps it inside the file.
9984                if page.offset < HEADER || end > size {
9985                    return Err(invalid("dictionary page range is outside the file"));
9986                }
9987                Some(page)
9988            }
9989            _ => return Err(invalid("dictionary page tag differs")),
9990        });
9991    }
9992    let mut distincts = Vec::with_capacity(width);
9993    for _ in 0..width {
9994        distincts.push(match cur.u8()? {
9995            0 => None,
9996            1 => Some(cur.u64()?),
9997            _ => return Err(invalid("distinct count tag differs")),
9998        });
9999    }
10000    let rows = usize::try_from(cur.u64()?).map_err(|_| invalid("row count does not fit"))?;
10001    let count = cur.u32()? as usize;
10002    let mut stripes = Vec::with_capacity(count);
10003    let mut total = 0_usize;
10004    for _ in 0..count {
10005        let count = cur.u32()? as usize;
10006        if count == 0 || count > STRIPE_PARTS {
10007            return Err(invalid("stripe part count is outside its bound"));
10008        }
10009        let mut parts = Vec::with_capacity(count);
10010        let mut stripe_rows = 0_usize;
10011        for _ in 0..count {
10012            let rows = cur.u32()?;
10013            if rows == 0 {
10014                return Err(invalid("empty part"));
10015            }
10016            parts.push(rows);
10017            stripe_rows = stripe_rows
10018                .checked_add(rows as usize)
10019                .ok_or_else(|| invalid("stripe row count overflow"))?;
10020        }
10021        total =
10022            total.checked_add(stripe_rows).ok_or_else(|| invalid("stripe row count overflow"))?;
10023        let index = Span { offset: cur.u64()?, length: cur.u32()? };
10024        let section = index_section(count)?;
10025        let wanted = section
10026            .checked_mul(width)
10027            .and_then(|bytes| u32::try_from(bytes).ok())
10028            .ok_or_else(|| invalid("index page length overflow"))?;
10029        let end = index
10030            .offset
10031            .checked_add(u64::from(index.length))
10032            .ok_or_else(|| invalid("index page offset overflow"))?;
10033        if index.offset < HEADER || end > size || index.length != wanted {
10034            return Err(invalid("index page range is outside the file"));
10035        }
10036        let mut pages = Vec::with_capacity(width);
10037        for _ in 0..width {
10038            let offset = cur.u64()?;
10039            let length = cur.u32()?;
10040            let end = offset
10041                .checked_add(u64::from(length))
10042                .ok_or_else(|| invalid("page offset overflow"))?;
10043            if offset < HEADER || end > size || length as usize > MAX_PAGE {
10044                return Err(invalid("page range is outside the file"));
10045            }
10046            pages.push(Span { offset, length });
10047        }
10048        let mut memberships = vec![None; width];
10049        for (column, field) in fields.iter().enumerate() {
10050            if !coded_type(&field.ty) || dictionaries[column].is_none() {
10051                continue;
10052            }
10053            let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
10054            let end = page
10055                .offset
10056                .checked_add(u64::from(page.length))
10057                .ok_or_else(|| invalid("membership page offset overflow"))?;
10058            if page.offset < HEADER || end > size || page.length as usize > MAX_PAGE {
10059                return Err(invalid("membership page range is outside the file"));
10060            }
10061            // No bytes is a stripe written after the column's dictionary was demoted, see
10062            // [`DEMOTED`], which is checked once the block that says so has been read.
10063            if page.length != 0 {
10064                memberships[column] = Some(page);
10065            }
10066        }
10067        let mut sieves = vec![None; width];
10068        for sieve in sieves.iter_mut().take(width) {
10069            match cur.u8()? {
10070                0 => continue,
10071                1 => {}
10072                _ => return Err(invalid("a sieve page has an unknown tag")),
10073            }
10074            let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
10075            let end = page
10076                .offset
10077                .checked_add(u64::from(page.length))
10078                .ok_or_else(|| invalid("sieve page offset overflow"))?;
10079            if page.offset < HEADER || end > size || page.length as usize > MAX_PAGE {
10080                return Err(invalid("sieve page range is outside the file"));
10081            }
10082            *sieve = Some(page);
10083        }
10084        let mut part_ranges = vec![None; width];
10085        for held in part_ranges.iter_mut().take(width) {
10086            match cur.u8()? {
10087                0 => continue,
10088                1 => {}
10089                _ => return Err(invalid("a part range page has an unknown tag")),
10090            }
10091            let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
10092            let end = page
10093                .offset
10094                .checked_add(u64::from(page.length))
10095                .ok_or_else(|| invalid("part range page offset overflow"))?;
10096            if page.offset < HEADER || end > size || page.length as usize > MAX_PAGE {
10097                return Err(invalid("part range page range is outside the file"));
10098            }
10099            *held = Some(page);
10100        }
10101        let mut ranges = Vec::with_capacity(width);
10102        for column in 0..width {
10103            let low = cur.bound()?;
10104            let high = cur.bound()?;
10105            let nulls = cur.u32()? as usize;
10106            if nulls > stripe_rows {
10107                return Err(invalid("null count exceeds stripe rows"));
10108            }
10109            let exact = cur.u8()? != 0;
10110            let sum = match cur.u8()? {
10111                0 => None,
10112                1 => Some(i128::from_le_bytes(
10113                    cur.take(16)?.try_into().map_err(|_| invalid("a stripe sum is truncated"))?,
10114                )),
10115                _ => return Err(invalid("a stripe sum has an unknown tag")),
10116            };
10117            // Files written before the ends of a decimal or a timestamp column carried their power
10118            // of ten hold a bare integer here, and that integer is the one the column holds, which
10119            // is what the power is over. So the type puts it back on the way in and an old file
10120            // prunes as well as a new one. A file that already wrote the power keeps it, because
10121            // this leaves anything that is not an integer alone.
10122            let ty = &fields.get(column).ok_or_else(|| invalid("a stripe range has no column"))?.ty;
10123            let low = low.map(|bound| scaled_as(bound, ty));
10124            let high = high.map(|bound| scaled_as(bound, ty));
10125            ranges.push(Range { low, high, nulls, exact, sum });
10126        }
10127        stripes.push(Stripe {
10128            rows: stripe_rows,
10129            parts,
10130            index,
10131            pages,
10132            memberships: Pages::from_slots(memberships)?,
10133            sieves: Pages::from_slots(sieves)?,
10134            part_ranges: Pages::from_slots(part_ranges)?,
10135            zone: Zone::from_ranges(ranges),
10136        });
10137    }
10138    if total != rows {
10139        return Err(invalid("table row count differs from stripes"));
10140    }
10141    // How many entries each column's synopsis lists, which is all a pair summary is checked against,
10142    // kept apart because the synopses themselves may be left in the file.
10143    let mut entry_counts = vec![0; width];
10144    let frequencies = if cur.done() {
10145        vec![None; width]
10146    } else {
10147        let frequency_magic = cur.take(8)?;
10148        let spanned = frequency_magic == FREQUENCIES_SPANS;
10149        let frequency_values = frequency_magic == FREQUENCIES || spanned;
10150        if !frequency_values && frequency_magic != FREQUENCIES_V2 {
10151            return Err(invalid("directory extension magic differs"));
10152        }
10153        if cur.u16()? as usize != width {
10154            return Err(invalid("frequency column count differs"));
10155        }
10156        let mut frequencies = Vec::with_capacity(width);
10157        for (field, entry_count) in fields.iter().zip(&mut entry_counts) {
10158            if spanned {
10159                let Some((length, entries)) = summary_span(&mut cur)? else {
10160                    frequencies.push(None);
10161                    continue;
10162                };
10163                *entry_count = entries;
10164                let start = cur.at;
10165                if let Some(offset) = stored_at {
10166                    cur.skip(length)?;
10167                    frequencies.push(Some(Frequencies::Stored {
10168                        span: Span {
10169                            offset: offset
10170                                .checked_add(start as u64)
10171                                .ok_or_else(|| invalid("frequency synopsis offset overflow"))?,
10172                            length: u32::try_from(length)
10173                                .map_err(|_| invalid("a frequency synopsis is too long"))?,
10174                        },
10175                        values: true,
10176                        entries,
10177                    }));
10178                } else {
10179                    let summary = decode_summary(&mut cur, field, rows, true)?
10180                        .ok_or_else(|| invalid("a stored synopsis is missing"))?;
10181                    if cur.at - start != length || summary.entries.len() != entries {
10182                        return Err(invalid("a stored synopsis differs from its directory span"));
10183                    }
10184                    frequencies.push(Some(Frequencies::Held(summary)));
10185                }
10186                continue;
10187            }
10188            let start = cur.at;
10189            let summary = decode_summary(&mut cur, field, rows, frequency_values)?;
10190            *entry_count = summary.as_ref().map_or(0, |summary| summary.entries.len());
10191            frequencies.push(match (summary, stored_at) {
10192                (None, _) => None,
10193                (Some(summary), None) => Some(Frequencies::Held(summary)),
10194                (Some(summary), Some(offset)) => Some(Frequencies::Stored {
10195                    span: Span {
10196                        offset: offset + start as u64,
10197                        length: u32::try_from(cur.at - start)
10198                            .map_err(|_| invalid("a frequency synopsis is too long"))?,
10199                    },
10200                    values: frequency_values,
10201                    entries: summary.entries.len(),
10202                }),
10203            });
10204        }
10205        frequencies
10206    };
10207    // There are two optional trailing blocks now rather than one, so the reader dispatches on the
10208    // magic it finds rather than on where the bytes ran out. That is what lets the two arrive
10209    // independently: a format 22 directory ends here and has neither, a directory written before
10210    // the section table has only the clustering declaration, and each one still opens without a
10211    // rewrite. It is the G1 exit criterion, which is that a reader that knows about sections opens
10212    // a file that predates them and answers every query, only without the graph path.
10213    //
10214    // A repeated block is refused rather than allowed to win, because two clustering declarations
10215    // in one directory is a torn directory and the only question is which of them is the lie.
10216    let mut clustering = None;
10217    let mut sections = Vec::new();
10218    let mut pair_frequencies = Vec::new();
10219    let mut seen_pair_frequencies = false;
10220    let mut frequency_texts = vec![Vec::new(); width];
10221    let mut seen_frequency_texts = false;
10222    let mut host_groups = None;
10223    let mut demoted = Vec::new();
10224    let mut seen_sections = false;
10225    let mut dictionary_payloads = Vec::new();
10226    let mut seen_payloads = false;
10227    let mut constraints = Constraints::default();
10228    // Zero until a section table says otherwise, which is what a format 22 table gets and what
10229    // makes every section stamp fail to match on one, because real generations start at one.
10230    let mut generation = 0;
10231    while !cur.done() {
10232        let mut tag = [0u8; 8];
10233        tag.copy_from_slice(cur.take(8)?);
10234        if &tag == PAIR_FREQUENCIES {
10235            if seen_pair_frequencies {
10236                return Err(invalid("directory names two pair frequency blocks"));
10237            }
10238            seen_pair_frequencies = true;
10239            let count = cur.u16()? as usize;
10240            if count > MAX_PAIR_FREQUENCIES {
10241                return Err(invalid("pair frequency count exceeds its bound"));
10242            }
10243            pair_frequencies = Vec::with_capacity(count);
10244            for _ in 0..count {
10245                let first = cur.u16()?;
10246                let second = cur.u16()?;
10247                let first_at = first as usize;
10248                let second_at = second as usize;
10249                if frequencies.get(first_at).and_then(Option::as_ref).is_none() {
10250                    return Err(invalid("pair frequency first column has no synopsis"));
10251                }
10252                let first_entries = entry_counts[first_at];
10253                if !matches!(fields.get(second_at), Some(field) if field.ty == LogicalType::Varchar)
10254                    || dictionaries.get(second_at).copied().flatten().is_none()
10255                {
10256                    return Err(invalid("pair frequency second column has no stable dictionary"));
10257                }
10258                if pair_frequencies
10259                    .iter()
10260                    .any(|held: &PairFrequencySummary| held.first == first && held.second == second)
10261                {
10262                    return Err(invalid("directory repeats a pair frequency summary"));
10263                }
10264                let omitted_max = cur.u64()?;
10265                if omitted_max > rows as u64 {
10266                    return Err(invalid("pair frequency omitted count exceeds the table"));
10267                }
10268                let entries_count = cur.u16()? as usize;
10269                if entries_count > FREQUENCY_ENTRIES {
10270                    return Err(invalid("pair frequency entry count exceeds its bound"));
10271                }
10272                let mut entries = Vec::with_capacity(entries_count);
10273                for _ in 0..entries_count {
10274                    let first_entry = cur.u16()?;
10275                    if first_entry as usize >= first_entries {
10276                        return Err(invalid("pair frequency anchor is outside its synopsis"));
10277                    }
10278                    let second = match cur.u8()? {
10279                        0 => None,
10280                        1 => Some(cur.u32()?),
10281                        _ => return Err(invalid("pair frequency string tag differs")),
10282                    };
10283                    let count = cur.u64()?;
10284                    if count == 0 || count > rows as u64 {
10285                        return Err(invalid("pair frequency count is outside the table"));
10286                    }
10287                    entries.push(PairFrequencyEntry { first_entry, second, count });
10288                }
10289                if entries.windows(2).any(|pair| pair[0].count < pair[1].count) {
10290                    return Err(invalid("pair frequency entries are not descending"));
10291                }
10292                pair_frequencies.push(PairFrequencySummary { first, second, entries, omitted_max });
10293            }
10294        } else if &tag == FREQUENCY_TEXTS {
10295            if seen_frequency_texts {
10296                return Err(invalid("directory names two frequency text blocks"));
10297            }
10298            seen_frequency_texts = true;
10299            let columns = cur.u16()? as usize;
10300            if columns > width {
10301                return Err(invalid("frequency text column count exceeds the schema"));
10302            }
10303            for _ in 0..columns {
10304                let column = cur.u16()? as usize;
10305                if !frequency_texts.get(column).is_some_and(Vec::is_empty) {
10306                    return Err(invalid("frequency text column is repeated or out of range"));
10307                }
10308                if !matches!(fields.get(column), Some(field) if coded_type(&field.ty))
10309                    || dictionaries.get(column).copied().flatten().is_none()
10310                    || frequencies.get(column).and_then(Option::as_ref).is_none()
10311                {
10312                    return Err(invalid("frequency texts belong to a non-string synopsis"));
10313                }
10314                let count = cur.u16()? as usize;
10315                if count == 0 || count != entry_counts[column] {
10316                    return Err(invalid("frequency text count differs from its synopsis"));
10317                }
10318                let mut texts = Vec::with_capacity(count);
10319                for _ in 0..count {
10320                    texts.push(match cur.u8()? {
10321                        0 => None,
10322                        1 => {
10323                            let length = cur.u32()? as usize;
10324                            let bytes = cur.take(length)?.to_vec();
10325                            if fields[column].ty == LogicalType::Varchar {
10326                                std::str::from_utf8(&bytes)
10327                                    .map_err(|_| invalid("frequency text is not UTF-8"))?;
10328                            }
10329                            Some(bytes)
10330                        }
10331                        _ => return Err(invalid("frequency text tag differs")),
10332                    });
10333                }
10334                frequency_texts[column] = texts;
10335            }
10336        } else if &tag == HOST_GROUPS {
10337            if host_groups.is_some() {
10338                return Err(invalid("directory names two host group blocks"));
10339            }
10340            let column = cur.u16()? as usize;
10341            if !matches!(fields.get(column), Some(field) if field.ty == LogicalType::Varchar)
10342                || dictionaries.get(column).copied().flatten().is_none()
10343            {
10344                return Err(invalid("host groups belong to a non-string dictionary"));
10345            }
10346            let omitted_max = cur.u64()?;
10347            if omitted_max > rows as u64 {
10348                return Err(invalid("host group bound exceeds the table"));
10349            }
10350            let count = cur.u16()? as usize;
10351            if count > host::CAPACITY {
10352                return Err(invalid("host group count exceeds its bound"));
10353            }
10354            let mut entries = Vec::with_capacity(count);
10355            let mut bytes = 0_usize;
10356            for _ in 0..count {
10357                let host_len = cur.u32()? as usize;
10358                bytes =
10359                    bytes.checked_add(host_len).ok_or_else(|| invalid("host bytes overflow"))?;
10360                if bytes > host::BYTE_BUDGET {
10361                    return Err(invalid("host groups exceed their byte budget"));
10362                }
10363                let host = std::str::from_utf8(cur.take(host_len)?)
10364                    .map_err(|_| invalid("host is not UTF-8"))?
10365                    .to_owned();
10366                let count = cur.u64()?;
10367                if count == 0 || count > rows as u64 {
10368                    return Err(invalid("host group count exceeds the table"));
10369                }
10370                let bytes_sum = i128::from_le_bytes(
10371                    cur.take(16)?
10372                        .try_into()
10373                        .map_err(|_| invalid("host length sum is truncated"))?,
10374                );
10375                if bytes_sum < 0 {
10376                    return Err(invalid("host length sum is negative"));
10377                }
10378                let minimum_len = cur.u32()? as usize;
10379                bytes =
10380                    bytes.checked_add(minimum_len).ok_or_else(|| invalid("host bytes overflow"))?;
10381                if bytes > host::BYTE_BUDGET {
10382                    return Err(invalid("host groups exceed their byte budget"));
10383                }
10384                let minimum = std::str::from_utf8(cur.take(minimum_len)?)
10385                    .map_err(|_| invalid("host minimum is not UTF-8"))?
10386                    .to_owned();
10387                entries.push(host::HostEntry { host, count, bytes_sum, minimum });
10388            }
10389            if entries.windows(2).any(|pair| pair[0].count < pair[1].count)
10390                || entries.iter().any(|entry| entry.host.is_empty() || entry.minimum.is_empty())
10391            {
10392                return Err(invalid("host groups are not in certified order"));
10393            }
10394            host_groups = Some(host::HostSummary { column, omitted_max, entries });
10395        } else if &tag == CLUSTERING {
10396            if clustering.is_some() {
10397                return Err(invalid("directory names two clustering declarations"));
10398            }
10399            let bucket = Width::from_tag(cur.u8()?)
10400                .ok_or_else(|| invalid("clustering width tag differs"))?;
10401            let count = cur.u16()? as usize;
10402            let mut columns = Vec::with_capacity(count.min(fields.len()));
10403            for _ in 0..count {
10404                columns.push(u32::from(cur.u16()?));
10405            }
10406            // Through the constructor and not built by hand, so that a file claiming a column the
10407            // table does not have is caught at open rather than at the first scan that trusted it.
10408            clustering = Some(Clustering::new(columns, bucket, &fields).map_err(|_| {
10409                invalid("stored clustering declaration does not match the table it is on")
10410            })?);
10411        } else if &tag == DEMOTED {
10412            if !demoted.is_empty() {
10413                return Err(invalid("directory names two demoted column blocks"));
10414            }
10415            let count = cur.u16()? as usize;
10416            if count == 0 || count > width {
10417                return Err(invalid("demoted column count is outside the schema"));
10418            }
10419            demoted = vec![false; width];
10420            for _ in 0..count {
10421                let column = cur.u16()? as usize;
10422                if dictionaries.get(column).copied().flatten().is_none() || demoted[column] {
10423                    return Err(invalid("a demoted column is repeated or has no dictionary"));
10424                }
10425                demoted[column] = true;
10426            }
10427        } else if &tag == SECTIONS {
10428            if seen_sections {
10429                return Err(invalid("directory names two section tables"));
10430            }
10431            seen_sections = true;
10432            generation = cur.u64()?;
10433            let count = cur.u16()? as usize;
10434            if count > MAX_SECTIONS {
10435                return Err(invalid("section count exceeds its bound"));
10436            }
10437            sections = Vec::with_capacity(count);
10438            // entry at a time: a malformed section entry is refused rather than turned into an
10439            // offset.
10440            for _ in 0..count {
10441                sections.push(Section::decode(cur.take(section::ENTRY_BYTES)?)?);
10442            }
10443            for held in &sections {
10444                let Some(end) = held.extent_page.checked_add(u64::from(held.extent_bytes)) else {
10445                    return Err(invalid("a section's extent table overflows the file"));
10446                };
10447                // The bound check is here and not in `section`, because only the caller knows how
10448                // big the file is. A section pointing past the end is a torn directory, and reading
10449                // the payload it names would be reading whatever else is at that offset.
10450                if held.extent_bytes != 0 && (held.extent_page < HEADER || end > size) {
10451                    return Err(invalid("a section's extent table is outside the file"));
10452                }
10453                if held.extents == 0 && held.extent_bytes != 0 {
10454                    return Err(invalid("a section with no extents names an extent table"));
10455                }
10456            }
10457        } else if &tag == DICTIONARY_PAYLOADS {
10458            if seen_payloads {
10459                return Err(invalid("directory names two dictionary payload blocks"));
10460            }
10461            seen_payloads = true;
10462            let count = cur.u16()? as usize;
10463            if count != fields.len() {
10464                return Err(invalid("dictionary payload block does not match the table's columns"));
10465            }
10466            dictionary_payloads = Vec::with_capacity(count);
10467            for _ in 0..count {
10468                let bytes = cur.u64()?;
10469                if bytes > size {
10470                    return Err(invalid("a dictionary payload is larger than the file"));
10471                }
10472                dictionary_payloads.push(bytes);
10473            }
10474        } else if &tag == KEYS {
10475            if !constraints.is_empty() {
10476                return Err(invalid("directory names two key blocks"));
10477            }
10478            let fits = |columns: &[u16]| {
10479                !columns.is_empty() && columns.iter().all(|&column| usize::from(column) < width)
10480            };
10481            let count = cur.u16()? as usize;
10482            for _ in 0..count {
10483                let primary = cur.u8()? != 0;
10484                let columns = columns_of(&mut cur)?;
10485                if !fits(&columns) {
10486                    return Err(invalid("a stored key names a column the table does not have"));
10487                }
10488                constraints.keys.push((columns, primary));
10489            }
10490            let count = cur.u16()? as usize;
10491            for _ in 0..count {
10492                let columns = columns_of(&mut cur)?;
10493                let referenced = columns_of(&mut cur)?;
10494                let len = cur.u32()? as usize;
10495                let table = std::str::from_utf8(cur.take(len)?)
10496                    .map_err(|_| invalid("a foreign key's table name is not UTF-8"))?
10497                    .to_owned();
10498                if !fits(&columns) || referenced.len() != columns.len() || table.is_empty() {
10499                    return Err(invalid("a stored foreign key does not match its table"));
10500                }
10501                constraints.foreign.push(StoredForeign { columns, table, referenced });
10502            }
10503            if constraints.is_empty() {
10504                return Err(invalid("a key block holds no key"));
10505            }
10506        } else {
10507            return Err(invalid("directory extension magic differs"));
10508        }
10509    }
10510    if !cur.done() {
10511        return Err(invalid("directory has trailing bytes"));
10512    }
10513    for stripe in &stripes {
10514        for (column, field) in fields.iter().enumerate() {
10515            if coded_type(&field.ty)
10516                && dictionaries[column].is_some()
10517                && stripe.memberships.get(column).is_none()
10518                && !demoted.get(column).copied().unwrap_or(false)
10519            {
10520                return Err(invalid("string page has no code membership index"));
10521            }
10522        }
10523    }
10524    Ok(Table {
10525        name,
10526        fields,
10527        stripes,
10528        rows,
10529        dictionaries,
10530        dictionary_payloads,
10531        demoted,
10532        distincts,
10533        frequencies,
10534        pair_frequencies,
10535        frequency_texts,
10536        host_groups,
10537        clustering,
10538        generation,
10539        sections,
10540        constraints,
10541    })
10542}
10543
10544/// How many keys or columns follow, in the key block.
10545fn put_count(out: &mut Vec<u8>, count: usize) -> Result<()> {
10546    put_u16(out, u16::try_from(count).map_err(|_| invalid("too many constraints"))?);
10547    Ok(())
10548}
10549
10550/// A count and then that many column places, the layout the key block uses for every list.
10551fn put_columns(out: &mut Vec<u8>, columns: &[u16]) -> Result<()> {
10552    put_count(out, columns.len())?;
10553    for &column in columns {
10554        put_u16(out, column);
10555    }
10556    Ok(())
10557}
10558
10559/// What [`put_columns`] wrote, for a list of columns.
10560fn columns_of(cur: &mut Cursor<'_>) -> Result<Vec<u16>> {
10561    let count = cur.u16()? as usize;
10562    (0..count).map(|_| cur.u16()).collect()
10563}
10564
10565/// A zone map's end, in the layout `rudb_common::bounds` defines. See [`Cursor::bound`].
10566fn put_bound(out: &mut Vec<u8>, bound: Option<&Bound>) -> Result<()> {
10567    bounds::put(out, bound)
10568}
10569
10570/// Which cascades are worth trying on a run of dictionary codes.
10571///
10572/// The exhaustive chooser encodes every candidate at every level of a cascade three deep and keeps
10573/// the smallest, which on a part of 1024 codes is around a hundred full encodes to decide something
10574/// three candidates were always going to win. It is the right default for a crate that does not
10575/// know what it is looking at. Here we do know. Codes are counted from zero in the order the values
10576/// were first seen, so a part of them is one value, or a narrow band, or a few long runs, and those
10577/// are constant, frame of reference and run length. Nothing else has ever come first on this data.
10578///
10579/// A dictionary of dictionary codes is the one candidate that can never pay, because the codes are
10580/// already the dictionary, and it is also the most expensive one to try. Below the top level the
10581/// streams are an RLE's run values and run lengths, which are integers in their own right with no
10582/// runs left in them, so only the two flat candidates go down there.
10583///
10584/// This is size given up for time on purpose, and the ablation is this chooser against
10585/// [`chooser::EXHAUSTIVE`] on the same file.
10586#[derive(Debug)]
10587struct Codes;
10588
10589impl chooser::Chooser for Codes {
10590    fn name(&self) -> &'static str {
10591        "codes"
10592    }
10593
10594    fn narrow_strings(
10595        &self,
10596        _values: &[&[u8]],
10597        offered: &[string::Kind],
10598        _depth: u8,
10599    ) -> Vec<string::Kind> {
10600        // Never reached, because nothing here encodes strings through the cascade. The trait asks
10601        // for it and the honest answer to a question we have no opinion on is the whole list.
10602        offered.to_vec()
10603    }
10604
10605    fn narrow_integers(
10606        &self,
10607        _values: &[i64],
10608        offered: &[integer::Kind],
10609        depth: u8,
10610    ) -> Vec<integer::Kind> {
10611        // The contract is a non empty subset, and a chunk that offers none of the three is a chunk
10612        // this has no opinion about rather than one that cannot be written.
10613        narrowed_to(Codes::keep(depth), offered)
10614    }
10615
10616    fn considers_integer(&self, kind: integer::Kind, depth: u8) -> bool {
10617        Codes::keep(depth).contains(&kind)
10618    }
10619}
10620
10621impl Codes {
10622    fn keep(depth: u8) -> &'static [integer::Kind] {
10623        if depth == 0 {
10624            &[integer::Kind::Constant, integer::Kind::Packed, integer::Kind::Rle]
10625        } else {
10626            &[integer::Kind::Constant, integer::Kind::Packed]
10627        }
10628    }
10629}
10630
10631/// The kinds of `offered` that are in `keep`, or all of `offered` when none of them are.
10632///
10633/// `Packed` applies to every chunk and both choosers keep it, so the fallback is never taken on a
10634/// chunk the cascade offers. It is there because the contract is a non empty subset and a chooser
10635/// that returned nothing would be a chunk that cannot be written. It is also why saying no to a kind
10636/// in `considers_integer` is safe: a kind that is never offered could only have been kept through
10637/// this fallback, and the fallback is never reached.
10638fn narrowed_to(keep: &[integer::Kind], offered: &[integer::Kind]) -> Vec<integer::Kind> {
10639    let narrowed: Vec<integer::Kind> =
10640        offered.iter().copied().filter(|kind| keep.contains(kind)).collect();
10641    if narrowed.is_empty() { offered.to_vec() } else { narrowed }
10642}
10643
10644/// Which cascades are worth trying on a part of plain integers.
10645///
10646/// Wider than [`Codes`] because the values are not codes and carry whatever shape the column has.
10647/// A timestamp column climbs, so delta is the one that matters and is the reason this exists at
10648/// all: three timestamp columns in ClickBench were coming out at exactly eight bytes a row with
10649/// nothing asked of them. The same three columns are why the stride is here, since a timestamp
10650/// loaded from a source that recorded whole seconds is microseconds with twenty zero bits under
10651/// every value. A column that is one value with a handful of exceptions is sparse. What is still
10652/// left out is the dictionary, for the same reason as in [`Codes`]: it is the most
10653/// expensive candidate to try and this file already puts the columns that want one through a
10654/// dictionary of their own before they ever reach here.
10655#[derive(Debug)]
10656struct Fixed;
10657
10658impl chooser::Chooser for Fixed {
10659    fn name(&self) -> &'static str {
10660        "fixed"
10661    }
10662
10663    fn narrow_strings(
10664        &self,
10665        _values: &[&[u8]],
10666        offered: &[string::Kind],
10667        _depth: u8,
10668    ) -> Vec<string::Kind> {
10669        offered.to_vec()
10670    }
10671
10672    fn narrow_integers(
10673        &self,
10674        _values: &[i64],
10675        offered: &[integer::Kind],
10676        depth: u8,
10677    ) -> Vec<integer::Kind> {
10678        narrowed_to(Fixed::keep(depth), offered)
10679    }
10680
10681    fn considers_integer(&self, kind: integer::Kind, depth: u8) -> bool {
10682        Fixed::keep(depth).contains(&kind)
10683    }
10684}
10685
10686impl Fixed {
10687    fn keep(depth: u8) -> &'static [integer::Kind] {
10688        if depth == 0 {
10689            &[
10690                integer::Kind::Constant,
10691                integer::Kind::Packed,
10692                integer::Kind::Delta,
10693                integer::Kind::Rle,
10694                integer::Kind::Sparse,
10695                integer::Kind::Strided,
10696            ]
10697        } else {
10698            &[integer::Kind::Constant, integer::Kind::Packed, integer::Kind::Delta]
10699        }
10700    }
10701}
10702
10703/// Every value of an integer part as an `i64`, or `None` for a part this cannot widen without
10704/// losing one.
10705///
10706/// `UBIGINT` is the only integer type left out, because half its range does not fit and a page that
10707/// silently wrapped would be worse than a page that stays plain. Booleans and strings are not
10708/// integers and have their own ways of being small.
10709fn widened(data: &Data) -> Option<Vec<i64>> {
10710    match data {
10711        Data::Int8(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
10712        Data::UInt8(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
10713        Data::Int16(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
10714        Data::UInt16(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
10715        Data::Int32(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
10716        Data::UInt32(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
10717        Data::Int64(values) => Some(values.to_vec()),
10718        _ => None,
10719    }
10720}
10721
10722/// A cascaded page decoded straight into the width the column is declared at.
10723///
10724/// A value that does not fit is a page that disagrees with the directory about what the column is,
10725/// which is a damaged file rather than a caller error, so it is refused rather than truncated. The
10726/// decoder does that check a block at a time where it can, see [`integer::decode_as`].
10727fn cascade(ty: &LogicalType, bytes: &[u8], rows: usize) -> Result<Data> {
10728    fn wanted<T: integer::Lane>(bytes: &[u8], rows: usize) -> Result<Vec<T>> {
10729        let values = integer::decode_as::<T>(bytes)
10730            .map_err(|error| invalid(&format!("page value is not of its type: {error}")))?;
10731        if values.len() != rows {
10732            return Err(invalid("cascade page holds the wrong number of rows"));
10733        }
10734        Ok(values)
10735    }
10736    Ok(match ty {
10737        LogicalType::TinyInt => Data::Int8(wanted::<i8>(bytes, rows)?.into()),
10738        LogicalType::UTinyInt => Data::UInt8(wanted::<u8>(bytes, rows)?.into()),
10739        LogicalType::SmallInt => Data::Int16(wanted::<i16>(bytes, rows)?.into()),
10740        LogicalType::USmallInt => Data::UInt16(wanted::<u16>(bytes, rows)?.into()),
10741        LogicalType::Integer | LogicalType::Date => Data::Int32(wanted::<i32>(bytes, rows)?.into()),
10742        LogicalType::UInteger => Data::UInt32(wanted::<u32>(bytes, rows)?.into()),
10743        LogicalType::BigInt
10744        | LogicalType::Timestamp
10745        | LogicalType::Time
10746        | LogicalType::TimeTz
10747        | LogicalType::TimestampTz
10748        | LogicalType::TimestampS
10749        | LogicalType::TimestampMs
10750        | LogicalType::TimestampNs => Data::Int64(wanted::<i64>(bytes, rows)?.into()),
10751        // A decimal is an integer of unscaled units, so the cascade reads back into whichever
10752        // integer the declared width says the column is stored as.
10753        LogicalType::Decimal { .. } => match ty.physical() {
10754            PhysicalType::Int16 => Data::Int16(wanted::<i16>(bytes, rows)?.into()),
10755            PhysicalType::Int32 => Data::Int32(wanted::<i32>(bytes, rows)?.into()),
10756            PhysicalType::Int64 => Data::Int64(wanted::<i64>(bytes, rows)?.into()),
10757            _ => return Err(invalid("cascade codec belongs to a decimal that is not an integer")),
10758        },
10759        _ => return Err(invalid("cascade codec belongs to a page that is not integers")),
10760    })
10761}
10762
10763/// How many bytes a part of this type costs written out plainly, which is what the cascade has to
10764/// beat before it is worth the decode.
10765fn plain_width(ty: &LogicalType) -> Option<usize> {
10766    Some(match ty {
10767        LogicalType::TinyInt | LogicalType::UTinyInt => 1,
10768        LogicalType::SmallInt | LogicalType::USmallInt => 2,
10769        LogicalType::Integer | LogicalType::UInteger | LogicalType::Date => 4,
10770        LogicalType::BigInt
10771        | LogicalType::Timestamp
10772        | LogicalType::Time
10773        | LogicalType::TimeTz
10774        | LogicalType::TimestampTz
10775        | LogicalType::TimestampS
10776        | LogicalType::TimestampMs
10777        | LogicalType::TimestampNs => 8,
10778        LogicalType::Decimal { .. } => match ty.physical() {
10779            PhysicalType::Int16 => 2,
10780            PhysicalType::Int32 => 4,
10781            PhysicalType::Int64 => 8,
10782            // The widest decimals are stored as `i128`, which the cascade does not widen into, so
10783            // they take the plain path and there is nothing here to compare against.
10784            _ => return None,
10785        },
10786        _ => return None,
10787    })
10788}
10789
10790/// A part's plain integers through the cascade, or `None` when nothing it offers is worth it.
10791///
10792/// What it has to beat is whatever the page would otherwise have cost, which is the bit packed form
10793/// where there is one and the plain width where there is not. Both are cheaper to decode than a
10794/// cascade, so a tie goes to them.
10795fn cascaded(
10796    flat: &Vector,
10797    ty: &LogicalType,
10798    packed: Option<&Packed<'_>>,
10799    settling: &mut Settling,
10800) -> Result<Option<Vec<u8>>> {
10801    let (Some(width), Some(data)) = (plain_width(ty), flat.data()) else { return Ok(None) };
10802    let Some(values) = widened(data) else { return Ok(None) };
10803    let plain = values.len().saturating_mul(width);
10804    let best = match packed {
10805        // The tag, the base, the word count and the words, which is what the codec 2 branch writes.
10806        Some(packed) => plain.min(21 + size_of_val(packed.words())),
10807        None => plain,
10808    };
10809    let out = settling.encode(&values)?;
10810    Ok((out.len() < best).then_some(out))
10811}
10812
10813/// How often the parts of one column in one stripe search the cascade again, in parts.
10814///
10815/// A stripe is 64 parts, so this is four searches a stripe where there were 64. The search is
10816/// what the cascade costs: on ClickBench `hits` the integer cascade was about a tenth of the load's
10817/// CPU and nearly all of it under `encode_pages`, trying six trees on every part to keep the one
10818/// the part before had kept.
10819const SEARCH_EVERY: usize = 16;
10820
10821/// What the parts of one column in one stripe have settled on in the integer cascade, and the
10822/// symbol table its text pages compress against.
10823///
10824/// One of these per column per stripe, used in part order, so what a part comes out as depends on
10825/// the stripe and not on which thread wrote it or on how many there were.
10826#[derive(Debug, Default)]
10827struct Settling {
10828    /// The shape of the last part that was searched, with what its top level offered, its length
10829    /// and its row count, which is the size a replay is held to.
10830    shape: Option<Shape>,
10831    /// Parts replayed since that search.
10832    since: usize,
10833    /// The FSST table of the last text page that trained one. See [`Settling::text`].
10834    symbols: Option<Symbols>,
10835}
10836
10837/// A table trained on one text page, with what that page came to and how many pages have used it
10838/// since.
10839#[derive(Debug)]
10840struct Symbols {
10841    shape: chooser::Settled,
10842    /// The trained page compressed and plain, in bytes, which is the ratio a later page is held to.
10843    /// Zero compressed when the table came out empty.
10844    len: usize,
10845    payload: usize,
10846    since: usize,
10847}
10848
10849impl Settling {
10850    /// A text page as one FSST chunk, against the table an earlier page of the stripe trained where
10851    /// there is one.
10852    ///
10853    /// The same rule as [`Self::encode`]: the table is used for [`SEARCH_EVERY`] pages and is kept
10854    /// while a page comes out no more than a quarter bigger a byte than the page it was trained on.
10855    /// Past that the page trains a table of its own and the pages after it use that one. Every page
10856    /// still carries the table it was compressed with, so nothing a reader does changes.
10857    fn text(&mut self, values: &[&[u8]], payload: usize) -> Result<Option<Vec<u8>>> {
10858        if let Some(symbols) = self.symbols.as_mut().filter(|symbols| symbols.since < SEARCH_EVERY)
10859        {
10860            let out = string::encode_fsst(values, &symbols.shape)?;
10861            // An empty table stays empty for the pages after, which are the same kind of text.
10862            let held = match &out {
10863                None => symbols.len == 0,
10864                Some(out) => {
10865                    (out.len() as u128) * (symbols.payload as u128) * 4
10866                        <= (symbols.len as u128) * (payload as u128) * 5
10867                }
10868            };
10869            if held {
10870                symbols.since += 1;
10871                return Ok(out);
10872            }
10873        }
10874        let shape = string::fsst_shape(values);
10875        let out = string::encode_fsst(values, &shape)?;
10876        let len = out.as_ref().map_or(0, Vec::len);
10877        self.symbols = Some(Symbols { shape, len, payload: payload.max(1), since: 0 });
10878        Ok(out)
10879    }
10880
10881    /// A part's integers through the cascade, replaying the settled shape where there is one.
10882    ///
10883    /// The replay is kept when it held and came out no more than a quarter bigger a row than the
10884    /// part the shape was searched on. Past that the column has changed under it and the part is
10885    /// searched. A replay that stopped fitting partway has already searched from where it stopped,
10886    /// so its shape is taken as the new one rather than searched a second time.
10887    fn encode(&mut self, values: &[i64]) -> Result<Vec<u8>> {
10888        if let Some(shape) = self.shape.as_ref().filter(|_| self.since < SEARCH_EVERY) {
10889            let replay = chooser::Replay::new(&shape.kinds, &Fixed).expecting(&shape.offered);
10890            let out = integer::encode_with(values, &replay)?;
10891            if !replay.held() {
10892                self.settle(&out, values.len(), replay.first_offered())?;
10893                return Ok(out);
10894            }
10895            let grown = (out.len() as u128) * (shape.rows as u128) * 4;
10896            if grown <= (shape.len as u128) * (values.len() as u128) * 5 {
10897                self.since += 1;
10898                return Ok(out);
10899            }
10900        }
10901        // A replay of nothing is the search, and says what the top level offered on the way.
10902        let search = chooser::Replay::new(&[], &Fixed);
10903        let out = integer::encode_with(values, &search)?;
10904        self.settle(&out, values.len(), search.first_offered())?;
10905        Ok(out)
10906    }
10907
10908    fn settle(&mut self, out: &[u8], rows: usize, offered: Vec<integer::Kind>) -> Result<()> {
10909        let kinds = integer::shape(out)?;
10910        self.shape = Some(Shape { kinds, offered, len: out.len().max(1), rows: rows.max(1) });
10911        self.since = 0;
10912        Ok(())
10913    }
10914}
10915
10916/// A searched part's cascade, what its top level was offered, and what it came to.
10917#[derive(Debug)]
10918struct Shape {
10919    kinds: Vec<integer::Kind>,
10920    offered: Vec<integer::Kind>,
10921    len: usize,
10922    rows: usize,
10923}
10924
10925/// A part's dictionary codes through the integer cascade, or `None` when the cascade did not pay.
10926///
10927/// Until now this stream was a `u32` a row with nothing asked of it, and on ClickBench that was
10928/// 400,185,326 bytes for every one of the 28 varchar columns, the same count for `URL` as for a
10929/// column holding the empty string in nearly every row. Codes are dense integers counted from zero
10930/// and a part holds 1024 of them, which is the shape frame of reference is best at, and a column
10931/// with one value everywhere comes back a constant costing nothing per row rather than four bytes.
10932///
10933/// The result is taken only when it is smaller than the plain form. A cascade is allowed to come
10934/// out larger on a part whose codes are genuinely wide, `URL` has about sixty million distinct
10935/// values, and there is no reason to pay for the decode when it does.
10936/// A varchar page as one FSST layer, or `None` when it did not pay.
10937///
10938/// Until now a varchar page that neither the global dictionary nor the per page dictionary claimed
10939/// was written out raw: four bytes of offset a row and then the bytes. That is the right answer for
10940/// a page of values with nothing in common and the wrong one for a page of English, and a column of
10941/// comments is the case this exists for.
10942///
10943/// One layer and not the full string cascade, which is what the payload blocks of a global
10944/// dictionary go through. The cascade is a search: it encodes the page under every candidate it has
10945/// and recurses into the integer cascade for the lengths of each one, and on TPC-H `orders` that
10946/// took the write from 6.9 s to 48.3 s. It reads back no faster than the dictionary it replaced
10947/// either, 1.807 G instructions against 1.810 G for `select o_comment from orders`, because
10948/// unpicking a nest of layers a value at a time costs what the dictionary's payload block decode
10949/// cost. Raw pages of the same column read in 0.686 G, which says the whole of the difference is
10950/// what the page has to be put back together from.
10951///
10952/// FSST alone keeps most of what the cascade found and gives all of that back. Decoding it is one
10953/// pass over the payload into one buffer, the values are laid end to end in it the way the raw form
10954/// already lays them out, and what the reader hands a chunk is views over that buffer.
10955///
10956/// The page dictionary gets first refusal because it is cheaper still, and it wins on a page whose
10957/// values repeat. What is left for this is the page whose values mostly do not, which is exactly the
10958/// page that was being written raw.
10959///
10960/// Taken only when it comes out smaller than the raw form, so a page of incompressible values pays
10961/// nothing at read time for having been offered.
10962///
10963/// The symbol table is trained once for several pages of the stripe rather than once a page. See
10964/// [`Settling::text`].
10965fn text_compressed(flat: &Vector, settling: &mut Settling) -> Result<Option<Vec<u8>>> {
10966    let mut values: Vec<&[u8]> = Vec::with_capacity(flat.len());
10967    let mut payload = 0_usize;
10968    for row in 0..flat.len() {
10969        // bytes_at: the rows were checked for UTF-8 on the way in, and checking them again here
10970        // was most of what the loop cost.
10971        let text = flat.bytes_at(row).unwrap_or(b"");
10972        payload = payload.saturating_add(text.len());
10973        values.push(text);
10974    }
10975    // What codec 0 writes for a varchar page: an offset a row and one more, then the payload.
10976    let plain = (flat.len() + 1).saturating_mul(4).saturating_add(payload);
10977    let Some(out) = settling.text(&values, payload)? else {
10978        return Ok(None);
10979    };
10980    Ok((out.len() < plain).then_some(out))
10981}
10982
10983fn encoded_codes(codes: &[u32]) -> Result<Option<Vec<u8>>> {
10984    let wide: Vec<i64> = codes.iter().map(|code| i64::from(*code)).collect();
10985    let coded = integer::encode_with(&wide, &Codes)?;
10986    let plain = codes.len().saturating_mul(size_of::<u32>());
10987    Ok((coded.len() < plain).then_some(coded))
10988}
10989
10990/// The validity of a page, which is a flag and then, when some rows are null and some are not, a
10991/// bit a row with the valid ones set.
10992fn push_validity(out: &mut Vec<u8>, flat: &Vector) {
10993    let flag = match flat.validity() {
10994        Validity::AllValid => 0,
10995        Validity::AllInvalid => 1,
10996        Validity::Mask(_) => 2,
10997    };
10998    out.push(flag);
10999    if flag == 2 {
11000        for group in (0..flat.len()).step_by(8) {
11001            let mut bits = 0_u8;
11002            for bit in 0..8 {
11003                if group + bit < flat.len() && !flat.is_null_at(group + bit) {
11004                    bits |= 1 << bit;
11005                }
11006            }
11007            out.push(bits);
11008        }
11009    }
11010}
11011
11012/// One part of a column coded against its global dictionary as a page, from the codes and the
11013/// validity [`push_validity`] wrote for it.
11014///
11015/// The codes go through the integer cascade when that comes out smaller than four bytes a code,
11016/// which on a column that repeats itself it nearly always does, and are written as they are when it
11017/// does not.
11018fn coded_page(codes: &[u32], validity: &[u8]) -> Result<Vec<u8>> {
11019    let coded = encoded_codes(codes)?;
11020    let mut out = Vec::with_capacity(
11021        1 + validity.len() + coded.as_ref().map_or(size_of_val(codes), Vec::len),
11022    );
11023    out.push(if coded.is_some() { 4 } else { 3 });
11024    out.extend_from_slice(validity);
11025    match coded {
11026        Some(coded) => out.extend_from_slice(&coded),
11027        None => {
11028            for &code in codes {
11029                put_u32(&mut out, code);
11030            }
11031        }
11032    }
11033    Ok(out)
11034}
11035
11036/// One part of one column as a page, for every column that is not coded against a global
11037/// dictionary. Those are built by [`coded_page`] from codes [`prepare`] handed out.
11038fn encode(vector: &Vector, settling: &mut Settling) -> Result<Vec<u8>> {
11039    let ty = vector.logical_type();
11040    // flatten: the file writer needs a uniform scalar page and does it once per loaded chunk.
11041    let flat = vector.flatten()?;
11042    let mut out = Vec::new();
11043    let dictionary = if coded_type(ty) { string_dictionary(&flat)? } else { None };
11044    let compressed_text = if dictionary.is_none() && coded_type(ty) {
11045        text_compressed(&flat, settling)?
11046    } else {
11047        None
11048    };
11049    let packed_vector = if dictionary.is_none() { Some(flat.bit_packed()?) } else { None };
11050    let packed = packed_vector.as_ref().and_then(Vector::packed_parts);
11051    // Only where nothing else has claimed the page, which is the plain integer case. A packed part
11052    // is still on the table because the cascade has to beat it too: the bit pack takes a part only
11053    // when it halves it, so a column that shrinks by a third was coming out whole.
11054    let cascade =
11055        if dictionary.is_none() { cascaded(&flat, ty, packed.as_ref(), settling)? } else { None };
11056    out.push(if cascade.is_some() {
11057        5
11058    } else if dictionary.is_some() {
11059        1
11060    } else if compressed_text.is_some() {
11061        6
11062    } else if packed.is_some() {
11063        2
11064    } else {
11065        0
11066    });
11067    push_validity(&mut out, &flat);
11068    if let Some(cascade) = cascade {
11069        out.extend_from_slice(&cascade);
11070        return Ok(out);
11071    }
11072    if let Some(dictionary) = dictionary {
11073        out.extend_from_slice(&dictionary);
11074        return Ok(out);
11075    }
11076    if let Some(compressed_text) = compressed_text {
11077        out.extend_from_slice(&compressed_text);
11078        return Ok(out);
11079    }
11080    if let Some(packed) = packed {
11081        if packed.offset() != 0 {
11082            return Err(invalid("writer received a sliced packed vector"));
11083        }
11084        out.push(u8::try_from(packed.width()).map_err(|_| invalid("packed width overflow"))?);
11085        out.extend_from_slice(&packed.base().to_le_bytes());
11086        put_u32(
11087            &mut out,
11088            u32::try_from(packed.words().len()).map_err(|_| invalid("too many packed words"))?,
11089        );
11090        for word in packed.words() {
11091            put_u64(&mut out, *word);
11092        }
11093        return Ok(out);
11094    }
11095    let data = flat.data().ok_or_else(|| invalid("scalar column did not flatten"))?;
11096    match (ty, data) {
11097        (LogicalType::TinyInt, Data::Int8(values)) => {
11098            for value in &**values {
11099                out.extend_from_slice(&value.to_le_bytes());
11100            }
11101        }
11102        (LogicalType::UTinyInt, Data::UInt8(values)) => {
11103            for value in &**values {
11104                out.extend_from_slice(&value.to_le_bytes());
11105            }
11106        }
11107        (LogicalType::SmallInt, Data::Int16(values)) => {
11108            for value in &**values {
11109                out.extend_from_slice(&value.to_le_bytes());
11110            }
11111        }
11112        (LogicalType::USmallInt, Data::UInt16(values)) => {
11113            for value in &**values {
11114                out.extend_from_slice(&value.to_le_bytes());
11115            }
11116        }
11117        (LogicalType::UInteger, Data::UInt32(values)) => {
11118            for value in &**values {
11119                out.extend_from_slice(&value.to_le_bytes());
11120            }
11121        }
11122        (LogicalType::UBigInt, Data::UInt64(values)) => {
11123            for value in &**values {
11124                out.extend_from_slice(&value.to_le_bytes());
11125            }
11126        }
11127        (LogicalType::Integer | LogicalType::Date, Data::Int32(values)) => {
11128            for value in &**values {
11129                out.extend_from_slice(&value.to_le_bytes());
11130            }
11131        }
11132        (
11133            LogicalType::BigInt
11134            | LogicalType::Timestamp
11135            | LogicalType::Time
11136            | LogicalType::TimeTz
11137            | LogicalType::TimestampTz
11138            | LogicalType::TimestampS
11139            | LogicalType::TimestampMs
11140            | LogicalType::TimestampNs,
11141            Data::Int64(values),
11142        ) => {
11143            for value in &**values {
11144                out.extend_from_slice(&value.to_le_bytes());
11145            }
11146        }
11147        // A hugeint and a uuid are both the 128 bit lane, and a uuid's bits are the ones the rest of
11148        // the engine already carries it in, so nothing about the value changes on the way down.
11149        (LogicalType::HugeInt | LogicalType::Uuid, Data::Int128(values)) => {
11150            for value in &**values {
11151                out.extend_from_slice(&value.to_le_bytes());
11152            }
11153        }
11154        (LogicalType::UHugeInt, Data::UInt128(values)) => {
11155            for value in &**values {
11156                out.extend_from_slice(&value.to_le_bytes());
11157            }
11158        }
11159        // Plainly, in the IEEE bytes. The integer encodings do not apply to a float and none of the
11160        // float codecs is worth having before somebody has measured a corpus of them.
11161        (LogicalType::Float, Data::Float32(values)) => {
11162            for value in &**values {
11163                out.extend_from_slice(&value.to_le_bytes());
11164            }
11165        }
11166        (LogicalType::Double, Data::Float64(values)) => {
11167            for value in &**values {
11168                out.extend_from_slice(&value.to_le_bytes());
11169            }
11170        }
11171        // Three counts and not one number. Months, days and microseconds stay apart on disk because
11172        // they are apart in the value: a month is not a fixed number of days and a day is not a
11173        // fixed number of microseconds, which is the whole reason the type has three fields.
11174        (LogicalType::Interval, Data::Interval(values)) => {
11175            for (months, days, micros) in &**values {
11176                out.extend_from_slice(&months.to_le_bytes());
11177                out.extend_from_slice(&days.to_le_bytes());
11178                out.extend_from_slice(&micros.to_le_bytes());
11179            }
11180        }
11181        (LogicalType::Boolean, Data::Bool(values)) => {
11182            for value in &**values {
11183                out.push(u8::from(*value));
11184            }
11185        }
11186        // The unscaled integer and nothing else. Scale is a property of the column and it is in the
11187        // directory already, so writing it a value at a time would be paying for it twice.
11188        (LogicalType::Decimal { .. }, Data::Int16(values)) => {
11189            for value in &**values {
11190                out.extend_from_slice(&value.to_le_bytes());
11191            }
11192        }
11193        (LogicalType::Decimal { .. }, Data::Int32(values)) => {
11194            for value in &**values {
11195                out.extend_from_slice(&value.to_le_bytes());
11196            }
11197        }
11198        (LogicalType::Decimal { .. }, Data::Int64(values)) => {
11199            for value in &**values {
11200                out.extend_from_slice(&value.to_le_bytes());
11201            }
11202        }
11203        (LogicalType::Decimal { .. }, Data::Int128(values)) => {
11204            for value in &**values {
11205                out.extend_from_slice(&value.to_le_bytes());
11206            }
11207        }
11208        // A blob and a bit string go down the way a varchar does, because the layout is the same
11209        // one: an offset a value and then the bytes. What is not the same is that nothing here may
11210        // read the payload as text, which is why this arm asks the column for bytes rather than for
11211        // a string, and why the codecs above that do read text are all asked of a varchar by name.
11212        (LogicalType::Varchar | LogicalType::Blob | LogicalType::Bit, Data::Varlen(values)) => {
11213            let mut bytes = Vec::new();
11214            put_u32(&mut out, 0);
11215            for row in 0..vector.len() {
11216                let value = values.bytes(row).ok_or_else(|| invalid("string view is invalid"))?;
11217                bytes.extend_from_slice(value);
11218                put_u32(
11219                    &mut out,
11220                    u32::try_from(bytes.len())
11221                        .map_err(|_| invalid("string payload exceeds 4GiB"))?,
11222                );
11223            }
11224            out.extend_from_slice(&bytes);
11225        }
11226        _ => return Err(Error::not_implemented(format!("native page for {ty}"))),
11227    }
11228    Ok(out)
11229}
11230
11231fn put_varint(out: &mut Vec<u8>, mut value: u32) {
11232    while value >= 0x80 {
11233        out.push((value as u8 & 0x7f) | 0x80);
11234        value >>= 7;
11235    }
11236    out.push(value as u8);
11237}
11238
11239/// The distinct codes of one part, which is what a stripe's membership index is merged from.
11240fn unique_codes(codes: &[u32]) -> Vec<u32> {
11241    let mut unique = codes.to_vec();
11242    unique.sort_unstable();
11243    unique.dedup();
11244    unique
11245}
11246
11247/// The union of the sorted distinct codes of every part in a stripe.
11248///
11249/// Pairwise up a tree rather than one long list concatenated and sorted. Both are the same order of
11250/// work on paper and the tree is the one that does not sort what is already in order: sixty four
11251/// sorted lists become one in six passes over the values.
11252fn merged_codes(lists: Vec<Vec<u32>>) -> Vec<u32> {
11253    let mut lists = lists;
11254    while lists.len() > 1 {
11255        let mut next = Vec::with_capacity(lists.len().div_ceil(2));
11256        for pair in lists.chunks(2) {
11257            match pair {
11258                [left, right] => next.push(merged_pair(left, right)),
11259                [only] => next.push(only.clone()),
11260                _ => {}
11261            }
11262        }
11263        lists = next;
11264    }
11265    lists.pop().unwrap_or_default()
11266}
11267
11268fn merged_pair(left: &[u32], right: &[u32]) -> Vec<u32> {
11269    let mut out = Vec::with_capacity(left.len().saturating_add(right.len()));
11270    let mut at = 0;
11271    let mut to = 0;
11272    while at < left.len() && to < right.len() {
11273        match left[at].cmp(&right[to]) {
11274            Ordering::Less => {
11275                out.push(left[at]);
11276                at += 1;
11277            }
11278            Ordering::Greater => {
11279                out.push(right[to]);
11280                to += 1;
11281            }
11282            Ordering::Equal => {
11283                out.push(left[at]);
11284                at += 1;
11285                to += 1;
11286            }
11287        }
11288    }
11289    out.extend_from_slice(&left[at..]);
11290    out.extend_from_slice(&right[to..]);
11291    out
11292}
11293
11294/// The widest bounds and the total null count of a stripe, from the bounds of its parts.
11295///
11296/// A bound that is missing from any part is missing from the stripe, because a missing bound means
11297/// nothing is known and a stripe that holds an unknown cannot claim one.
11298fn merged_range(ranges: impl Iterator<Item = Range>) -> Range {
11299    let mut merged = Range::default();
11300    let mut first = true;
11301    for range in ranges {
11302        merged.nulls = merged.nulls.saturating_add(range.nulls);
11303        // Both of these have to survive every part, so one part that could not say anything makes
11304        // the stripe unable to say it either. A sum is dropped on overflow rather than wrapped,
11305        // which leaves the stripe with exact ends and no total, which is a true thing to say.
11306        merged.sum = match (merged.sum.take(), range.sum) {
11307            (Some(held), Some(next)) if !first => held.checked_add(next),
11308            (_, next) if first => next,
11309            _ => None,
11310        };
11311        merged.exact = if first { range.exact } else { merged.exact && range.exact };
11312        if first {
11313            merged.low = range.low;
11314            merged.high = range.high;
11315            first = false;
11316            continue;
11317        }
11318        merged.low = match (merged.low.take(), range.low) {
11319            (Some(held), Some(next)) => Some(held.smaller(next)),
11320            _ => None,
11321        };
11322        merged.high = match (merged.high.take(), range.high) {
11323            (Some(held), Some(next)) => Some(held.larger(next)),
11324            _ => None,
11325        };
11326    }
11327    merged
11328}
11329
11330/// One stripe's sieves for one column: the part count, a length for each part, then their bytes.
11331///
11332/// One page for the whole stripe rather than one per part, because a part's sieve is a few hundred
11333/// bytes and sixty four of those are sixty four directory entries and sixty four reads for something
11334/// a scan walks straight through. A part with no sieve writes a length of zero and costs four bytes.
11335/// `bound` cut down to [`PART_BOUND_BYTES`], still a bound of the side it was.
11336///
11337/// A prefix of a string sorts at or before the string, so cutting one down leaves a low end that is
11338/// still a low end. A high end has to go the other way, so the cut prefix is stepped up at the last
11339/// byte that can carry it, and a prefix of nothing but `0xFF` has no such byte and gives up the
11340/// bound rather than claiming one that is too small. Anything that is not a string is already a
11341/// fixed width and is left alone.
11342fn shortened(bound: Option<Bound>, high: bool) -> Option<Bound> {
11343    match bound {
11344        Some(Bound::Bytes(mut value)) if value.len() > PART_BOUND_BYTES => {
11345            value.truncate(PART_BOUND_BYTES);
11346            if !high {
11347                return Some(Bound::Bytes(value));
11348            }
11349            while let Some(last) = value.pop() {
11350                if last < u8::MAX {
11351                    value.push(last + 1);
11352                    return Some(Bound::Bytes(value));
11353                }
11354            }
11355            None
11356        }
11357        other => other,
11358    }
11359}
11360
11361/// The ranges of one column's parts of one stripe, as a page.
11362///
11363/// The two ends and the null count, and not `exact` or the total. Those two answer a `MIN` or a
11364/// `SUM` out of the directory, and the directory already answers those per stripe, where the same
11365/// number costs sixty times less to keep. What a part range is for is skipping the part, and
11366/// skipping needs the ends. So a range read back from here says it is not exact, which is true of a
11367/// string end that was cut down anyway.
11368fn encode_part_ranges(ranges: &[Range]) -> Result<Vec<u8>> {
11369    let mut out = Vec::new();
11370    put_u32(
11371        &mut out,
11372        u32::try_from(ranges.len()).map_err(|_| invalid("too many parts in a stripe"))?,
11373    );
11374    for range in ranges {
11375        put_bound(&mut out, shortened(range.low.clone(), false).as_ref())?;
11376        put_bound(&mut out, shortened(range.high.clone(), true).as_ref())?;
11377        put_u32(&mut out, u32::try_from(range.nulls).map_err(|_| invalid("null count overflow"))?);
11378    }
11379    Ok(out)
11380}
11381
11382/// The ranges one encoded page holds, one entry per part of the stripe.
11383fn decode_part_ranges(bytes: &[u8]) -> Result<Vec<Range>> {
11384    let mut cur = Cursor::new(bytes);
11385    let parts = cur.u32()? as usize;
11386    let mut out = Vec::new();
11387    for _ in 0..parts {
11388        let low = cur.bound()?;
11389        let high = cur.bound()?;
11390        let nulls = cur.u32()? as usize;
11391        out.push(Range { low, high, nulls, exact: false, sum: None });
11392    }
11393    Ok(out)
11394}
11395
11396fn encode_sieves<'a>(sieves: impl Iterator<Item = &'a Option<Sieve>>) -> Result<Vec<u8>> {
11397    let held: Vec<&Option<Sieve>> = sieves.collect();
11398    let mut out = Vec::new();
11399    put_u32(
11400        &mut out,
11401        u32::try_from(held.len()).map_err(|_| invalid("too many parts in a stripe"))?,
11402    );
11403    for sieve in &held {
11404        let length = sieve.as_ref().map_or(0, Sieve::len);
11405        put_u32(&mut out, u32::try_from(length).map_err(|_| invalid("sieve length overflow"))?);
11406    }
11407    // flatten: a part with no sieve wrote a length of zero above and contributes no bytes here.
11408    for sieve in held.into_iter().flatten() {
11409        out.extend_from_slice(&sieve.to_bytes());
11410    }
11411    Ok(out)
11412}
11413
11414/// The sieves one encoded page holds, one entry per part of the stripe.
11415///
11416/// A part whose bytes are not a sieve this version understands comes back as `None`, which is a part
11417/// that gets read. That is how a file written by a later version of the sieve stays readable rather
11418/// than being a corrupt page.
11419fn decode_sieves(bytes: &[u8]) -> Result<Vec<Option<Sieve>>> {
11420    let parts = u32::from_le_bytes(
11421        bytes
11422            .get(..4)
11423            .ok_or_else(|| invalid("sieve page is truncated"))?
11424            .try_into()
11425            .map_err(|_| invalid("sieve page is truncated"))?,
11426    ) as usize;
11427    let mut lengths = Vec::with_capacity(parts);
11428    for part in 0..parts {
11429        let at = 4 + part * 4;
11430        let field = bytes.get(at..at + 4).ok_or_else(|| invalid("sieve page is truncated"))?;
11431        lengths.push(u32::from_le_bytes(
11432            field.try_into().map_err(|_| invalid("sieve page is truncated"))?,
11433        ) as usize);
11434    }
11435    let mut at = 4 + parts * 4;
11436    let mut out = Vec::with_capacity(parts);
11437    for length in lengths {
11438        if length == 0 {
11439            out.push(None);
11440            continue;
11441        }
11442        let end = at.checked_add(length).ok_or_else(|| invalid("sieve page is truncated"))?;
11443        let field = bytes.get(at..end).ok_or_else(|| invalid("sieve page is truncated"))?;
11444        out.push(Sieve::from_bytes(field));
11445        at = end;
11446    }
11447    if at != bytes.len() {
11448        return Err(invalid("sieve page has trailing bytes"));
11449    }
11450    Ok(out)
11451}
11452
11453/// One stripe's membership index: the code count and then the codes as ascending deltas.
11454///
11455/// The codes have to be sorted and distinct already, which is what [`unique_codes`] and
11456/// [`merged_codes`] hand over. Anything else decodes as different codes, so neither of those two is
11457/// a step a caller can skip.
11458fn encode_membership(unique: &[u32]) -> Vec<u8> {
11459    let mut out = Vec::with_capacity(unique.len().saturating_mul(2).saturating_add(5));
11460    put_varint(&mut out, u32::try_from(unique.len()).unwrap_or(u32::MAX));
11461    let mut previous = 0;
11462    for (at, &code) in unique.iter().enumerate() {
11463        put_varint(&mut out, if at == 0 { code } else { code - previous });
11464        previous = code;
11465    }
11466    out
11467}
11468
11469fn take_varint(bytes: &[u8], at: &mut usize) -> Result<u32> {
11470    let mut value = 0_u32;
11471    for shift in (0..35).step_by(7) {
11472        let byte = *bytes.get(*at).ok_or_else(|| invalid("membership varint is truncated"))?;
11473        *at += 1;
11474        let part = u32::from(byte & 0x7f);
11475        if shift == 28 && part > 0x0f {
11476            return Err(invalid("membership varint overflow"));
11477        }
11478        value = value
11479            .checked_add(
11480                part.checked_shl(shift).ok_or_else(|| invalid("membership varint overflow"))?,
11481            )
11482            .ok_or_else(|| invalid("membership varint overflow"))?;
11483        if byte & 0x80 == 0 {
11484            return Ok(value);
11485        }
11486    }
11487    Err(invalid("membership varint is too long"))
11488}
11489
11490fn decode_membership(bytes: &[u8]) -> Result<Vec<u32>> {
11491    let mut at = 0;
11492    let count = take_varint(bytes, &mut at)? as usize;
11493    let mut codes = Vec::with_capacity(count);
11494    let mut previous = 0_u32;
11495    for index in 0..count {
11496        let delta = take_varint(bytes, &mut at)?;
11497        let code = if index == 0 {
11498            delta
11499        } else {
11500            previous.checked_add(delta).ok_or_else(|| invalid("membership code overflow"))?
11501        };
11502        if index > 0 && code <= previous {
11503            return Err(invalid("membership codes are not increasing"));
11504        }
11505        codes.push(code);
11506        previous = code;
11507    }
11508    if at != bytes.len() {
11509        return Err(invalid("membership page has trailing bytes"));
11510    }
11511    Ok(codes)
11512}
11513
11514/// A varchar page as a dictionary of its distinct values and a code a row, or `None` when that does
11515/// not come out smaller than the raw form.
11516///
11517/// Every text page that no global dictionary claims asks this first, including the page of
11518/// comments that never has a repeat, so the map is hashed with [`Spread`] rather than SipHash and
11519/// sized for the page up front. With the default hasher and growth it was 4% of the instructions of
11520/// a `lineitem` load from CSV, all of it on `l_comment` pages this then refused.
11521fn string_dictionary(vector: &Vector) -> Result<Option<Vec<u8>>> {
11522    let mut by_text: HashMap<&[u8], u32, Spread> =
11523        HashMap::with_capacity_and_hasher(vector.len(), Spread);
11524    let mut values = Vec::new();
11525    let mut codes = Vec::with_capacity(vector.len());
11526    let mut plain_bytes = 0_usize;
11527    for row in 0..vector.len() {
11528        let text = vector.bytes_at(row).unwrap_or(b"");
11529        plain_bytes = plain_bytes.saturating_add(text.len());
11530        let code = match by_text.get(text) {
11531            Some(&code) => code,
11532            None => {
11533                let code = u32::try_from(values.len())
11534                    .map_err(|_| invalid("too many dictionary values"))?;
11535                by_text.insert(text, code);
11536                values.push(text);
11537                code
11538            }
11539        };
11540        codes.push(code);
11541    }
11542    let dictionary_bytes = values.iter().map(|value| value.len()).sum::<usize>();
11543    let encoded = 8_usize
11544        .saturating_add((values.len() + 1).saturating_mul(4))
11545        .saturating_add(dictionary_bytes)
11546        .saturating_add(codes.len().saturating_mul(4));
11547    let plain = (vector.len() + 1).saturating_mul(4).saturating_add(plain_bytes);
11548    if encoded >= plain {
11549        return Ok(None);
11550    }
11551    let mut out = Vec::with_capacity(encoded);
11552    put_u32(
11553        &mut out,
11554        u32::try_from(values.len()).map_err(|_| invalid("too many dictionary values"))?,
11555    );
11556    put_u32(
11557        &mut out,
11558        u32::try_from(dictionary_bytes).map_err(|_| invalid("dictionary payload exceeds 4GiB"))?,
11559    );
11560    let mut offset = 0_u32;
11561    put_u32(&mut out, offset);
11562    for value in &values {
11563        offset = offset
11564            .checked_add(
11565                u32::try_from(value.len()).map_err(|_| invalid("dictionary value is too long"))?,
11566            )
11567            .ok_or_else(|| invalid("dictionary payload exceeds 4GiB"))?;
11568        put_u32(&mut out, offset);
11569    }
11570    for value in values {
11571        out.extend_from_slice(value);
11572    }
11573    for code in codes {
11574        put_u32(&mut out, code);
11575    }
11576    Ok(Some(out))
11577}
11578
11579/// The room one closing column takes under [`CLOSE_BYTES`], given back when dropped.
11580struct Room<'a, T> {
11581    state: &'a Mutex<(T, usize)>,
11582    finished: &'a Condvar,
11583    bytes: usize,
11584}
11585
11586impl<T> Drop for Room<'_, T> {
11587    fn drop(&mut self) {
11588        let mut held = self.state.lock().unwrap_or_else(PoisonError::into_inner);
11589        held.1 -= self.bytes;
11590        drop(held);
11591        self.finished.notify_all();
11592    }
11593}
11594
11595/// One column's work at the end of a load, as [`Writer::close_columns`] schedules it.
11596enum Closing<'a> {
11597    /// A numeric column's frequencies, whether to count its distinct values exactly, and the
11598    /// range to count them in a flat array when it is short enough.
11599    Numeric {
11600        column: usize,
11601        counted: bool,
11602        dense: Option<(u64, usize)>,
11603    },
11604    Dictionary {
11605        index: usize,
11606        dictionary: &'a GlobalDictionary,
11607    },
11608}
11609
11610/// What one [`Closing`] came back with, by column.
11611enum Closed {
11612    Numeric(usize, (Option<FrequencySummary>, Option<u64>)),
11613    Dictionary(usize, ClosedDictionary),
11614}
11615
11616/// What [`Writer::close_dictionary`] builds for one column and [`Writer::close`] writes.
11617struct ClosedDictionary {
11618    /// `None` for a demoted dictionary, which holds only some of the column. See [`DEMOTED`].
11619    distinct: Option<u64>,
11620    frequencies: Option<FrequencySummary>,
11621    texts: Vec<Option<Vec<u8>>>,
11622    hosts: Option<host::HostSummary>,
11623    encoded: EncodedDictionary,
11624    /// The bytes of the column's payload blocks, which are already in the file.
11625    payload: u64,
11626}
11627
11628struct EncodedDictionary {
11629    index: Vec<u8>,
11630    ranks: Vec<u8>,
11631    grams: Vec<u8>,
11632}
11633
11634/// Sorts codes into the byte order of the values they name, eight bytes of depth at a time.
11635///
11636/// # What the shape of the data does to a comparison sort
11637///
11638/// Distinct values against distinct prefixes, on the eight million row `hits`:
11639///
11640/// ```text
11641///   distinct   first 8   first 16   first 32   column
11642///  2,266,417        50      8,892    232,630   URL
11643///  2,346,025        49      8,534    204,060   Referer
11644///  1,357,764    81,362    348,340    861,579   Title
11645/// ```
11646///
11647/// Two and a quarter million URLs have fifty distinct first eight bytes between them, because they
11648/// all begin `http://` and then a host and there are not many hosts. So a sort that leads with
11649/// those eight bytes settles almost nothing on `URL` and `Referer`, whatever the comment on it used
11650/// to say, and almost every pair falls through to a comparison of whole values that agree for most
11651/// of their length. `Title` is free text and separates at eight bytes, which is why the design
11652/// looked right when it was written.
11653///
11654/// # What is done about it
11655///
11656/// Sort on eight bytes of the value at the current depth, held beside the code, and then take each
11657/// run that those eight bytes leave tied and sort it again on the next eight. A value is fetched
11658/// from the payload once per eight bytes of depth rather than once per comparison, and the sort
11659/// itself runs over an array of integers that is in cache rather than over pointers into a payload
11660/// that is hundreds of megabytes.
11661///
11662/// That is the whole trick, and it matters because the payload touch is the expensive part. The
11663/// bytes themselves are nearly free once the line is in cache, so reading eight at a time and
11664/// throwing away the ones that were not needed beats going back for each one.
11665///
11666/// # Why the length has to be carried
11667///
11668/// The eight bytes are padded with zero when the value has fewer than eight left, and a zero byte
11669/// can appear in a value, so equal keys do not mean equal bytes. What is true is that a value which
11670/// ran out inside the window is a prefix of any other value with the same key, and a prefix sorts
11671/// first, so how many of the eight bytes were real is the tie break and nothing further is needed.
11672/// A run is only worth another pass when all eight were real, because otherwise the run is one
11673/// value: a dictionary holds a value once.
11674fn sort_by_value<'a>(codes: &mut [u32], values: impl Fn(u32) -> &'a [u8]) {
11675    let mut work = vec![(0, codes.len(), 0)];
11676    let mut keyed: Vec<(u64, u8, u32)> = Vec::new();
11677    while let Some((from, to, depth)) = work.pop() {
11678        let part = &mut codes[from..to];
11679        keyed.clear();
11680        keyed.extend(part.iter().map(|&code| {
11681            let value = values(code);
11682            let rest = value.get(depth..).unwrap_or_default();
11683            (head(rest), rest.len().min(8) as u8, code)
11684        }));
11685        keyed.sort_unstable();
11686        for (slot, entry) in part.iter_mut().zip(keyed.iter()) {
11687            *slot = entry.2;
11688        }
11689        let mut start = 0;
11690        while start < keyed.len() {
11691            let (key, taken, _) = keyed[start];
11692            let mut end = start + 1;
11693            while end < keyed.len() && keyed[end].0 == key && keyed[end].1 == taken {
11694                end += 1;
11695            }
11696            if taken == 8 && end - start > 1 {
11697                work.push((from + start, from + end, depth + 8));
11698            }
11699            start = end;
11700        }
11701    }
11702}
11703
11704/// How few codes are worth sorting on more than one thread.
11705const PARALLEL_SORT_MIN: usize = 1 << 16;
11706
11707/// How many buckets a thread gets in [`sort_by_value_across`], so that a thread that drew a slow
11708/// bucket is not what the others wait for.
11709const BUCKETS_PER_WORKER: usize = 4;
11710
11711/// How many sampled codes stand for each bucket when the splitters are picked.
11712const SAMPLES_PER_BUCKET: usize = 32;
11713
11714/// [`sort_by_value`] over `workers` threads, with the same answer.
11715///
11716/// A sample sort. A sample of the codes is sorted and cut into as many equal runs as there are
11717/// buckets, and the values at the cuts are the splitters. Every code goes to the bucket its value
11718/// falls in by a binary search of the splitters, the buckets are laid end to end in splitter order,
11719/// and each bucket is then sorted on its own by whichever thread takes it. Every value in a bucket
11720/// sorts after every value in the bucket before, so the buckets sorted one by one are the codes
11721/// sorted.
11722///
11723/// The answer is the one [`sort_by_value`] gives down to the order of equal values, not only the
11724/// order of different ones. A global dictionary holds each value once, so there are none, but the
11725/// sort does not rely on it: equal values land in the same bucket in code order, which is the order
11726/// [`sort_by_value`] leaves them in, since the code is the last thing it sorts on.
11727///
11728/// On the 10m ClickBench sample the close sorts five columns of one to three and a half million
11729/// distinct values, one column at a time, and until this each sort ran on one thread while the
11730/// other thirty one waited for it.
11731fn sort_by_value_across<'a>(
11732    codes: &mut [u32],
11733    values: impl Fn(u32) -> &'a [u8] + Sync,
11734    workers: usize,
11735) {
11736    if workers <= 1 || codes.len() < PARALLEL_SORT_MIN {
11737        sort_by_value(codes, values);
11738        return;
11739    }
11740    let buckets = workers * BUCKETS_PER_WORKER;
11741    let wanted = buckets * SAMPLES_PER_BUCKET;
11742    let mut sample = (0..wanted).map(|at| codes[at * codes.len() / wanted]).collect::<Vec<_>>();
11743    sort_by_value(&mut sample, &values);
11744    let splitters =
11745        (1..buckets).map(|cut| values(sample[cut * sample.len() / buckets])).collect::<Vec<_>>();
11746    let values = &values;
11747    let splitters = &splitters;
11748    let per = codes.len().div_ceil(workers);
11749    // Which bucket each code goes to, a run of the codes per thread.
11750    let places = std::thread::scope(|scope| {
11751        codes
11752            .chunks(per)
11753            .map(|run| {
11754                scope.spawn(move || {
11755                    run.iter()
11756                        .map(|&code| {
11757                            let value = values(code);
11758                            splitters.partition_point(|splitter| *splitter <= value) as u32
11759                        })
11760                        .collect::<Vec<_>>()
11761                })
11762            })
11763            .collect::<Vec<_>>()
11764            .into_iter()
11765            .flat_map(|handle| {
11766                handle.join().unwrap_or_else(|panic| std::panic::resume_unwind(panic))
11767            })
11768            .collect::<Vec<_>>()
11769    });
11770    let mut starts = vec![0_usize; buckets + 1];
11771    for &place in &places {
11772        starts[place as usize + 1] += 1;
11773    }
11774    for bucket in 0..buckets {
11775        starts[bucket + 1] += starts[bucket];
11776    }
11777    let mut laid = vec![0_u32; codes.len()];
11778    let mut next = starts.clone();
11779    for (&code, &place) in codes.iter().zip(&places) {
11780        laid[next[place as usize]] = code;
11781        next[place as usize] += 1;
11782    }
11783    drop(places);
11784    let mut runs = Vec::with_capacity(buckets);
11785    let mut rest = laid.as_mut_slice();
11786    for bucket in 0..buckets {
11787        let (run, after) = rest.split_at_mut(starts[bucket + 1] - starts[bucket]);
11788        runs.push(run);
11789        rest = after;
11790    }
11791    // The largest buckets first, since they are taken from the back.
11792    runs.sort_by_key(|run| run.len());
11793    let queue = Mutex::new(runs);
11794    std::thread::scope(|scope| {
11795        for _ in 0..workers {
11796            scope.spawn(|| {
11797                loop {
11798                    let taken = queue.lock().unwrap_or_else(PoisonError::into_inner).pop();
11799                    let Some(run) = taken else { break };
11800                    sort_by_value(run, values);
11801                }
11802            });
11803        }
11804    });
11805    codes.copy_from_slice(&laid);
11806}
11807
11808/// The first eight bytes of a value as an integer that sorts the way the bytes sort.
11809///
11810/// A value shorter than eight bytes is padded with zeros after it. The short case is two loads of
11811/// four that overlap rather than a copy of however many bytes there are, because a copy of a length
11812/// the compiler cannot see is a call to `memcpy`, and this runs once a value at every level of the
11813/// sort in [`sort_by_value`]. On a load of a million rows of `hits` that call was 2.9 percent of
11814/// the load's cycles, and the loads that replace it put 1.5 percent on the sort itself.
11815fn head(bytes: &[u8]) -> u64 {
11816    if let Some(word) = bytes.first_chunk::<8>() {
11817        return u64::from_be_bytes(*word);
11818    }
11819    let len = bytes.len();
11820    if len >= 4 {
11821        let front = u64::from(u32::from_be_bytes([bytes[0], bytes[1], bytes[2], bytes[3]]));
11822        let back = &bytes[len - 4..];
11823        let back = u64::from(u32::from_be_bytes([back[0], back[1], back[2], back[3]]));
11824        return (front << 32) | (back << (8 * (8 - len)));
11825    }
11826    bytes.iter().enumerate().fold(0, |word, (at, &byte)| word | (u64::from(byte) << (56 - 8 * at)))
11827}
11828
11829/// One column's dictionary page, which is its index and its sorted order.
11830///
11831/// The payload is not in it. Its blocks are in the file already, written as each was encoded, and
11832/// `places` says where, in block order. With `scattered` set the index records each block's start
11833/// and length, so a reader can find one wherever it went.
11834///
11835/// `scattered` false lays the blocks out the way a file written before format 26 has them, one
11836/// behind the next with only the ends recorded. Nothing in the writer asks for that any more. It is
11837/// kept because [`open_global_dictionary`] still reads those files and a reading path that nothing
11838/// can produce is a reading path nothing tests.
11839fn encode_global_dictionary(
11840    dictionary: &GlobalDictionary,
11841    order: &[(u64, u32)],
11842    places: &[Placed],
11843    scattered: bool,
11844) -> Result<EncodedDictionary> {
11845    let values = dictionary.values();
11846    if order.len() != values {
11847        return Err(invalid("global dictionary order does not cover its values"));
11848    }
11849    let blocks = values.div_ceil(TEXT_PAYLOAD_VALUES);
11850    if places.len() != blocks {
11851        return Err(invalid("global dictionary payload is not the blocks it says it is"));
11852    }
11853    if dictionary.grams.len() != blocks {
11854        return Err(invalid("global dictionary signatures do not cover its blocks"));
11855    }
11856    let (ranks, rank_ends) = encode_ranks(order, code_width(values))?;
11857    let rank_blocks = values.div_ceil(TEXT_RANK_BLOCK);
11858    let offset_bits = offset_width(&dictionary.ends);
11859    let payload_words = if scattered { 3 } else { 2 };
11860    let index_len = DICTIONARY_HEADER
11861        .checked_add(offset_bytes(values, offset_bits))
11862        .and_then(|len| len.checked_add(blocks.checked_mul(payload_words * 8)?))
11863        .and_then(|len| len.checked_add(rank_blocks.checked_mul(16)?))
11864        .and_then(|len| len.checked_add(8))
11865        .ok_or_else(|| invalid("global dictionary index length overflow"))?;
11866    let mut index = Vec::with_capacity(index_len);
11867    put_u32(
11868        &mut index,
11869        u32::try_from(values).map_err(|_| invalid("global dictionary has too many values"))?,
11870    );
11871    put_u32(&mut index, TEXT_PAYLOAD_VALUES as u32);
11872    put_u32(
11873        &mut index,
11874        u32::try_from(blocks).map_err(|_| invalid("global dictionary has too many blocks"))?,
11875    );
11876    let flag = (if scattered { DICTIONARY_SCATTERED } else { 0 })
11877        | DICTIONARY_GRAMS
11878        | DICTIONARY_WIDE_GRAMS;
11879    put_u32(&mut index, offset_bits as u32 | flag);
11880    encode_offsets(&dictionary.ends, offset_bits, &mut index)?;
11881    // Where each block is and how long it is, so a reader can find one. The stored blocks are
11882    // shorter than the decoded ones and by a different amount each, so their lengths are the one
11883    // thing the offsets above no longer say, and where they start is no longer arithmetic on the
11884    // block before once a block is written the moment it is encoded.
11885    let mut end = 0_u64;
11886    for place in places {
11887        if scattered {
11888            put_u64(&mut index, place.start);
11889            put_u64(&mut index, place.length);
11890        } else {
11891            end = end
11892                .checked_add(place.length)
11893                .ok_or_else(|| invalid("global dictionary payload overflow"))?;
11894            put_u64(&mut index, end);
11895        }
11896    }
11897    for place in places {
11898        put_u64(&mut index, place.hash);
11899    }
11900    // The same two lists for the sorted order. A rank block is packed at whatever width its own
11901    // heads need, so where one ends is no longer arithmetic on the block number.
11902    if rank_ends.len() != rank_blocks {
11903        return Err(invalid("global dictionary order is not the blocks it says it is"));
11904    }
11905    for end in &rank_ends {
11906        put_u64(&mut index, *end);
11907    }
11908    let mut at = 0_usize;
11909    for end in &rank_ends {
11910        let end = usize::try_from(*end).map_err(|_| invalid("global dictionary order overflow"))?;
11911        put_u64(&mut index, checksum(&ranks[at..end]));
11912        at = end;
11913    }
11914    let gram_len = blocks
11915        .checked_mul(TEXT_GRAM_BYTES)
11916        .ok_or_else(|| invalid("global dictionary signature count overflow"))?;
11917    let mut grams = Vec::with_capacity(gram_len);
11918    for block in &dictionary.grams {
11919        grams.extend_from_slice(block);
11920    }
11921    put_u64(&mut index, checksum(&grams));
11922    if index.len() != index_len {
11923        return Err(invalid("global dictionary index is not the length it was laid out for"));
11924    }
11925    Ok(EncodedDictionary { index, ranks, grams })
11926}
11927
11928/// How many blocks of the payload the shape is settled on.
11929///
11930/// Eight blocks is 8,192 values, which is the sample `chooser::Sampled` draws and is that size for
11931/// the same reason. They are spread across the dictionary rather than taken off the front, because
11932/// a dictionary is in the order values were first seen and the front of it is the first morsel of
11933/// the load.
11934const PAYLOAD_SAMPLE_BLOCKS: usize = 8;
11935
11936/// The shapes the payload encoder picks between.
11937///
11938/// Narrow on purpose. The exhaustive search encodes every candidate at every level and runs at two
11939/// to six megabytes a second on this data, which over the twelve gigabytes of dictionary `hits`
11940/// carries is about an hour of processor time, so it cannot be what a load does. Each of these
11941/// settles the outer level and the one below it, which is where almost all of that hour goes, and
11942/// leaves the levels under them to the exhaustive search where the chunks are small enough for it
11943/// to cost nothing.
11944///
11945/// Measured on the five ClickBench columns that have a dictionary worth the name, at 1,024 values a
11946/// block, against the exhaustive search over the same blocks:
11947///
11948/// | column | exhaustive | FRONT then LZ | LZ then FSST | LZ then PLAIN |
11949/// |---|---|---|---|---|
11950/// | 2 | 2.923 at 4.3 MB/s | 2.587 at 21.2 | 2.593 at 36.1 | 2.538 at 53.6 |
11951/// | 13 | 3.093 at 3.1 | 3.029 at 36.4 | 2.921 at 35.7 | 2.770 at 82.9 |
11952/// | 14 | 2.330 at 2.1 | 2.283 at 24.3 | 2.213 at 23.5 | 2.113 at 67.6 |
11953/// | 39 | 2.459 at 5.3 | 2.147 at 10.6 | 2.145 at 29.3 | 2.088 at 43.1 |
11954/// | 56 | 4.694 at 6.3 | 4.381 at 51.0 | 4.172 at 50.6 | 3.983 at 86.8 |
11955///
11956/// The best of the three per column is 98 percent of the exhaustive ratio for a tenth of the time.
11957/// `FSST` and `PLAIN` on their own are in the list as a floor rather than to win. `FSST` is the
11958/// right answer for text that does not share prefixes with its neighbours, and `PLAIN` is there so
11959/// that a column nothing compresses is found out in the sample and written at a gigabyte a second
11960/// rather than searched for an answer that does not exist.
11961fn payload_shapes() -> Vec<chooser::Settled> {
11962    let integers = vec![integer::Kind::Packed];
11963    [
11964        vec![string::Kind::Front, string::Kind::Lz],
11965        vec![string::Kind::Lz, string::Kind::Fsst],
11966        vec![string::Kind::Lz, string::Kind::Plain],
11967        vec![string::Kind::Fsst],
11968        vec![string::Kind::Plain],
11969    ]
11970    .into_iter()
11971    .map(|strings| chooser::Settled::new(strings, integers.clone()))
11972    .collect()
11973}
11974
11975/// Syncs the file, and counts the sync and how long it took as a publish wait when a load is being
11976/// profiled.
11977///
11978/// A wait rather than time, because the time is already in the publish span around it. What the
11979/// wait columns add is how much of publish was the device, which on the WSL2 disk of the gaming PC
11980/// is most of it: a sync there costs about two milliseconds (see `rudb_device_card`).
11981fn synced(file: &dyn rudb_io::File, profile: Option<&LoadProfile>) -> Result<()> {
11982    let started = profile.map(|_| std::time::Instant::now());
11983    file.sync()?;
11984    if let (Some(profile), Some(started)) = (profile, started) {
11985        profile.waited(
11986            Stage::Publish,
11987            u64::try_from(started.elapsed().as_nanos()).unwrap_or(u64::MAX),
11988        );
11989    }
11990    Ok(())
11991}
11992
11993/// One sealed dictionary block on its way to being encoded outside the writer's lock.
11994///
11995/// Handed out by the merge that sealed it and encoded with the pages of the same stripe. See
11996/// [`GlobalDictionary::hand_out`].
11997#[derive(Debug)]
11998pub(crate) struct Unencoded {
11999    column: usize,
12000    at: usize,
12001    ends: Vec<u32>,
12002    bytes: Vec<u8>,
12003    shape: chooser::Settled,
12004}
12005
12006impl Unencoded {
12007    /// The encoded block and its signature.
12008    pub(crate) fn encode(&self) -> Result<EncodedBlock> {
12009        let values = block_values(&self.ends, &self.bytes);
12010        Ok((string::encode_with(&values, &self.shape)?, block_grams(&values)))
12011    }
12012
12013    /// The column and the block number the encoded block goes back to.
12014    pub(crate) fn place(&self) -> (usize, usize) {
12015        (self.column, self.at)
12016    }
12017}
12018
12019/// One encoded dictionary block and the signature of the values in it.
12020///
12021/// Boxed because it is carried around in things that are otherwise small.
12022pub(crate) type EncodedBlock = (Vec<u8>, Box<[u8; TEXT_GRAM_BYTES]>);
12023
12024/// The conservative four-byte substring signature of one block's values.
12025fn block_grams(values: &[&[u8]]) -> Box<[u8; TEXT_GRAM_BYTES]> {
12026    let mut grams = Box::new([0_u8; TEXT_GRAM_BYTES]);
12027    for value in values {
12028        for gram in value.windows(4) {
12029            for bit in gram_bits(gram, TEXT_GRAM_BYTES) {
12030                grams[bit / 8] |= 1 << (bit % 8);
12031            }
12032        }
12033    }
12034    grams
12035}
12036
12037/// The values of one block, given where each of them ends relative to the block.
12038fn block_values<'a>(ends: &[u32], bytes: &'a [u8]) -> Vec<&'a [u8]> {
12039    let mut out = Vec::with_capacity(ends.len());
12040    let mut from = 0;
12041    for &to in ends {
12042        out.push(&bytes[from..to as usize]);
12043        from = to as usize;
12044    }
12045    out
12046}
12047
12048/// Encodes every block still raw at the end of a load: the part block each column ends on and,
12049/// for a column too small to have settled a shape, every block it has.
12050///
12051/// Across threads, the way [`encode_ready`] does it. This ran one column at a time on the thread
12052/// closing the table, and a column that never settled a shape encodes each block by trying every
12053/// candidate, so on a million rows of `hits` it was most of the load's CPU on one core.
12054fn finish_dictionaries(dictionaries: &mut [Option<GlobalDictionary>]) -> Result<()> {
12055    for dictionary in dictionaries.iter_mut().flatten() {
12056        if !dictionary.early.is_empty() {
12057            return Err(Error::internal("a dictionary block handed out never came back"));
12058        }
12059        dictionary.seal_rest();
12060        dictionary.settle_rest()?;
12061    }
12062    encode_waiting(dictionaries)?;
12063    // A block handed out and never given back leaves a gap nothing above would notice when it was
12064    // the last one, so the count is checked against the values as well.
12065    if dictionaries
12066        .iter()
12067        .flatten()
12068        .any(|dictionary| dictionary.encoded() != dictionary.values().div_ceil(TEXT_PAYLOAD_VALUES))
12069    {
12070        return Err(Error::internal("a dictionary block handed out never came back"));
12071    }
12072    Ok(())
12073}
12074
12075/// Encodes the waiting blocks of every dictionary across threads, and appends them to their columns
12076/// in order.
12077fn encode_waiting(dictionaries: &mut [Option<GlobalDictionary>]) -> Result<()> {
12078    let jobs = dictionaries
12079        .iter()
12080        .enumerate()
12081        .flat_map(|(column, held)| {
12082            (0..held.as_ref().map_or(0, |held| held.waiting.len())).map(move |at| (column, at))
12083        })
12084        .collect::<Vec<_>>();
12085    if jobs.is_empty() {
12086        return Ok(());
12087    }
12088    let one = |column: usize, at: usize| -> Result<(usize, usize, EncodedBlock)> {
12089        let held = dictionaries[column].as_ref().ok_or_else(|| Error::internal("no dictionary"))?;
12090        Ok((column, at, held.encode_waiting(at)?))
12091    };
12092    let workers = std::thread::available_parallelism()
12093        .map_or(1, usize::from)
12094        .min(MAX_FREQUENCY_WORKERS)
12095        .min(jobs.len());
12096    let made = if workers <= 1 {
12097        jobs.iter().map(|&(column, at)| one(column, at)).collect::<Result<Vec<_>>>()?
12098    } else {
12099        let next = AtomicUsize::new(0);
12100        let jobs = &jobs;
12101        let pieces = std::thread::scope(|scope| {
12102            (0..workers)
12103                .map(|_| {
12104                    scope.spawn(|| {
12105                        let mut mine = Vec::new();
12106                        loop {
12107                            let job = next.fetch_add(1, Atomic::Relaxed);
12108                            let Some(&(column, at)) = jobs.get(job) else { break };
12109                            mine.push(one(column, at)?);
12110                        }
12111                        Ok(mine)
12112                    })
12113                })
12114                .collect::<Vec<_>>()
12115                .into_iter()
12116                .map(|handle| {
12117                    handle
12118                        .join()
12119                        .map_err(|_| Error::internal("a dictionary encode worker panicked"))?
12120                })
12121                .collect::<Result<Vec<_>>>()
12122        })?;
12123        pieces.into_iter().flatten().collect()
12124    };
12125    let mut done: Vec<Vec<(usize, EncodedBlock)>> =
12126        (0..dictionaries.len()).map(|_| Vec::new()).collect();
12127    for (column, at, bytes) in made {
12128        done[column].push((at, bytes));
12129    }
12130    for (column, mut made) in done.into_iter().enumerate() {
12131        if made.is_empty() {
12132            continue;
12133        }
12134        let Some(held) = dictionaries[column].as_mut() else { continue };
12135        made.sort_by_key(|(at, _)| *at);
12136        let waiting = std::mem::take(&mut held.waiting);
12137        for ((at, _), (_, block)) in waiting.into_iter().zip(made) {
12138            if held.encoded() != at {
12139                return Err(Error::internal("a dictionary block was encoded out of order"));
12140            }
12141            held.push_block(block);
12142        }
12143    }
12144    Ok(())
12145}
12146
12147/// Which of [`payload_shapes`] comes out smallest over a sample of the blocks.
12148///
12149/// Every shape is encoded over the same sample and the smallest wins, which is the exhaustive
12150/// search moved up a level: over shapes of a column rather than over candidates of a chunk. The
12151/// sample is spread across the dictionary so that the first and last blocks are both in it, because
12152/// a dictionary written in first seen order has its common values at the front and its long tail at
12153/// the back, and those do not compress alike. Which blocks those are is
12154/// [`GlobalDictionary::seal`]'s to decide, because by the time this is called the rest of them have
12155/// been encoded and the raw bytes are gone.
12156fn settle_shape(sample: &[Vec<&[u8]>]) -> Result<chooser::Settled> {
12157    let mut best: Option<(chooser::Settled, usize)> = None;
12158    for shape in payload_shapes() {
12159        let mut size = 0;
12160        for block in sample {
12161            size += string::encode_with(block, &shape)?.len();
12162        }
12163        if best.as_ref().is_none_or(|(_, smallest)| size < *smallest) {
12164            best = Some((shape, size));
12165        }
12166    }
12167    best.map(|(shape, _)| shape)
12168        .ok_or_else(|| invalid("no shape applies to a global dictionary payload"))
12169}
12170
12171/// The sorted order laid out the way a reader reads it, in blocks of [`TEXT_RANK_BLOCK`] entries.
12172///
12173/// Each block holds its heads first and then its codes, rather than pairing them, because a search
12174/// asks for a head at every probe and for a code about once a search. Keeping the heads together
12175/// means a probe touches eight bytes of a block rather than twelve spread over it, and the last few
12176/// probes of a search, which are the ones that land in the same block, touch the same cache line.
12177fn encode_ranks(order: &[(u64, u32)], code_bits: usize) -> Result<(Vec<u8>, Vec<u64>)> {
12178    let mut out = Vec::with_capacity(order.len() * 4);
12179    let mut ends = Vec::with_capacity(order.len().div_ceil(TEXT_RANK_BLOCK));
12180    let mut heads = Vec::with_capacity(TEXT_RANK_BLOCK);
12181    let mut codes = Vec::with_capacity(TEXT_RANK_BLOCK);
12182    for block in order.chunks(TEXT_RANK_BLOCK) {
12183        // The order is sorted by value and a head is a prefix of a value, so the heads of a block
12184        // rise, the smallest is the first and the largest is the last.
12185        let base = block.first().map_or(0, |&(head, _)| head);
12186        let span = block.last().map_or(0, |&(head, _)| head.wrapping_sub(base));
12187        let width = (u64::BITS - span.leading_zeros()) as usize;
12188        heads.clear();
12189        codes.clear();
12190        for &(head, code) in block {
12191            heads.push(head.wrapping_sub(base));
12192            codes.push(u64::from(code));
12193        }
12194        put_u64(&mut out, base);
12195        out.push(width as u8);
12196        bitpack::pack_tail(&heads, width, &mut out)
12197            .map_err(|_| invalid("global dictionary heads do not pack"))?;
12198        bitpack::pack_tail(&codes, code_bits, &mut out)
12199            .map_err(|_| invalid("global dictionary codes do not pack"))?;
12200        ends.push(out.len() as u64);
12201    }
12202    Ok((out, ends))
12203}
12204
12205/// Opens a column's global dictionary, which reads its index and none of its payload.
12206///
12207/// `keep_budget` is how many decoded payload bytes this dictionary may hold on to, and every
12208/// caller bar the test of the ceiling passes [`TEXT_KEEP_BUDGET`]. It is a parameter rather than
12209/// the constant read where it is used because a test of a ceiling that cannot be moved has to build
12210/// a quarter of a gigabyte of dictionary to reach it.
12211fn open_global_dictionary(
12212    file: Arc<File>,
12213    page: Page,
12214    ty: &LogicalType,
12215    keep_budget: usize,
12216) -> Result<Vector> {
12217    if !coded_type(ty) {
12218        return Err(invalid("global dictionary belongs to a non-string column"));
12219    }
12220    let mut header = [0; DICTIONARY_HEADER];
12221    read_at(&file, page.offset, &mut header)?;
12222    let count = u32::from_le_bytes(header[0..4].try_into().expect("four bytes")) as usize;
12223    let per_block = u32::from_le_bytes(header[4..8].try_into().expect("four bytes")) as usize;
12224    let blocks = u32::from_le_bytes(header[8..12].try_into().expect("four bytes")) as usize;
12225    let width = u32::from_le_bytes(header[12..16].try_into().expect("four bytes"));
12226    let scattered = width & DICTIONARY_SCATTERED != 0;
12227    let has_grams = width & DICTIONARY_GRAMS != 0;
12228    let gram_width =
12229        if width & DICTIONARY_WIDE_GRAMS != 0 { TEXT_GRAM_BYTES } else { NARROW_GRAM_BYTES };
12230    let offset_bits = (width & !DICTIONARY_FLAGS) as usize;
12231    if per_block != TEXT_PAYLOAD_VALUES {
12232        return Err(invalid("global dictionary block width differs"));
12233    }
12234    if blocks != count.div_ceil(TEXT_PAYLOAD_VALUES) {
12235        return Err(invalid("global dictionary block count differs from its value count"));
12236    }
12237    if offset_bits > u32::BITS as usize {
12238        return Err(invalid("global dictionary packs offsets past a payload"));
12239    }
12240    let offset_len = offset_bytes(count, offset_bits);
12241    // The sorted order is kept out of the index on purpose. The index is read and checksummed in
12242    // full the moment the column is first touched, and the order is half again the size of the
12243    // offsets, so putting it there would make every query that reads a string column pay for a
12244    // search that most of them never make.
12245    let ranks = count;
12246    let rank_blocks = ranks.div_ceil(TEXT_RANK_BLOCK);
12247    // Three words a payload block, for where it starts, how long it is and what it hashes to, or
12248    // two of them on a file that has the blocks back to back and needs no start. Two a rank block
12249    // either way, since those are still one run.
12250    let payload_words = if scattered { 3 } else { 2 };
12251    let hash_len = blocks
12252        .checked_mul(payload_words * 8)
12253        .and_then(|len| len.checked_add(rank_blocks.checked_mul(16)?))
12254        .and_then(|len| len.checked_add(usize::from(has_grams) * 8))
12255        .ok_or_else(|| invalid("global dictionary block count overflow"))?;
12256    let gram_len = if has_grams {
12257        blocks
12258            .checked_mul(gram_width)
12259            .ok_or_else(|| invalid("global dictionary signature count overflow"))?
12260    } else {
12261        0
12262    };
12263    let index_len = DICTIONARY_HEADER
12264        .checked_add(offset_len)
12265        .and_then(|len| len.checked_add(hash_len))
12266        .ok_or_else(|| invalid("global dictionary header overflow"))?;
12267    if index_len > page.length as usize {
12268        return Err(invalid("global dictionary offset index exceeds its page"));
12269    }
12270    let mut index = vec![0; index_len];
12271    index[..DICTIONARY_HEADER].copy_from_slice(&header);
12272    read_at(&file, page.offset + DICTIONARY_HEADER as u64, &mut index[DICTIONARY_HEADER..])?;
12273    if checksum(&index) != page.hash {
12274        return Err(invalid("global dictionary index checksum differs"));
12275    }
12276    let word_end = index_len - usize::from(has_grams) * 8;
12277    let gram_hash = has_grams
12278        .then(|| u64::from_le_bytes(index[word_end..index_len].try_into().expect("eight bytes")));
12279    let mut words = index[DICTIONARY_HEADER + offset_len..word_end]
12280        .chunks_exact(8)
12281        .map(|part| u64::from_le_bytes(part.try_into().expect("eight bytes")))
12282        .collect::<Vec<_>>();
12283    let mut rest = words.split_off(blocks * payload_words);
12284    let rank_hashes = rest.split_off(rank_blocks);
12285    let rank_ends = rest;
12286    // A rank block packs its heads at whatever width its own values need, so its length is no longer
12287    // arithmetic on the block number and the reader has to be told where each one ends.
12288    if rank_ends.windows(2).any(|pair| pair[0] >= pair[1]) {
12289        return Err(invalid("global dictionary order blocks do not rise"));
12290    }
12291    let rank_len = usize::try_from(rank_ends.last().copied().unwrap_or_default())
12292        .map_err(|_| invalid("global dictionary rank overflow"))?;
12293    let body_len = index_len
12294        .checked_add(rank_len)
12295        .ok_or_else(|| invalid("global dictionary header overflow"))?;
12296    if body_len > page.length as usize {
12297        return Err(invalid("global dictionary order exceeds its page"));
12298    }
12299    let gram_end = body_len
12300        .checked_add(gram_len)
12301        .ok_or_else(|| invalid("global dictionary signature length overflow"))?;
12302    if gram_end > page.length as usize {
12303        return Err(invalid("global dictionary signatures exceed their page"));
12304    }
12305    let grams = gram_hash.map(|hash| NativeGrams {
12306        start: page.offset + body_len as u64,
12307        length: gram_len,
12308        width: gram_width,
12309        hash,
12310        verdicts: Mutex::new(Vec::new()),
12311    });
12312    // The offsets stay where they were read, behind the header, rather than being copied out. On a
12313    // dictionary of millions of values they are megabytes, and a copy is as many fresh pages to
12314    // fault in again on a query that may want a handful of strings.
12315    let mut offsets = index;
12316    offsets.truncate(DICTIONARY_HEADER + offset_len);
12317    let hashes = words.split_off(blocks * (payload_words - 1));
12318    let (starts, lengths) = if scattered {
12319        let mut starts = Vec::with_capacity(blocks);
12320        let mut lengths = Vec::with_capacity(blocks);
12321        for pair in words.chunks_exact(2) {
12322            starts.push(pair[0]);
12323            lengths.push(pair[1]);
12324        }
12325        (starts, lengths)
12326    } else {
12327        // A file written before the blocks said where they were has them behind one another at the
12328        // end of the page, so the base is where the sorted order stops and each end is the start of
12329        // the one after it. Turning them round here is what lets everything below take one shape.
12330        let base = page.offset + gram_end as u64;
12331        let mut starts = Vec::with_capacity(blocks);
12332        let mut lengths = Vec::with_capacity(blocks);
12333        let mut at = 0_u64;
12334        for &end in &words {
12335            let len = end
12336                .checked_sub(at)
12337                .ok_or_else(|| invalid("global dictionary block ends before it starts"))?;
12338            starts.push(base + at);
12339            lengths.push(len);
12340            at = end;
12341        }
12342        (starts, lengths)
12343    };
12344    // What the offsets bound is the decoded payload, and what the page length counts is the stored
12345    // one, so on a format 26 file the block lengths adding up to the rest of the page is the one
12346    // thing that ties the index to the page. From format 27 the blocks are written during the load
12347    // and the page is only the index and the order, so there the most that can be said is that
12348    // every block is somewhere in the file past its header.
12349    let stored_len = page.length as u64 - gram_end as u64;
12350    if scattered && stored_len == 0 {
12351        let size = file.metadata().map_err(io)?.len();
12352        let inside = starts.iter().zip(&lengths).all(|(&start, &len)| {
12353            start >= HEADER && start.checked_add(len).is_some_and(|end| end <= size)
12354        });
12355        if !inside {
12356            return Err(invalid("global dictionary block lies outside the file"));
12357        }
12358    } else if lengths.iter().try_fold(0_u64, |sum, len| sum.checked_add(*len)) != Some(stored_len) {
12359        return Err(invalid("global dictionary blocks do not bound the payload"));
12360    }
12361    Vector::external_text(
12362        ty.clone(),
12363        Arc::new(NativeText {
12364            file,
12365            values: count,
12366            offsets,
12367            offset_bits,
12368            value_ends: OnceLock::new(),
12369            value_lens: OnceLock::new(),
12370            ends_asked: AtomicUsize::new(0),
12371            ranks,
12372            rank_at: page.offset + index_len as u64,
12373            rank_ends,
12374            rank_hashes,
12375            rank_blocks: (0..rank_blocks).map(|_| OnceLock::new()).collect(),
12376            code_bits: code_width(count),
12377            code_ranks: OnceLock::new(),
12378            starts,
12379            lengths,
12380            hashes,
12381            grams,
12382            blocks: (0..blocks).map(|_| OnceLock::new()).collect(),
12383            char_lens: (0..blocks).map(|_| OnceLock::new()).collect(),
12384            keep_budget,
12385            payload_kept: AtomicUsize::new(0),
12386            swept: (0..blocks).map(|_| AtomicBool::new(false)).collect(),
12387            visit_dropped: AtomicUsize::new(0),
12388            searched: Mutex::new(HashMap::new()),
12389        }),
12390    )
12391}
12392
12393/// What a stored page is, without decoding a value out of it.
12394///
12395/// Two layers, and both of them belong in the answer. The codec byte at the front of every page is
12396/// the format's own choice, and it is what says whether the column came back as codes into a table
12397/// wide dictionary, as a bit packed page, as an encoding cascade or as the bytes themselves. Under
12398/// the cascade codecs there is a second choice the encoder made per chunk, and that is what
12399/// [`integer::describe`] and [`string::describe`] already write out as `DICT(PACKED, PACKED)`.
12400///
12401/// This mirrors the tags [`decode`] reads and has to be kept beside it. A page whose header this
12402/// cannot walk comes back as text rather than as an error, because a caller asking what a file
12403/// looks like is usually asking because something is wrong with it, and a report that stops at the
12404/// first bad page is a report that says nothing about the other nine hundred.
12405fn page_encoding(ty: &LogicalType, rows: usize, bytes: &[u8]) -> String {
12406    /// The page header is the codec, the validity tag and, for a page that stores a mask, the mask.
12407    fn cascade_at(rows: usize, bytes: &[u8]) -> Result<(u8, usize)> {
12408        let mut cur = Cursor::new(bytes);
12409        let codec = cur.u8()?;
12410        if cur.u8()? == 2 {
12411            cur.take(rows.div_ceil(8))?;
12412        }
12413        Ok((codec, cur.at))
12414    }
12415    let Ok((codec, at)) = cascade_at(rows, bytes) else {
12416        return "UNREADABLE".to_string();
12417    };
12418    let tail = &bytes[at..];
12419    let described = |described: Result<String>| described.unwrap_or_else(|_| "UNREADABLE".into());
12420    match codec {
12421        0 => match ty {
12422            LogicalType::Varchar | LogicalType::Blob => "PLAIN".to_string(),
12423            _ => "FIXED".to_string(),
12424        },
12425        1 => "DICT(PLAIN)".to_string(),
12426        2 => "FOR+BITPACK".to_string(),
12427        3 => "TABLE DICT".to_string(),
12428        4 => format!("TABLE DICT({})", described(integer::describe(tail))),
12429        5 => described(integer::describe(tail)),
12430        6 => described(string::describe(tail)),
12431        other => format!("CODEC {other}"),
12432    }
12433}
12434
12435/// Selected stable dictionary codes from one page.
12436///
12437/// Pair-frequency construction needs at most the bounded heavy-hitter rows. Reading those code
12438/// positions directly avoids materializing every code in each part that contains a candidate.
12439fn decode_selected_stable_codes(
12440    rows: usize,
12441    bytes: &[u8],
12442    positions: &[usize],
12443    out: &mut Vec<Option<u32>>,
12444) -> Result<bool> {
12445    if positions.windows(2).any(|pair| pair[0] >= pair[1])
12446        || positions.last().is_some_and(|&position| position >= rows)
12447    {
12448        return Err(invalid("selected code positions are not sorted and in range"));
12449    }
12450    let mut cur = Cursor::new(bytes);
12451    let codec = cur.u8()?;
12452    if codec != 3 && codec != 4 {
12453        return Ok(false);
12454    }
12455    let flag = cur.u8()?;
12456    let mask = match flag {
12457        0 | 1 => None,
12458        2 => {
12459            let at = cur.at;
12460            let len = rows.div_ceil(8);
12461            cur.take(len)?;
12462            Some((at, len))
12463        }
12464        _ => return Err(invalid("page validity tag differs")),
12465    };
12466    let valid = |row: usize| match flag {
12467        0 => true,
12468        1 => false,
12469        2 => mask.is_some_and(|(at, _)| bytes[at + row / 8] >> (row % 8) & 1 == 1),
12470        _ => unreachable!("the validity tag was checked"),
12471    };
12472    if codec == 4 {
12473        let wide = integer::decode_selected(&bytes[cur.at..], positions)?;
12474        for (&row, code) in positions.iter().zip(wide) {
12475            let code = u32::try_from(code).map_err(|_| invalid("code is not a code"))?;
12476            out.push(valid(row).then_some(code));
12477        }
12478        return Ok(true);
12479    }
12480    let codes_at = cur.at;
12481    let codes_len = rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?;
12482    cur.take(codes_len)?;
12483    if cur.at != bytes.len() {
12484        return Err(invalid("global code page has trailing bytes"));
12485    }
12486    let codes = &bytes[codes_at..codes_at + codes_len];
12487    for &row in positions {
12488        let at = row.checked_mul(4).ok_or_else(|| invalid("dictionary code offset overflow"))?;
12489        let code = u32::from_le_bytes(
12490            codes[at..at + 4].try_into().map_err(|_| invalid("dictionary code is truncated"))?,
12491        );
12492        out.push(valid(row).then_some(code));
12493    }
12494    Ok(true)
12495}
12496
12497/// [`decode`] of only the rows at `positions`, which rise.
12498///
12499/// A compressed text page decompresses only those rows, see [`string::decode_flat_at`], and checks
12500/// only those rows are text. Every other page is decoded whole and gathered, since its values are
12501/// fixed width or its strings are shared through a dictionary, and there picking comes after.
12502fn decode_at(
12503    ty: &LogicalType,
12504    rows: usize,
12505    bytes: &[u8],
12506    global: Option<Arc<Vector>>,
12507    positions: &[u32],
12508) -> Result<Vector> {
12509    if positions.last().is_some_and(|&last| last as usize >= rows) {
12510        return Err(invalid("a position is past the end of the part"));
12511    }
12512    // Past about one row in eight, unpacking the whole part and picking the rows out is the cheaper
12513    // of the two, since a unit unpacks at a fraction of what a row unpacked on its own costs.
12514    if bytes.first() == Some(&5)
12515        && positions.len().saturating_mul(8) <= rows
12516        // Past the codec, the validity flag and the mask a flag of 2 has.
12517        && bytes
12518            .get(2 + if bytes.get(1) == Some(&2) { rows.div_ceil(8) } else { 0 }..)
12519            .is_some_and(integer::pointed)
12520    {
12521        return cascade_at(ty, rows, bytes, positions);
12522    }
12523    if bytes.first() != Some(&6) {
12524        return decode(ty, rows, bytes, global)?.gather(positions);
12525    }
12526    if !coded_type(ty) {
12527        return Err(invalid("compressed text codec belongs to a non-string page"));
12528    }
12529    let mut cur = Cursor::new(bytes);
12530    cur.u8()?;
12531    let validity = match cur.u8()? {
12532        0 => Validity::AllValid,
12533        1 => Validity::AllInvalid,
12534        2 => {
12535            let mask = cur.take(rows.div_ceil(8))?;
12536            Validity::from_iter(positions.len(), |at| {
12537                let row = positions[at] as usize;
12538                mask[row / 8] >> (row % 8) & 1 == 1
12539            })
12540        }
12541        _ => return Err(invalid("page validity tag differs")),
12542    };
12543    let (payload, ends) = string::decode_flat_at(&bytes[cur.at..], positions)?.into_parts();
12544    let mut values = StringColumn::over(Buffer::from_vec(payload).into_page());
12545    push_values(&mut values, ty, &ends)?;
12546    Ok(Vector::flat(ty.clone(), Data::Varlen(values))?.with_validity(validity))
12547}
12548
12549/// The rows `positions` names of an integer cascade page, unpacked at those rows alone.
12550///
12551/// A scan whose join keeps a few rows in a thousand reads its other columns only at those rows, and
12552/// decoding the whole part to pick them out afterwards was most of what it cost. In TPC-H q17 the
12553/// bitmap over the parts of one brand and container keeps about one `lineitem` row in a thousand.
12554fn cascade_at(ty: &LogicalType, rows: usize, bytes: &[u8], positions: &[u32]) -> Result<Vector> {
12555    fn wanted<T: integer::Lane>(values: &[i64]) -> Result<Vec<T>> {
12556        values
12557            .iter()
12558            .map(|&value| T::fit(value).ok_or_else(|| invalid("page value is not of its type")))
12559            .collect()
12560    }
12561    let mut cur = Cursor::new(bytes);
12562    cur.u8()?;
12563    let validity = match cur.u8()? {
12564        0 => Validity::AllValid,
12565        1 => Validity::AllInvalid,
12566        2 => {
12567            let mask = cur.take(rows.div_ceil(8))?;
12568            Validity::from_iter(positions.len(), |at| {
12569                let row = positions[at] as usize;
12570                mask[row / 8] >> (row % 8) & 1 == 1
12571            })
12572        }
12573        _ => return Err(invalid("page validity tag differs")),
12574    };
12575    let at: Vec<usize> = positions.iter().map(|&row| row as usize).collect();
12576    let values = integer::decode_selected(&bytes[cur.at..], &at)
12577        .map_err(|error| invalid(&format!("page value is not of its type: {error}")))?;
12578    if values.len() != positions.len() {
12579        return Err(invalid("cascade page holds the wrong number of rows"));
12580    }
12581    let data = match ty {
12582        LogicalType::TinyInt => Data::Int8(wanted::<i8>(&values)?.into()),
12583        LogicalType::UTinyInt => Data::UInt8(wanted::<u8>(&values)?.into()),
12584        LogicalType::SmallInt => Data::Int16(wanted::<i16>(&values)?.into()),
12585        LogicalType::USmallInt => Data::UInt16(wanted::<u16>(&values)?.into()),
12586        LogicalType::Integer | LogicalType::Date => Data::Int32(wanted::<i32>(&values)?.into()),
12587        LogicalType::UInteger => Data::UInt32(wanted::<u32>(&values)?.into()),
12588        LogicalType::BigInt
12589        | LogicalType::Timestamp
12590        | LogicalType::Time
12591        | LogicalType::TimeTz
12592        | LogicalType::TimestampTz
12593        | LogicalType::TimestampS
12594        | LogicalType::TimestampMs
12595        | LogicalType::TimestampNs => Data::Int64(values.into()),
12596        LogicalType::Decimal { .. } => match ty.physical() {
12597            PhysicalType::Int16 => Data::Int16(wanted::<i16>(&values)?.into()),
12598            PhysicalType::Int32 => Data::Int32(wanted::<i32>(&values)?.into()),
12599            PhysicalType::Int64 => Data::Int64(values.into()),
12600            _ => return Err(invalid("cascade codec belongs to a decimal that is not an integer")),
12601        },
12602        _ => return Err(invalid("cascade codec belongs to a page that is not integers")),
12603    };
12604    Ok(Vector::flat(ty.clone(), data)?.with_validity(validity))
12605}
12606
12607/// The values of a string or blob page, laid end to end in the page's payload from its start, each
12608/// ending where `ends` says. A varchar is checked for text on the way in, once over the whole run,
12609/// and a blob or a bit string is not, since neither ever claimed to hold any.
12610fn push_values(values: &mut StringColumn, ty: &LogicalType, ends: &[usize]) -> Result<()> {
12611    if ty == &LogicalType::Varchar {
12612        return values.push_run_in_place(0, ends);
12613    }
12614    let mut start = 0;
12615    for &end in ends {
12616        let len = end
12617            .checked_sub(start)
12618            .ok_or_else(|| invalid("a string value ends before it starts"))?;
12619        values.push_bytes_in_place(start, len)?;
12620        start = end;
12621    }
12622    Ok(())
12623}
12624
12625fn decode(
12626    ty: &LogicalType,
12627    rows: usize,
12628    bytes: &[u8],
12629    global: Option<Arc<Vector>>,
12630) -> Result<Vector> {
12631    let mut cur = Cursor::new(bytes);
12632    let codec = cur.u8()?;
12633    let flag = cur.u8()?;
12634    let validity = match flag {
12635        0 => Validity::AllValid,
12636        1 => Validity::AllInvalid,
12637        2 => {
12638            let mask = cur.take(rows.div_ceil(8))?;
12639            Validity::from_iter(rows, |row| mask[row / 8] >> (row % 8) & 1 == 1)
12640        }
12641        _ => return Err(invalid("page validity tag differs")),
12642    };
12643    if codec == 1 {
12644        if !coded_type(ty) {
12645            return Err(invalid("dictionary codec belongs to a non-string page"));
12646        }
12647        let count = cur.u32()? as usize;
12648        let payload_len = cur.u32()? as usize;
12649        let offset_bytes = cur.take(
12650            (count + 1)
12651                .checked_mul(4)
12652                .ok_or_else(|| invalid("dictionary offset count overflow"))?,
12653        )?;
12654        let offsets = offset_bytes
12655            .chunks_exact(4)
12656            .map(|part| u32::from_le_bytes(part.try_into().expect("four bytes")))
12657            .collect::<Vec<_>>();
12658        let payload = cur.take(payload_len)?.to_vec();
12659        if offsets.first() != Some(&0)
12660            || offsets.last().copied().map(|last| last as usize) != Some(payload.len())
12661            || offsets.windows(2).any(|pair| pair[0] > pair[1])
12662        {
12663            return Err(invalid("dictionary offsets do not bound the payload"));
12664        }
12665        // A page, because every chunk cut out of this dictionary points at the same payload and a
12666        // page is what lets a cut be the views and nothing else.
12667        let mut strings = StringColumn::over(Buffer::from_vec(payload).into_page());
12668        let ends: Vec<usize> = offsets[1..].iter().map(|&end| end as usize).collect();
12669        push_values(&mut strings, ty, &ends)?;
12670        let mut codes = Vec::with_capacity(rows);
12671        for _ in 0..rows {
12672            codes.push(cur.u32()?);
12673        }
12674        if codes.iter().any(|code| *code as usize >= count) {
12675            return Err(invalid("dictionary code is out of range"));
12676        }
12677        if cur.at != bytes.len() {
12678            return Err(invalid("dictionary page has trailing bytes"));
12679        }
12680        let dictionary = Vector::flat(ty.clone(), Data::Varlen(strings))?;
12681        return Ok(Vector::dictionary(codes, dictionary)?.with_validity(validity));
12682    }
12683    if codec == 3 || codec == 4 {
12684        let dictionary = global.ok_or_else(|| invalid("global code page has no dictionary"))?;
12685        let codes = if codec == 4 {
12686            // The cascade holds the whole tail of the page and says how long it is itself, so the
12687            // check that nothing is left over is the one the decoder already makes.
12688            // Straight into `u32`, which is also the check that every code is one: a code outside
12689            // it is a corrupt file and the decoder refuses it, a block at a time where it can.
12690            let codes = integer::decode_as::<u32>(&bytes[cur.at..])
12691                .map_err(|error| invalid(&format!("code is not a code: {error}")))?;
12692            if codes.len() != rows {
12693                return Err(invalid("encoded code page holds the wrong number of rows"));
12694            }
12695            codes
12696        } else {
12697            let mut codes = Vec::with_capacity(rows);
12698            for _ in 0..rows {
12699                codes.push(cur.u32()?);
12700            }
12701            if cur.at != bytes.len() {
12702                return Err(invalid("global code page has trailing bytes"));
12703            }
12704            codes
12705        };
12706        let highest = codes.iter().copied().max();
12707        return Ok(Vector::stable_dictionary_validated(codes, dictionary, highest)?
12708            .with_validity(validity));
12709    }
12710    if codec == 6 {
12711        if !coded_type(ty) {
12712            return Err(invalid("compressed text codec belongs to a non-string page"));
12713        }
12714        // As codec 5, the layer holds the whole tail of the page and says how long it is itself.
12715        // It comes back as one buffer with the values laid end to end and where each one ends, which
12716        // is the raw form's layout, so what is left to do here is what codec 0 does.
12717        let (payload, ends) = string::decode_flat(&bytes[cur.at..])?.into_parts();
12718        if ends.len() != rows {
12719            return Err(invalid("compressed text page holds the wrong number of rows"));
12720        }
12721        // A page, because this is read once and handed out a chunk at a time, and a cut of a paged
12722        // payload moves views rather than bytes.
12723        let mut values = StringColumn::over(Buffer::from_vec(payload).into_page());
12724        push_values(&mut values, ty, &ends)?;
12725        return Ok(Vector::flat(ty.clone(), Data::Varlen(values))?.with_validity(validity));
12726    }
12727    if codec == 5 {
12728        // The cascade holds the whole tail of the page and says how long it is itself.
12729        let data = cascade(ty, &bytes[cur.at..], rows)?;
12730        return Ok(Vector::flat(ty.clone(), data)?.with_validity(validity));
12731    }
12732    if codec == 2 {
12733        let width = u32::from(cur.u8()?);
12734        let base = i128::from_le_bytes(cur.take(16)?.try_into().expect("sixteen bytes"));
12735        let count = cur.u32()? as usize;
12736        let length = count.checked_mul(8).ok_or_else(|| invalid("packed page is too long"))?;
12737        let words: Vec<u64> = cur
12738            .take(length)?
12739            .chunks_exact(8)
12740            .map(|word| u64::from_le_bytes(word.try_into().expect("eight bytes")))
12741            .collect();
12742        if cur.at != bytes.len() {
12743            return Err(invalid("packed page has trailing bytes"));
12744        }
12745        return Ok(Vector::packed(ty.clone(), words, width, base, rows)?.with_validity(validity));
12746    }
12747    if codec != 0 {
12748        return Err(invalid("page codec is unknown"));
12749    }
12750    let data = match ty {
12751        LogicalType::TinyInt => {
12752            let values = cur.take(rows)?;
12753            Data::Int8(values.iter().map(|item| *item as i8).collect::<Vec<_>>().into())
12754        }
12755        LogicalType::UTinyInt => Data::UInt8(cur.take(rows)?.to_vec().into()),
12756        LogicalType::SmallInt => {
12757            let values =
12758                cur.take(rows.checked_mul(2).ok_or_else(|| invalid("page size overflow"))?)?;
12759            Data::Int16(
12760                values
12761                    .chunks_exact(2)
12762                    .map(|item| i16::from_le_bytes(item.try_into().expect("two bytes")))
12763                    .collect::<Vec<_>>()
12764                    .into(),
12765            )
12766        }
12767        LogicalType::USmallInt => {
12768            let values =
12769                cur.take(rows.checked_mul(2).ok_or_else(|| invalid("page size overflow"))?)?;
12770            Data::UInt16(
12771                values
12772                    .chunks_exact(2)
12773                    .map(|item| u16::from_le_bytes(item.try_into().expect("two bytes")))
12774                    .collect::<Vec<_>>()
12775                    .into(),
12776            )
12777        }
12778        LogicalType::UInteger => {
12779            let values =
12780                cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
12781            Data::UInt32(
12782                values
12783                    .chunks_exact(4)
12784                    .map(|item| u32::from_le_bytes(item.try_into().expect("four bytes")))
12785                    .collect::<Vec<_>>()
12786                    .into(),
12787            )
12788        }
12789        LogicalType::UBigInt => {
12790            let values =
12791                cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
12792            Data::UInt64(
12793                values
12794                    .chunks_exact(8)
12795                    .map(|item| u64::from_le_bytes(item.try_into().expect("eight bytes")))
12796                    .collect::<Vec<_>>()
12797                    .into(),
12798            )
12799        }
12800        LogicalType::Integer | LogicalType::Date => {
12801            let values =
12802                cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
12803            Data::Int32(
12804                values
12805                    .chunks_exact(4)
12806                    .map(|item| i32::from_le_bytes(item.try_into().expect("four bytes")))
12807                    .collect::<Vec<_>>()
12808                    .into(),
12809            )
12810        }
12811        LogicalType::BigInt
12812        | LogicalType::Timestamp
12813        | LogicalType::Time
12814        | LogicalType::TimeTz
12815        | LogicalType::TimestampTz
12816        | LogicalType::TimestampS
12817        | LogicalType::TimestampMs
12818        | LogicalType::TimestampNs => {
12819            let values =
12820                cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
12821            Data::Int64(
12822                values
12823                    .chunks_exact(8)
12824                    .map(|item| i64::from_le_bytes(item.try_into().expect("eight bytes")))
12825                    .collect::<Vec<_>>()
12826                    .into(),
12827            )
12828        }
12829        LogicalType::HugeInt | LogicalType::Uuid => {
12830            let values =
12831                cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
12832            Data::Int128(
12833                values
12834                    .chunks_exact(16)
12835                    .map(|item| i128::from_le_bytes(item.try_into().expect("sixteen bytes")))
12836                    .collect::<Vec<_>>()
12837                    .into(),
12838            )
12839        }
12840        LogicalType::UHugeInt => {
12841            let values =
12842                cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
12843            Data::UInt128(
12844                values
12845                    .chunks_exact(16)
12846                    .map(|item| u128::from_le_bytes(item.try_into().expect("sixteen bytes")))
12847                    .collect::<Vec<_>>()
12848                    .into(),
12849            )
12850        }
12851        LogicalType::Float => {
12852            let values =
12853                cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
12854            Data::Float32(
12855                values
12856                    .chunks_exact(4)
12857                    .map(|item| f32::from_le_bytes(item.try_into().expect("four bytes")))
12858                    .collect::<Vec<_>>()
12859                    .into(),
12860            )
12861        }
12862        LogicalType::Double => {
12863            let values =
12864                cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
12865            Data::Float64(
12866                values
12867                    .chunks_exact(8)
12868                    .map(|item| f64::from_le_bytes(item.try_into().expect("eight bytes")))
12869                    .collect::<Vec<_>>()
12870                    .into(),
12871            )
12872        }
12873        LogicalType::Interval => {
12874            let values =
12875                cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
12876            Data::Interval(
12877                values
12878                    .chunks_exact(16)
12879                    .map(|item| {
12880                        (
12881                            i32::from_le_bytes(item[..4].try_into().expect("four bytes")),
12882                            i32::from_le_bytes(item[4..8].try_into().expect("four bytes")),
12883                            i64::from_le_bytes(item[8..].try_into().expect("eight bytes")),
12884                        )
12885                    })
12886                    .collect::<Vec<_>>()
12887                    .into(),
12888            )
12889        }
12890        LogicalType::Boolean => {
12891            let values = cur.take(rows)?;
12892            if values.iter().any(|value| *value > 1) {
12893                return Err(invalid("boolean page has another value"));
12894            }
12895            Data::Bool(values.iter().map(|value| *value == 1).collect::<Vec<_>>().into())
12896        }
12897        // Whichever integer the declared width says, which is the mapping the rest of the engine
12898        // already uses for a decimal in memory.
12899        LogicalType::Decimal { .. } => match ty.physical() {
12900            PhysicalType::Int16 => {
12901                let values =
12902                    cur.take(rows.checked_mul(2).ok_or_else(|| invalid("page size overflow"))?)?;
12903                Data::Int16(
12904                    values
12905                        .chunks_exact(2)
12906                        .map(|item| i16::from_le_bytes(item.try_into().expect("two bytes")))
12907                        .collect::<Vec<_>>()
12908                        .into(),
12909                )
12910            }
12911            PhysicalType::Int32 => {
12912                let values =
12913                    cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
12914                Data::Int32(
12915                    values
12916                        .chunks_exact(4)
12917                        .map(|item| i32::from_le_bytes(item.try_into().expect("four bytes")))
12918                        .collect::<Vec<_>>()
12919                        .into(),
12920                )
12921            }
12922            PhysicalType::Int64 => {
12923                let values =
12924                    cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
12925                Data::Int64(
12926                    values
12927                        .chunks_exact(8)
12928                        .map(|item| i64::from_le_bytes(item.try_into().expect("eight bytes")))
12929                        .collect::<Vec<_>>()
12930                        .into(),
12931                )
12932            }
12933            _ => {
12934                let values =
12935                    cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
12936                Data::Int128(
12937                    values
12938                        .chunks_exact(16)
12939                        .map(|item| i128::from_le_bytes(item.try_into().expect("sixteen bytes")))
12940                        .collect::<Vec<_>>()
12941                        .into(),
12942                )
12943            }
12944        },
12945        LogicalType::Varchar | LogicalType::Blob | LogicalType::Bit => {
12946            let offset_bytes = cur
12947                .take((rows + 1).checked_mul(4).ok_or_else(|| invalid("offset count overflow"))?)?;
12948            let offsets = offset_bytes
12949                .chunks_exact(4)
12950                .map(|part| u32::from_le_bytes(part.try_into().expect("four bytes")))
12951                .collect::<Vec<_>>();
12952            let payload = cur.take(bytes.len() - cur.at)?.to_vec();
12953            if offsets.first() != Some(&0)
12954                || offsets.last().copied().map(|last| last as usize) != Some(payload.len())
12955                || offsets.windows(2).any(|pair| pair[0] > pair[1])
12956            {
12957                return Err(invalid("string offsets do not bound the payload"));
12958            }
12959            // A page for the reason the dictionary payload above is one: the page is read once and
12960            // handed out a chunk at a time, and a cut of a paged payload moves views rather than
12961            // bytes.
12962            //
12963            // A varchar is checked for text on the way in and a blob and a bit string are not,
12964            // because the second pair never claimed to hold any. Reading them through the checking
12965            // seam would refuse a column for holding exactly what it was told to hold.
12966            let mut values = StringColumn::over(Buffer::from_vec(payload).into_page());
12967            let ends: Vec<usize> = offsets[1..].iter().map(|&end| end as usize).collect();
12968            push_values(&mut values, ty, &ends)?;
12969            Data::Varlen(values)
12970        }
12971        _ => return Err(Error::not_implemented(format!("native page for {ty}"))),
12972    };
12973    if cur.at != bytes.len() {
12974        return Err(invalid("page has trailing bytes"));
12975    }
12976    Ok(Vector::flat(ty.clone(), data)?.with_validity(validity))
12977}
12978
12979#[cfg(test)]
12980mod tests {
12981    use std::fs::{self, OpenOptions};
12982    use std::io::{Seek, SeekFrom, Write};
12983    use std::path::PathBuf;
12984    use std::time::{SystemTime, UNIX_EPOCH};
12985
12986    use rudb_common::Stat;
12987    use rudb_common::Value;
12988    use rudb_common::bounds::{Frequencies, Op, Remainder, Zones};
12989    use rudb_common::stat::Provenance;
12990
12991    use super::*;
12992
12993    #[test]
12994    fn head_is_the_value_padded_to_eight_bytes() {
12995        let bytes: Vec<u8> = (1..=12).collect();
12996        for len in 0..=bytes.len() {
12997            let value = &bytes[..len];
12998            let mut word = [0; 8];
12999            let take = len.min(8);
13000            word[..take].copy_from_slice(&value[..take]);
13001            assert_eq!(head(value), u64::from_be_bytes(word), "{len} bytes");
13002        }
13003        assert!(head(b"ab") < head(b"ab\x01"));
13004        assert!(head(b"abcd") < head(b"abce"));
13005    }
13006
13007    #[test]
13008    fn spanned_frequency_header_rejects_missing_or_out_of_bounds_payloads() {
13009        for (length, entries) in [(0_u32, 1_u32), (9, 0), (1, FREQUENCY_ENTRIES as u32 + 1)] {
13010            let mut bytes = Vec::new();
13011            put_u32(&mut bytes, length);
13012            put_u32(&mut bytes, entries);
13013            bytes.push(1);
13014            assert!(summary_span(&mut Cursor::new(&bytes)).is_err());
13015        }
13016        let mut bytes = Vec::new();
13017        put_u32(&mut bytes, 1);
13018        put_u32(&mut bytes, 0);
13019        bytes.push(1);
13020        assert_eq!(summary_span(&mut Cursor::new(&bytes)).expect("one byte"), Some((1, 0)));
13021    }
13022
13023    #[test]
13024    fn a_name_taken_in_pieces_is_the_name_of_the_pieces_joined() {
13025        let bytes: Vec<u8> =
13026            (0..300_u32).map(|at| (at.wrapping_mul(2_654_435_761) >> 13) as u8).collect();
13027        for length in [0, 1, 7, 31, 32, 33, 63, 64, 65, 100, 300] {
13028            let whole = content_name(&bytes[..length]);
13029            for step in [1, 3, 8, 31, 32, 33, 64, 301] {
13030                let mut namer = ContentNamer::default();
13031                bytes[..length].chunks(step).for_each(|piece| namer.update(piece));
13032                assert_eq!(namer.finish(), whole, "{length} bytes in pieces of {step}");
13033            }
13034        }
13035    }
13036
13037    /// The chooser as it was before it could rule kinds out up front: the same narrowing, with every
13038    /// kind tested for. What it writes is what the file used to hold.
13039    #[derive(Debug)]
13040    struct TestsEverything<'a>(&'a dyn chooser::Chooser);
13041
13042    impl chooser::Chooser for TestsEverything<'_> {
13043        fn name(&self) -> &'static str {
13044            "tests everything"
13045        }
13046
13047        fn narrow_strings(
13048            &self,
13049            values: &[&[u8]],
13050            offered: &[string::Kind],
13051            depth: u8,
13052        ) -> Vec<string::Kind> {
13053            self.0.narrow_strings(values, offered, depth)
13054        }
13055
13056        fn narrow_integers(
13057            &self,
13058            values: &[i64],
13059            offered: &[integer::Kind],
13060            depth: u8,
13061        ) -> Vec<integer::Kind> {
13062            self.0.narrow_integers(values, offered, depth)
13063        }
13064    }
13065
13066    #[test]
13067    fn ruling_kinds_out_before_testing_for_them_writes_the_same_bytes() {
13068        let columns: Vec<Vec<i64>> = vec![
13069            vec![],
13070            vec![5; 1000],
13071            (0..1000).collect(),
13072            (0..1000).map(|row| 1_600_000_000_000_000 + row * 1_000_000).collect(),
13073            (0..1000).map(|row| row / 50).collect(),
13074            (0..1000).map(|row| if row % 97 == 0 { row } else { 0 }).collect(),
13075            (0..1000).map(|row| (row * 7919) % 13).collect(),
13076            (0..1000).map(|row| (row * 2_654_435_761) % 1_000_003).collect(),
13077            (0..1000).map(|row| [3, 3, 3, 9, 9, 1][row as usize % 6]).collect(),
13078            (0..1000).map(|row| i64::MIN + row % 3).collect(),
13079        ];
13080        let choosers: [&dyn chooser::Chooser; 2] = [&Fixed, &Codes];
13081        for column in &columns {
13082            for chooser in choosers {
13083                let quick = integer::encode_with(column, chooser).unwrap();
13084                let full = integer::encode_with(column, &TestsEverything(chooser)).unwrap();
13085                assert_eq!(
13086                    quick,
13087                    full,
13088                    "{} on {:?}",
13089                    chooser.name(),
13090                    &column[..column.len().min(8)]
13091                );
13092            }
13093        }
13094    }
13095
13096    /// Parts of a column that all look alike come out of a settled shape byte for byte as they
13097    /// come out of a search, because the search would have kept the same tree on every one.
13098    #[test]
13099    fn parts_that_look_alike_replay_to_the_bytes_a_search_writes() {
13100        let mut settling = Settling::default();
13101        for part in 0..STRIPE_PARTS as i64 {
13102            let values: Vec<i64> = (0..2048)
13103                .map(|row| 1_600_000_000_000_000 + (part * 2048 + row) * 1_000_000 + row % 7)
13104                .collect();
13105            let searched = integer::encode_with(&values, &Fixed).unwrap();
13106            assert_eq!(settling.encode(&values).unwrap(), searched, "part {part}");
13107        }
13108    }
13109
13110    /// Text pages compressed against a table an earlier page trained read back as they went in, and
13111    /// a page of different text trains a table of its own rather than coming out as big as the
13112    /// earlier table would make it.
13113    #[test]
13114    fn text_pages_share_a_table_until_the_text_changes() {
13115        let words = ["carefully", "final", "deposits", "sleep", "furiously", "quickly", "among"];
13116        let english: Vec<Vec<u8>> = (0..1024)
13117            .map(|row: usize| {
13118                let pick = |at: usize| words[(row * 7 + at * 3) % words.len()];
13119                format!("{} {} {} the {}", pick(0), pick(1), pick(2), pick(3)).into_bytes()
13120            })
13121            .collect();
13122        let digits: Vec<Vec<u8>> =
13123            (0..1024_u64).map(|row| format!("{:020}", row * 7_919_993).into_bytes()).collect();
13124        let mut settling = Settling::default();
13125        for page in 0..8 {
13126            let values: Vec<&[u8]> =
13127                if page < 4 { &english } else { &digits }.iter().map(Vec::as_slice).collect();
13128            let payload = values.iter().map(|value| value.len()).sum();
13129            let out = settling.text(&values, payload).unwrap().unwrap();
13130            assert_eq!(string::decode(&out).unwrap(), values, "page {page}");
13131            let alone = string::encode_only(string::Kind::Fsst, &values).unwrap().unwrap();
13132            assert!(
13133                out.len() * 4 <= alone.len() * 5,
13134                "page {page}: {} against {}",
13135                out.len(),
13136                alone.len()
13137            );
13138            let since = settling.symbols.as_ref().unwrap().since;
13139            assert_eq!(since, page % 4, "page {page}");
13140        }
13141    }
13142
13143    /// A column that changes shape partway through a stripe still reads back, and no part comes
13144    /// out much bigger than a search would have made it, because a replay that stops fitting or
13145    /// grows past a quarter a row is searched.
13146    #[test]
13147    fn a_column_that_changes_under_the_shape_is_searched_again() {
13148        let mut state = 0x9e37_79b9_7f4a_7c15_u64;
13149        let mut noise = move || {
13150            state ^= state << 13;
13151            state ^= state >> 7;
13152            state ^= state << 17;
13153            (state % 1_000_000) as i64
13154        };
13155        let mut settling = Settling::default();
13156        for part in 0..STRIPE_PARTS as i64 {
13157            let values: Vec<i64> = match part / 16 {
13158                0 => (0..2048).map(|row| (part * 2048 + row) / 300).collect(),
13159                1 => (0..2048).map(|_| noise()).collect(),
13160                2 => (0..2048).map(|row| if row % 97 == 0 { row } else { 42 }).collect(),
13161                _ => (0..2048).map(|row| 5 + (part * 2048 + row) * 1_000_000).collect(),
13162            };
13163            let settled = settling.encode(&values).unwrap();
13164            assert_eq!(integer::decode(&settled).unwrap(), values, "part {part}");
13165            let searched = integer::encode_with(&values, &Fixed).unwrap();
13166            assert!(
13167                settled.len() * 4 <= searched.len() * 5,
13168                "part {part}: {} settled against {} searched, {} against {}",
13169                settled.len(),
13170                searched.len(),
13171                integer::describe(&settled).unwrap(),
13172                integer::describe(&searched).unwrap(),
13173            );
13174        }
13175    }
13176
13177    #[test]
13178    fn checksum_matches_fixed_vectors() {
13179        assert_eq!(checksum(b""), 0xef46_db37_51d8_e999);
13180        assert_eq!(checksum(b"a"), 0xd24e_c4f1_a98c_6e5b);
13181        assert_eq!(checksum(b"abc"), 0x44bc_2cf5_ad77_0999);
13182    }
13183
13184    #[test]
13185    fn sorting_across_threads_matches_sorting_on_one() {
13186        let mut state = 0x9e37_79b9_7f4a_7c15_u64;
13187        let mut next = move || {
13188            state ^= state << 13;
13189            state ^= state >> 7;
13190            state ^= state << 17;
13191            state
13192        };
13193        let mut values = Vec::new();
13194        for at in 0..150_000_u64 {
13195            let value = match next() % 6 {
13196                0 => Vec::new(),
13197                1 => format!("https://example.com/{}", next() % 5_000).into_bytes(),
13198                2 => format!("https://example.com/path/{at}").into_bytes(),
13199                3 => b"same".to_vec(),
13200                4 => vec![0xff; (next() % 12) as usize],
13201                _ => (0..next() % 20).map(|_| (next() % 3) as u8).collect(),
13202            };
13203            values.push(value);
13204        }
13205        let value = |code: u32| values[code as usize].as_slice();
13206        for workers in [1, 2, 3, 8, 32] {
13207            let mut one = (0..values.len() as u32).rev().collect::<Vec<_>>();
13208            let mut across = one.clone();
13209            sort_by_value(&mut one, value);
13210            sort_by_value_across(&mut across, value, workers);
13211            assert_eq!(one, across, "{workers} workers");
13212        }
13213        let mut sorted = (0..values.len() as u32).collect::<Vec<_>>();
13214        sort_by_value_across(&mut sorted, value, 8);
13215        assert!(sorted.windows(2).all(|pair| value(pair[0]) <= value(pair[1])));
13216    }
13217
13218    fn path(label: &str) -> PathBuf {
13219        let stamp = SystemTime::now().duration_since(UNIX_EPOCH).expect("time advances").as_nanos();
13220        std::env::temp_dir().join(format!("rudb-native-{label}-{}-{stamp}.rdb", std::process::id()))
13221    }
13222
13223    /// Every value of a dictionary in code order, which the tests have no other way to ask for now
13224    /// that a dictionary does not keep the bytes of the values it has seen.
13225    ///
13226    /// Only valid once `finish_blocks` has run, because until then the last part block is still raw.
13227    fn dictionary_values(dictionary: &GlobalDictionary) -> Vec<Vec<u8>> {
13228        let (flat, bases) = dictionary.decoded(None).expect("the blocks decode");
13229        (0..dictionary.values())
13230            .map(|code| {
13231                let (from, to) = GlobalDictionary::value_span(&dictionary.ends, &bases, code);
13232                flat[from..to].to_vec()
13233            })
13234            .collect()
13235    }
13236
13237    /// The sections a test put in the table, which is every one the writer did not.
13238    ///
13239    /// A table now carries a summary and a sketch per column out of the write itself, and a test
13240    /// about the section table is not about those. Filtering by kind rather than by count, so a
13241    /// table that turns out to have no room for its summaries does not quietly change what these
13242    /// tests are asserting over.
13243    fn attached(table: &Table) -> Vec<&Section> {
13244        table.sections().iter().filter(|held| !held.among(section::STATISTICS_KINDS)).collect()
13245    }
13246
13247    /// A read names the offset it wants, so a cursor somebody else moved cannot reach it.
13248    #[test]
13249    fn a_read_at_an_offset_ignores_where_another_thread_left_the_cursor() {
13250        const SPANS: usize = 64;
13251        const SPAN: usize = 512;
13252        let path = path("positional");
13253        let content: Vec<u8> =
13254            (0..SPANS).flat_map(|span| std::iter::repeat_n(span as u8, SPAN)).collect();
13255        fs::write(&path, &content).expect("the file is written");
13256        let file = Arc::new(File::open(&path).expect("the file opens"));
13257        std::thread::scope(|scope| {
13258            for _ in 0..8 {
13259                let file = Arc::clone(&file);
13260                scope.spawn(move || {
13261                    for _ in 0..64 {
13262                        for span in 0..SPANS {
13263                            let mut bytes = [0_u8; SPAN];
13264                            read_at(&file, (span * SPAN) as u64, &mut bytes)
13265                                .expect("the span reads");
13266                            assert!(
13267                                bytes.iter().all(|byte| *byte == span as u8),
13268                                "span {span} came back as {}",
13269                                bytes[0],
13270                            );
13271                        }
13272                    }
13273                });
13274            }
13275        });
13276        let mut past = [0_u8; SPAN];
13277        let end = (SPANS * SPAN) as u64;
13278        let error = read_at(&file, end, &mut past).expect_err("a read past the end is refused");
13279        assert!(error.message().contains("ends before its declared length"), "{error}");
13280        drop(file);
13281        let _ = fs::remove_file(&path);
13282    }
13283
13284    /// The writer records where it put a page and puts it there.
13285    ///
13286    /// This used to move the file's cursor between the steps that record an offset, which is what
13287    /// reading the pages back to build the frequencies did on a platform with no `pread`, and the
13288    /// directory landed on top of a page. The writer's file is an `rudb_io` file now and has no
13289    /// cursor to move, so what is left is the check that every page is where the directory says.
13290    #[test]
13291    fn a_writer_puts_a_page_where_it_said_it_did_wherever_the_cursor_has_got_to() {
13292        let path = path("cursor");
13293        let mut writer = Writer::create(
13294            &path,
13295            "items",
13296            vec![
13297                Field::required("id", LogicalType::Integer),
13298                Field::new("text", LogicalType::Varchar),
13299            ],
13300        )
13301        .expect("new file");
13302        writer.append(&sample()).expect("first part");
13303        writer.append(&sample()).expect("second part");
13304        writer.finish().expect("commit");
13305        let reader = Reader::open(&path).expect("reopen from disk");
13306        assert_eq!(reader.table().rows(), 6);
13307        let ids = reader.read(0, &[0]).expect("the integer page reads back");
13308        assert_eq!(ids.value_at(0, 0), Value::Integer(4));
13309        assert_eq!(ids.value_at(2, 0), Value::Integer(-2));
13310        let text = reader.read(1, &[1]).expect("the text page reads back");
13311        assert_eq!(text.value_at(1, 0), Value::Null);
13312        assert_eq!(text.value_at(2, 0), Value::Varchar("long text after a slash".into()));
13313        // Nothing the directory points at may run past the end of the file, which is the shape the
13314        // failure took: a page recorded at an offset the directory had already been written over.
13315        let end = reader.table().stripes().iter().flat_map(|stripe| {
13316            stripe
13317                .pages
13318                .iter()
13319                .map(|page| page.offset + u64::from(page.length))
13320                .chain(std::iter::once(stripe.index.offset + u64::from(stripe.index.length)))
13321        });
13322        let last = end.fold(HEADER, u64::max);
13323        let directory = fs::metadata(&path).expect("the file is there").len();
13324        assert!(last <= directory, "a page runs to {last} in a file of {directory} bytes");
13325        fs::remove_file(path).expect("remove scratch file");
13326    }
13327
13328    /// How long a global dictionary index is, read out of the page's own header.
13329    ///
13330    /// The tests below damage a byte of the order or of the payload, so they need to know where each
13331    /// one starts, and working it out here rather than writing a number down means adding something
13332    /// to the index does not quietly turn one of them into a test that damages the index instead.
13333    fn dictionary_index_len(header: &[u8; DICTIONARY_HEADER]) -> u64 {
13334        let count = u64::from(u32::from_le_bytes(header[0..4].try_into().expect("four bytes")));
13335        let blocks = u64::from(u32::from_le_bytes(header[8..12].try_into().expect("four bytes")));
13336        let width = u32::from_le_bytes(header[12..16].try_into().expect("four bytes"));
13337        let bits = (width & !DICTIONARY_FLAGS) as usize;
13338        let payload_words = if width & DICTIONARY_SCATTERED == 0 { 2 } else { 3 };
13339        let rank_blocks = count.div_ceil(TEXT_RANK_BLOCK as u64);
13340        DICTIONARY_HEADER as u64
13341            + offset_bytes(count as usize, bits) as u64
13342            + blocks * payload_words * 8
13343            + rank_blocks * 16
13344            + if width & DICTIONARY_GRAMS == 0 { 0 } else { 8 }
13345    }
13346
13347    fn sample() -> Chunk {
13348        Chunk::new(vec![
13349            Vector::from_values(
13350                LogicalType::Integer,
13351                &[Value::Integer(4), Value::Integer(9), Value::Integer(-2)],
13352            )
13353            .expect("integers"),
13354            Vector::from_values(
13355                LogicalType::Varchar,
13356                &[
13357                    Value::Varchar("alpha".into()),
13358                    Value::Null,
13359                    Value::Varchar("long text after a slash".into()),
13360                ],
13361            )
13362            .expect("strings"),
13363        ])
13364        .expect("matching rows")
13365    }
13366
13367    fn sample_ids() -> Chunk {
13368        Chunk::new(vec![
13369            Vector::flat(LogicalType::Integer, Data::Int32(vec![7, 8, 9].into()))
13370                .expect("integers"),
13371        ])
13372        .expect("one column")
13373    }
13374
13375    #[test]
13376    fn the_planner_gets_the_null_count_off_the_same_directory_the_bounds_are_in() {
13377        // Six rows, two of them null. `IS NULL` used to get the same fifth any unreadable
13378        // condition gets, and the number was in the stripe entry next to the bounds all along.
13379        let path = path("nulls_for_the_planner");
13380        let mut writer =
13381            Writer::create(&path, "items", vec![Field::new("a", LogicalType::Integer)])
13382                .expect("new file");
13383        let rows = Chunk::new(vec![
13384            Vector::from_values(
13385                LogicalType::Integer,
13386                &[
13387                    Value::Integer(4),
13388                    Value::Null,
13389                    Value::Integer(9),
13390                    Value::Null,
13391                    Value::Integer(1),
13392                    Value::Integer(2),
13393                ],
13394            )
13395            .expect("integers"),
13396        ])
13397        .expect("one column");
13398        writer.append(&rows).expect("the only part");
13399        writer.finish().expect("commit");
13400        let reader = Reader::open(&path).expect("reopen from disk");
13401        let stripes = Stripes::new(reader);
13402        let column = stripes.column("a").expect("the file has that column");
13403        assert_eq!(stripes.nulls(column), Stat::exact(2, Provenance::NullCount));
13404        // A column the file does not have. Zero here would be a fact about a column that is not
13405        // there, which the planner would then divide by.
13406        assert_eq!(stripes.nulls(column + 1), Stat::Unknown);
13407        fs::remove_file(&path).expect("clean up");
13408    }
13409
13410    #[test]
13411    fn the_planner_gets_a_leading_count_without_a_complete_numeric_synopsis() {
13412        // Six rows hold three values. The two leading counts help equality planning, while the
13413        // omitted value keeps the directory from being a complete grouped-count result.
13414        let path = path("frequencies_for_the_planner");
13415        let mut writer =
13416            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
13417                .expect("new file");
13418        let rows = Chunk::new(vec![
13419            Vector::from_values(
13420                LogicalType::Integer,
13421                &[
13422                    Value::Integer(4),
13423                    Value::Integer(4),
13424                    Value::Integer(4),
13425                    Value::Integer(9),
13426                    Value::Integer(9),
13427                    Value::Integer(1),
13428                ],
13429            )
13430            .expect("integers"),
13431        ])
13432        .expect("one column");
13433        writer.append(&rows).expect("the only part");
13434        writer.finish().expect("commit");
13435        let reader = Reader::open(&path).expect("reopen from disk");
13436        let common = Common::new(reader);
13437        assert_eq!(common.rows(), 6);
13438        let column = common.column("id").expect("the file has that column");
13439        assert_eq!(common.column("nothing"), None);
13440        assert_eq!(
13441            common.rows_with(column, &Bound::Int(4)),
13442            Stat::exact(3, Provenance::FrequencySynopsis)
13443        );
13444        // An absent value cannot be distinguished from the omitted one by the synopsis.
13445        assert_eq!(common.rows_with(column, &Bound::Int(7)), Stat::Unknown);
13446        // A constant of another domain against an integer column. Nothing in the list compares
13447        // with it, so the zero above would be an artefact of the mismatch rather than a fact.
13448        assert_eq!(common.rows_with(column, &Bound::Bytes(b"four".to_vec())), Stat::Unknown);
13449        assert!(common.remainder(column).is_some());
13450        fs::remove_file(&path).expect("clean up");
13451    }
13452
13453    #[test]
13454    fn string_frequency_estimates_do_not_open_the_global_dictionary() {
13455        let path = path("string_frequencies_for_the_planner");
13456        let mut writer =
13457            Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
13458                .expect("new file");
13459        let rows = Chunk::new(vec![
13460            Vector::from_values(
13461                LogicalType::Varchar,
13462                &[
13463                    Value::Varchar(String::new()),
13464                    Value::Varchar("alpha".into()),
13465                    Value::Varchar(String::new()),
13466                    Value::Varchar("beta".into()),
13467                    Value::Varchar(String::new()),
13468                ],
13469            )
13470            .expect("strings"),
13471        ])
13472        .expect("one column");
13473        writer.append(&rows).expect("the only part");
13474        writer.finish().expect("commit");
13475
13476        let reader = Reader::open(&path).expect("reopen from disk");
13477        assert_eq!(reader.reads().dictionaries, 0, "open reads only the directory");
13478        let common = Common::new(reader.clone());
13479        let column = common.column("text").expect("the file has that column");
13480        assert_eq!(
13481            common.rows_with(column, &Bound::Bytes(Vec::new())),
13482            Stat::exact(3, Provenance::FrequencySynopsis)
13483        );
13484        assert_eq!(
13485            common.rows_with(column, &Bound::Bytes(b"missing".to_vec())),
13486            Stat::exact(0, Provenance::FrequencySynopsis)
13487        );
13488        assert_eq!(
13489            reader.reads().dictionaries,
13490            0,
13491            "the bounded spellings answer without opening the dictionary index"
13492        );
13493        fs::remove_file(&path).expect("clean up");
13494    }
13495
13496    #[test]
13497    fn host_groups_certify_omitted_hosts_and_keep_exact_aggregates() {
13498        let path = path("certified_host_groups");
13499        let mut writer =
13500            Writer::create(&path, "hits", vec![Field::required("Referer", LogicalType::Varchar)])
13501                .expect("new file");
13502        let mut values = vec![Value::Varchar("http://www.example.com/a".into()); 150];
13503        values.extend(vec![Value::Varchar("https://example.com/b".into()); 70]);
13504        values.extend((0..550).map(|at| Value::Varchar(format!("https://site{at}.test/x"))));
13505        values.push(Value::Varchar(String::new()));
13506        for part in values.chunks(512) {
13507            writer
13508                .append(
13509                    &Chunk::new(vec![
13510                        Vector::from_values(LogicalType::Varchar, part).expect("strings"),
13511                    ])
13512                    .expect("one column"),
13513                )
13514                .expect("part written");
13515        }
13516        writer.finish().expect("commit");
13517        let reader = Reader::open(&path).expect("reopen");
13518        assert!(reader.table.host_groups.is_none(), "no query-specific host result is stored");
13519        fs::remove_file(&path).expect("clean up");
13520    }
13521
13522    /// A table directory with nothing in it but a name and one column, for the section tests.
13523    ///
13524    /// The section table is orthogonal to everything else in a directory, so the tests that pin it
13525    /// say so by starting from the emptiest table that encodes.
13526    fn bare_table(sections: Vec<Section>) -> Table {
13527        Table {
13528            name: "linked".to_owned(),
13529            fields: vec![Field::required("id", LogicalType::Integer)],
13530            stripes: Vec::new(),
13531            rows: 0,
13532            dictionaries: vec![None],
13533            dictionary_payloads: Vec::new(),
13534            demoted: Vec::new(),
13535            distincts: vec![None],
13536            frequencies: vec![None],
13537            pair_frequencies: Vec::new(),
13538            frequency_texts: Vec::new(),
13539            host_groups: None,
13540            clustering: None,
13541            constraints: Constraints::default(),
13542            generation: 1,
13543            sections,
13544        }
13545    }
13546
13547    fn a_key_map_section() -> Section {
13548        Section {
13549            kind: *section::KEY_MAP,
13550            id: 1,
13551            generation: 3,
13552            extents: 1,
13553            extent_page: HEADER,
13554            extent_bytes: section::EXTENT_BYTES as u32,
13555            hash: 0x1234_5678_9abc_def0,
13556            flags: 0,
13557            header_bytes: 24,
13558        }
13559    }
13560
13561    #[test]
13562    fn a_section_table_round_trips_through_a_directory() {
13563        let mut later = a_key_map_section();
13564        later.kind = *b"RUDBZZ9\0";
13565        later.id = 2;
13566        let table = bare_table(vec![a_key_map_section(), later]);
13567        let directory = encode_directory(&table).expect("directory");
13568        let decoded = decode_directory(&directory, 1 << 20).expect("reopen");
13569        assert_eq!(decoded.sections(), &[a_key_map_section(), later]);
13570        // The second is a kind this build has no name for, and it survived the round trip anyway.
13571        // That is what keeps an old build from silently discarding a newer build's work when it
13572        // rewrites a directory.
13573        assert!(decoded.sections()[0].known());
13574        assert!(!decoded.sections()[1].known());
13575    }
13576
13577    #[test]
13578    fn a_directory_written_before_the_section_table_reads_as_a_table_with_none() {
13579        // The G1 exit criterion, at the directory level. A format 22 directory is exactly this
13580        // build's directory with the trailing section block cut off, so cutting it off is the
13581        // honest way to make one: no fixture to go stale, and no separate encoder to drift.
13582        let directory = encode_directory(&bare_table(Vec::new())).expect("directory");
13583        let block = SECTIONS.len() + size_of::<u64>() + size_of::<u16>();
13584        let older = &directory[..directory.len() - block];
13585        let decoded = decode_directory(older, 1 << 20).expect("a directory from before sections");
13586        assert!(decoded.sections().is_empty());
13587        assert_eq!(decoded.generation(), 0, "a format 22 table recorded no generation");
13588        assert_eq!(decoded.name(), "linked");
13589        assert_eq!(decoded.fields().len(), 1, "everything before the block still decodes");
13590    }
13591
13592    #[test]
13593    fn a_file_stamped_with_the_previous_format_still_opens_and_reads() {
13594        // The same criterion end to end, which is the one the milestone actually asks for: a build
13595        // that knows about sections opens a file written by a build that did not, with no rewrite
13596        // and no repair, and answers from it. The version field is patched rather than a file
13597        // committed by an old binary because the bytes either side of it are identical: format 22
13598        // and format 23 differ only in a trailing directory block, and a reader that stops before
13599        // that block gets a table with no sections.
13600        let path = path("format_twenty_two");
13601        let mut writer =
13602            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
13603                .expect("new file");
13604        let rows = Chunk::new(vec![
13605            Vector::from_values(
13606                LogicalType::Integer,
13607                &[Value::Integer(1), Value::Integer(2), Value::Integer(3)],
13608            )
13609            .expect("integers"),
13610        ])
13611        .expect("one column");
13612        writer.append(&rows).expect("the only part");
13613        writer.finish().expect("commit");
13614
13615        let file = OpenOptions::new().write(true).open(&path).expect("reopen to patch");
13616        write_at(&file, 8, &22_u32.to_le_bytes()).expect("stamp the older format");
13617        drop(file);
13618
13619        let reader = Reader::open(&path).expect("a format 22 file opens unchanged");
13620        assert_eq!(reader.table().rows(), 3);
13621        // The rows and not the section table, because the section block is found by the magic at
13622        // the end of the directory rather than by the number in the header, so stamping the header
13623        // back does not take away the summaries this writer put there. What the test is about is
13624        // that the version check accepts 22, and the rows coming back is what says it did.
13625        assert_eq!(reader.read(0, &[0]).expect("the part still reads").len(), 3);
13626
13627        // And a format this build has never written is still refused, so the accept set is a list
13628        // and not an absence of a check.
13629        let file = OpenOptions::new().write(true).open(&path).expect("reopen to patch");
13630        write_at(&file, 8, &21_u32.to_le_bytes()).expect("stamp an unreadable format");
13631        drop(file);
13632        let error = Reader::open(&path).expect_err("format 21 is not readable");
13633        assert!(error.to_string().contains("format 21"), "{error}");
13634
13635        fs::remove_file(&path).expect("clean up");
13636    }
13637
13638    #[test]
13639    fn a_section_whose_extent_table_is_outside_the_file_is_refused() {
13640        // The bound the format has to check and `section` cannot, because only the reader knows how
13641        // big the file is. Reading the payload a section like this names would be reading whatever
13642        // else happens to be at that offset, which is the one way a graph section could turn into a
13643        // wrong answer rather than a slow one.
13644        let mut past = a_key_map_section();
13645        past.extent_page = 1 << 30;
13646        let directory = encode_directory(&bare_table(vec![past])).expect("directory");
13647        let error = decode_directory(&directory, 1 << 20).expect_err("refused");
13648        assert!(error.to_string().contains("outside the file"), "{error}");
13649
13650        let mut inside_the_header = a_key_map_section();
13651        inside_the_header.extent_page = 8;
13652        let directory = encode_directory(&bare_table(vec![inside_the_header])).expect("directory");
13653        assert!(
13654            decode_directory(&directory, 1 << 20).is_err(),
13655            "a section may not overlap a header"
13656        );
13657    }
13658
13659    #[test]
13660    fn a_section_recorded_as_not_built_is_legal_and_names_no_bytes() {
13661        // Section 3.7: a relationship that does not fit the budget is recorded with its size so
13662        // that `rudb_links()` can report what a larger budget would buy. That record is a section
13663        // entry with no extents, so it has to survive a round trip while naming nothing.
13664        let not_built = Section {
13665            kind: *section::FORWARD_LINK,
13666            id: 9,
13667            generation: 3,
13668            extents: 0,
13669            extent_page: 0,
13670            extent_bytes: 0,
13671            hash: 0,
13672            flags: 0,
13673            header_bytes: 0,
13674        };
13675        let directory = encode_directory(&bare_table(vec![not_built])).expect("directory");
13676        let decoded = decode_directory(&directory, 1 << 20).expect("reopen");
13677        assert_eq!(decoded.sections(), &[not_built]);
13678
13679        // But a section with no extents that still names an extent table is incoherent, and an
13680        // incoherent entry is a torn directory rather than a relationship that was skipped.
13681        let mut incoherent = not_built;
13682        incoherent.extent_bytes = 28;
13683        incoherent.extent_page = HEADER;
13684        let directory = encode_directory(&bare_table(vec![incoherent])).expect("directory");
13685        assert!(decode_directory(&directory, 1 << 20).is_err());
13686    }
13687
13688    #[test]
13689    fn a_directory_naming_more_sections_than_the_bound_is_refused() {
13690        let directory = encode_directory(&bare_table(Vec::new())).expect("directory");
13691        let mut torn = directory.clone();
13692        let count_at = torn.len() - size_of::<u16>();
13693        torn[count_at..].copy_from_slice(&u16::MAX.to_le_bytes());
13694        // Not an allocation of sixty five thousand entries off a torn count: either the bound
13695        // refuses it or the bytes run out, and both are errors rather than a read past the end.
13696        assert!(decode_directory(&torn, 1 << 20).is_err());
13697    }
13698
13699    /// A committed one column file of `rows` integers, for the attach tests.
13700    fn linked_file(label: &str, rows: i32) -> PathBuf {
13701        let path = path(label);
13702        let mut writer =
13703            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
13704                .expect("new file");
13705        let values = (0..rows).map(Value::Integer).collect::<Vec<_>>();
13706        let chunk =
13707            Chunk::new(vec![Vector::from_values(LogicalType::Integer, &values).expect("integers")])
13708                .expect("one column");
13709        writer.append(&chunk).expect("the only part");
13710        writer.finish().expect("commit");
13711        path
13712    }
13713
13714    fn a_key_map_payload() -> Vec<u8> {
13715        // Shaped like one without being one: this crate never reads a payload, so what matters here
13716        // is that every byte comes back and that the header the entry measures is at the front.
13717        (0..512_u32).flat_map(u32::to_le_bytes).collect()
13718    }
13719
13720    #[test]
13721    fn a_section_attached_to_a_committed_file_reads_back_byte_for_byte() {
13722        let path = linked_file("attach", 64);
13723        let payload = a_key_map_payload();
13724        let table = attach(
13725            &path,
13726            "items",
13727            &[section::Attachment {
13728                kind: *section::KEY_MAP,
13729                id: 0,
13730                flags: 2,
13731                header_bytes: 40,
13732                bytes: &payload,
13733            }],
13734        )
13735        .expect("attach a key map");
13736        assert_eq!(attached(&table).len(), 1);
13737
13738        let reader = Reader::open(&path).expect("reopen after the attach");
13739        let held = attached(reader.table());
13740        assert_eq!(held.len(), 1);
13741        assert_eq!(held[0].kind, *section::KEY_MAP);
13742        assert_eq!(held[0].flags, 2, "the form a reader must not have to guess");
13743        assert_eq!(held[0].header_bytes, 40);
13744        // The generation is the one the pages were written at, not the one the attach committed at.
13745        // Attaching a section moved no row, so a section written by it is current, and a second
13746        // table added to this file later would not make it stale.
13747        assert_eq!(held[0].generation, 1);
13748        assert!(held[0].usable(reader.table().generation()));
13749        assert_eq!(reader.payload(held[0]).expect("read the payload"), payload);
13750        assert_eq!(reader.extents(held[0]).expect("extent table").len(), 1);
13751
13752        fs::remove_file(&path).expect("clean up");
13753    }
13754
13755    #[test]
13756    fn attaching_a_section_answers_every_row_exactly_as_before() {
13757        // Section 3.1 end to end, and the reason the whole layer is safe to build incrementally. A
13758        // file with a section in it and the same file without one have to agree row for row, so the
13759        // comparison is made against the answers taken before the attach rather than against a
13760        // constant somebody typed.
13761        let path = linked_file("attach_changes_nothing", 300);
13762        let before = Reader::open(&path).expect("open before");
13763        let rows = before.table().rows();
13764        let first = before.read(0, &[0]).expect("read before");
13765        let values = (0..rows).map(|at| first.value_at(at, 0)).collect::<Vec<_>>();
13766        let layout = before.layout().columns_total();
13767        drop(before);
13768
13769        let payload = a_key_map_payload();
13770        attach(
13771            &path,
13772            "items",
13773            &[section::Attachment {
13774                kind: *section::KEY_MAP,
13775                id: 0,
13776                flags: 0,
13777                header_bytes: 0,
13778                bytes: &payload,
13779            }],
13780        )
13781        .expect("attach");
13782
13783        let after = Reader::open(&path).expect("open after");
13784        assert_eq!(after.table().rows(), rows);
13785        let read = after.read(0, &[0]).expect("read after");
13786        for (at, value) in values.iter().enumerate() {
13787            assert_eq!(&read.value_at(at, 0), value, "row {at} moved");
13788        }
13789        assert_eq!(
13790            after.layout().columns_total(),
13791            layout,
13792            "an attach appends and does not rewrite a column page"
13793        );
13794
13795        fs::remove_file(&path).expect("clean up");
13796    }
13797
13798    #[test]
13799    fn a_rebuilt_section_replaces_the_one_it_supersedes() {
13800        // Rebuilding a key map has to be a write and not a question. If an attach added rather than
13801        // replaced, a table rebuilt a few times would name several maps for one column and a reader
13802        // would have to pick, which is a decision with no right answer in it.
13803        let path = linked_file("attach_twice", 32);
13804        let one = a_key_map_payload();
13805        let two = vec![7_u8; 1024];
13806        let entry = |bytes| section::Attachment {
13807            kind: *section::KEY_MAP,
13808            id: 4,
13809            flags: 1,
13810            header_bytes: 0,
13811            bytes,
13812        };
13813        attach(&path, "items", &[entry(&one)]).expect("first build");
13814        attach(&path, "items", &[entry(&two)]).expect("rebuild");
13815
13816        let reader = Reader::open(&path).expect("reopen");
13817        let held = attached(reader.table());
13818        assert_eq!(held.len(), 1, "one map per column and not one per build");
13819        assert_eq!(reader.payload(held[0]).expect("payload"), two);
13820
13821        fs::remove_file(&path).expect("clean up");
13822    }
13823
13824    #[test]
13825    fn an_attach_carries_through_a_kind_it_does_not_know() {
13826        // The first of section 3.2's three rules, at the point where it is easiest to break: a build
13827        // that rewrites a directory has to carry an entry it has no name for, or opening a file with
13828        // an older binary and attaching one section quietly deletes the work of a newer one.
13829        let path = linked_file("attach_unknown", 16);
13830        let payload = vec![3_u8; 96];
13831        attach(
13832            &path,
13833            "items",
13834            &[section::Attachment {
13835                kind: *b"RUDBZZ9\0",
13836                id: 1,
13837                flags: 0,
13838                header_bytes: 0,
13839                bytes: &payload,
13840            }],
13841        )
13842        .expect("a kind this build does not know still writes");
13843        let key_map = a_key_map_payload();
13844        attach(
13845            &path,
13846            "items",
13847            &[section::Attachment {
13848                kind: *section::KEY_MAP,
13849                id: 0,
13850                flags: 0,
13851                header_bytes: 0,
13852                bytes: &key_map,
13853            }],
13854        )
13855        .expect("attach beside it");
13856
13857        let reader = Reader::open(&path).expect("reopen");
13858        let held = attached(reader.table());
13859        assert_eq!(held.len(), 2, "the unfamiliar entry survived a directory rewrite");
13860        let unknown = held.iter().find(|one| !one.known()).expect("the unfamiliar one");
13861        assert_eq!(reader.payload(unknown).expect("its bytes are still there"), payload);
13862
13863        fs::remove_file(&path).expect("clean up");
13864    }
13865
13866    #[test]
13867    fn a_payload_of_nothing_is_a_relationship_recorded_as_not_built() {
13868        let path = linked_file("attach_not_built", 8);
13869        attach(
13870            &path,
13871            "items",
13872            &[section::Attachment {
13873                kind: *section::FORWARD_LINK,
13874                id: 2,
13875                flags: 0,
13876                header_bytes: 0,
13877                bytes: &[],
13878            }],
13879        )
13880        .expect("record a link that did not fit the budget");
13881
13882        let reader = Reader::open(&path).expect("reopen");
13883        let held = attached(reader.table());
13884        assert_eq!(held.len(), 1);
13885        assert_eq!(held[0].extents, 0);
13886        assert_eq!(held[0].extent_page, 0, "an entry that names no bytes points at none");
13887        assert!(reader.extents(held[0]).expect("no extent table").is_empty());
13888        assert!(reader.payload(held[0]).expect("no payload").is_empty());
13889
13890        fs::remove_file(&path).expect("clean up");
13891    }
13892
13893    #[test]
13894    fn a_payload_past_one_extent_is_split_and_joined_back() {
13895        // Issue #745's rule, exercised rather than argued. One byte past the bound is the smallest
13896        // payload that has to be two extents, and it is the case a split written for the common
13897        // size gets wrong.
13898        let path = linked_file("attach_two_extents", 8);
13899        let payload = vec![0x5a_u8; section::MAX_EXTENT as usize + 1];
13900        attach(
13901            &path,
13902            "items",
13903            &[section::Attachment {
13904                kind: *section::KEY_MAP,
13905                id: 0,
13906                flags: 0,
13907                header_bytes: 0,
13908                bytes: &payload,
13909            }],
13910        )
13911        .expect("attach a payload past the bound");
13912
13913        let reader = Reader::open(&path).expect("reopen");
13914        let held = attached(reader.table());
13915        let extents = reader.extents(held[0]).expect("extent table");
13916        assert_eq!(extents.len(), 2, "one byte past the bound is two extents");
13917        assert_eq!(extents[0].length, section::MAX_EXTENT);
13918        assert_eq!(extents[1].length, 1);
13919        assert_eq!(extents[1].first, u64::from(section::MAX_EXTENT));
13920        // And the extent the caller wants is readable on its own, which is the point of the split.
13921        assert_eq!(reader.extent(&extents[1]).expect("the last extent"), vec![0x5a]);
13922        assert_eq!(reader.payload(held[0]).expect("the whole payload").len(), payload.len());
13923
13924        fs::remove_file(&path).expect("clean up");
13925    }
13926
13927    #[test]
13928    fn a_torn_extent_is_refused_rather_than_decoded() {
13929        let path = linked_file("attach_torn", 8);
13930        let payload = a_key_map_payload();
13931        attach(
13932            &path,
13933            "items",
13934            &[section::Attachment {
13935                kind: *section::KEY_MAP,
13936                id: 0,
13937                flags: 0,
13938                header_bytes: 0,
13939                bytes: &payload,
13940            }],
13941        )
13942        .expect("attach");
13943
13944        let reader = Reader::open(&path).expect("reopen");
13945        let extent = reader.extents(&reader.table().sections()[0]).expect("extent table")[0];
13946        let file = OpenOptions::new().write(true).open(&path).expect("reopen to corrupt");
13947        write_at(&file, extent.offset + 7, &[0xff]).expect("flip a byte of the payload");
13948        drop(file);
13949
13950        let reader = Reader::open(&path).expect("the table still opens");
13951        let error = reader
13952            .payload(&reader.table().sections()[0])
13953            .expect_err("a corrupt payload is not handed out");
13954        assert!(error.to_string().contains("checksum"), "{error}");
13955        // And the table is still readable, which is section 3.1: a section that cannot be trusted
13956        // costs the query its shortcut and nothing else.
13957        assert_eq!(reader.read(0, &[0]).expect("the column is untouched").width(), 1);
13958
13959        fs::remove_file(&path).expect("clean up");
13960    }
13961
13962    #[test]
13963    fn attaching_to_a_file_of_the_previous_format_is_refused_rather_than_done() {
13964        // Readable is not writable. A format 22 directory has no section block, and adding one
13965        // without moving the number in the header would leave a file claiming a format it is not.
13966        let path = linked_file("attach_old_format", 8);
13967        let file = OpenOptions::new().write(true).open(&path).expect("reopen to patch");
13968        write_at(&file, 8, &22_u32.to_le_bytes()).expect("stamp the older format");
13969        drop(file);
13970
13971        let payload = a_key_map_payload();
13972        let error = attach(
13973            &path,
13974            "items",
13975            &[section::Attachment {
13976                kind: *section::KEY_MAP,
13977                id: 0,
13978                flags: 0,
13979                header_bytes: 0,
13980                bytes: &payload,
13981            }],
13982        )
13983        .expect_err("format 22 cannot gain a section");
13984        assert!(error.to_string().contains("format 22"), "{error}");
13985        assert!(Reader::open(&path).expect("and the file is untouched").table().rows() == 8);
13986
13987        fs::remove_file(&path).expect("clean up");
13988    }
13989
13990    #[test]
13991    fn a_section_header_longer_than_its_payload_is_refused_at_the_write() {
13992        let path = linked_file("attach_bad_header", 8);
13993        let error = attach(
13994            &path,
13995            "items",
13996            &[section::Attachment {
13997                kind: *section::KEY_MAP,
13998                id: 0,
13999                flags: 0,
14000                header_bytes: 40,
14001                bytes: &[1, 2, 3],
14002            }],
14003        )
14004        .expect_err("a writer's bug stops at the write");
14005        assert!(error.to_string().contains("header is longer"), "{error}");
14006
14007        fs::remove_file(&path).expect("clean up");
14008    }
14009
14010    #[test]
14011    fn attaching_to_a_name_the_file_does_not_hold_says_so() {
14012        let path = linked_file("attach_wrong_name", 8);
14013        let error = attach(&path, "orders", &[]).expect_err("no such table");
14014        assert!(error.to_string().contains("orders"), "{error}");
14015        fs::remove_file(&path).expect("clean up");
14016    }
14017
14018    #[test]
14019    fn the_planner_gets_an_exact_count_for_a_leading_value_of_an_incomplete_synopsis() {
14020        // The case a complete synopsis does not cover, and the one worth the most. 16,000 rows over
14021        // 601 distinct values, 10,000 of them holding a single value and the rest spread ten apiece
14022        // over six hundred more. The writer holds 512 values, so the list is a prefix and most of
14023        // the tail is outside it. The counts inside it are still exact, because the pass recounts
14024        // the candidates that survived it, so `id = 1` is ten thousand rows rather than the
14025        // twenty six a distinct count of 601 would divide its way to.
14026        let path = path("frequency_prefix_for_the_planner");
14027        let mut writer =
14028            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
14029                .expect("new file");
14030        let mut values = vec![Value::Integer(1); 10_000];
14031        for _ in 0..10 {
14032            values.extend((0..600).map(|tail| Value::Integer(1_000 + tail)));
14033        }
14034        // A vector holds 8,192 rows, so this goes in as several parts. The pass that takes the
14035        // synopsis walks the whole column rather than a part, so the counts are the same either way.
14036        for part in values.chunks(8_000) {
14037            let rows = Chunk::new(vec![
14038                Vector::from_values(LogicalType::Integer, part).expect("integers"),
14039            ])
14040            .expect("one column");
14041            writer.append(&rows).expect("a part");
14042        }
14043        writer.finish().expect("commit");
14044        let reader = Reader::open(&path).expect("reopen from disk");
14045        let prefix =
14046            reader.frequency_prefix(0).expect("a readable synopsis").expect("the column has one");
14047        // A prefix and not the whole column, and the writer said how many rows anything left out of
14048        // it can hold.
14049        assert_eq!(prefix.entries.len(), 512);
14050        assert_eq!(prefix.omitted_max, 10);
14051        let common = Common::new(reader);
14052        assert_eq!(common.rows(), 16_000);
14053        let column = common.column("id").expect("the file has that column");
14054        assert_eq!(
14055            common.rows_with(column, &Bound::Int(1)),
14056            Stat::exact(10_000, Provenance::FrequencySynopsis)
14057        );
14058        // In the prefix, because ties go to the smaller value and the prefix reaches 1,510.
14059        assert_eq!(
14060            common.rows_with(column, &Bound::Int(1_100)),
14061            Stat::exact(10, Provenance::FrequencySynopsis)
14062        );
14063        // Outside it, and a prefix says nothing about a value it does not list. Not zero, which is
14064        // what a complete list would say, and the file holds ten rows of this one.
14065        assert_eq!(common.rows_with(column, &Bound::Int(1_550)), Stat::Unknown);
14066        // Not in the file at all, and still nothing rather than a zero. A prefix cannot tell the
14067        // two apart, which is the whole of what it gives up.
14068        assert_eq!(common.rows_with(column, &Bound::Int(9_999)), Stat::Unknown);
14069        // What the prefix left out, which is what turns the unknown above into a number. The 512
14070        // entries account for 15,110 rows, so 890 are left for the 89 values the writer dropped,
14071        // and 890 over 89 is the ten rows each of them really holds.
14072        let remainder = common.remainder(column).expect("the list is a prefix");
14073        assert_eq!(remainder, Remainder { rows: 890, listed: 512, most: 10 });
14074        assert_eq!(remainder.rows / (601 - remainder.listed), 10);
14075        fs::remove_file(&path).expect("clean up");
14076    }
14077
14078    /// A file with no table in it is a file, and opening it says so rather than failing.
14079    #[test]
14080    fn a_file_holding_no_table_commits_and_opens_and_a_table_can_be_added_to_it() {
14081        let path = path("empty");
14082        Writer::empty(&path, &[]).expect("a file with nothing in it");
14083        let catalog = Catalog::open(&path).expect("the empty file opens");
14084        assert_eq!(catalog.len(), 0);
14085        assert!(catalog.is_empty());
14086        assert_eq!(catalog.names().count(), 0);
14087        // The next generation goes over the top of it the way it goes over any other, which is what
14088        // says this is a committed file and not a special case somebody has to know about.
14089        let mut writer =
14090            Writer::open(&path, "items", vec![Field::required("id", LogicalType::Integer)])
14091                .expect("a table goes into the empty file");
14092        writer.append(&sample_ids()).expect("rows");
14093        writer.finish().expect("commit");
14094        let catalog = Catalog::open(&path).expect("the file opens again");
14095        assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
14096        fs::remove_file(&path).expect("clean up");
14097    }
14098
14099    /// A committed table with no rows is a name the next generation takes over, and one with rows
14100    /// is a name it refuses.
14101    ///
14102    /// The refusal is what it always was and it is load bearing: carrying a table that holds rows
14103    /// forward means reading and rewriting its pages, and a writer that quietly wrote a second
14104    /// entry under the same name would leave a file with two tables a reader cannot tell apart. An
14105    /// empty one has no pages and no reader, so there is nothing to carry and nothing to lose, and
14106    /// taking its place is what lets a schema committed by an earlier session be loaded by a stream
14107    /// instead of through memory.
14108    #[test]
14109    fn a_committed_empty_table_gives_up_its_name_and_one_with_rows_does_not() {
14110        let path = path("empty-name");
14111        let field = || vec![Field::required("id", LogicalType::Integer)];
14112        Writer::create(&path, "items", field()).expect("new file").finish().expect("commit");
14113        let catalog = Catalog::open(&path).expect("the file opens");
14114        assert_eq!(catalog.rows().collect::<Vec<_>>(), vec![("items", 0)]);
14115
14116        let mut writer = Writer::open(&path, "items", field()).expect("the empty name is free");
14117        writer.append(&sample_ids()).expect("rows");
14118        writer.finish().expect("commit");
14119        let catalog = Catalog::open(&path).expect("the file opens again");
14120        // One entry and not two. The generation replaced the empty table rather than joining it.
14121        assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
14122        let held = catalog.rows().collect::<Vec<_>>();
14123        assert_eq!(held.len(), 1);
14124        assert!(held[0].1 > 0, "the rows that were appended are the ones the catalog counts");
14125
14126        // The same call against the same name now that it holds rows, which is still refused.
14127        let error = Writer::open(&path, "items", field()).expect_err("a name with rows is taken");
14128        assert!(error.to_string().contains("same name"), "{error}");
14129        fs::remove_file(&path).expect("clean up");
14130    }
14131
14132    #[test]
14133    fn a_device_card_rides_the_catalog_and_an_older_catalog_has_none() {
14134        let entry = || Entry {
14135            name: "items".to_string(),
14136            fields: vec![Field::required("id", LogicalType::Integer)],
14137            rows: 1,
14138            directory: Page { offset: HEADER, length: 8, hash: 0 },
14139            nonzero: vec![None],
14140            aggregates: vec![None],
14141            distincts: vec![None],
14142            extremes: vec![None],
14143            frequencies: vec![None],
14144        };
14145        let card = KeptCard { device: "dev:42".to_string(), bytes: vec![1, 2, 3] };
14146        let bytes = encode_catalog(&[entry()], &[], Some(&card)).expect("encodes");
14147        let (entries, views, kept) = decode_catalog(&bytes, HEADER + 8).expect("decodes");
14148        assert_eq!((entries.len(), views.len()), (1, 0));
14149        assert_eq!(kept, Some(card));
14150        let bytes = encode_catalog(&[entry()], &[], None).expect("encodes");
14151        assert_eq!(decode_catalog(&bytes, HEADER + 8).expect("decodes").2, None);
14152    }
14153
14154    /// A view, with everything about it that a reopened catalog has to be able to answer from.
14155    fn sample_view(name: &str) -> ViewEntry {
14156        ViewEntry {
14157            name: name.to_string(),
14158            sql: "SELECT id FROM items WHERE id > 0".to_string(),
14159            statement: format!("CREATE VIEW {name} AS SELECT id FROM items WHERE (id > 0);"),
14160            aliases: vec!["n".to_string()],
14161            columns: vec![Field::new("n", LogicalType::Integer)],
14162        }
14163    }
14164
14165    #[test]
14166    fn a_view_written_into_the_catalog_comes_back_whole() {
14167        let path = path("views");
14168        let mut writer =
14169            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
14170                .expect("new file");
14171        writer.append(&sample_ids()).expect("rows");
14172        writer.with_views(vec![sample_view("v")]).finish().expect("commit");
14173        let catalog = Catalog::open(&path).expect("reopen");
14174        assert_eq!(catalog.views().cloned().collect::<Vec<_>>(), vec![sample_view("v")]);
14175        // The tables are still there and are still read the same way, so the section on the end did
14176        // not move anything in front of it.
14177        assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
14178        fs::remove_file(&path).expect("clean up");
14179    }
14180
14181    /// A writer opened to append a table says nothing about views and must not lose them.
14182    #[test]
14183    fn appending_a_table_carries_the_views_forward() {
14184        let path = path("viewscarry");
14185        let mut writer =
14186            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
14187                .expect("new file");
14188        writer.append(&sample_ids()).expect("rows");
14189        writer.with_views(vec![sample_view("v")]).finish().expect("commit");
14190        let mut writer =
14191            Writer::open(&path, "other", vec![Field::required("id", LogicalType::Integer)])
14192                .expect("a second table");
14193        writer.append(&sample_ids()).expect("rows");
14194        writer.finish().expect("commit");
14195        let catalog = Catalog::open(&path).expect("reopen");
14196        assert_eq!(catalog.views().count(), 1);
14197        assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items", "other"]);
14198        fs::remove_file(&path).expect("clean up");
14199    }
14200
14201    /// The whole point of [`Writer::restate`]: the views change and the pages do not move.
14202    #[test]
14203    fn restating_the_views_leaves_every_table_where_it_was() {
14204        let path = path("restate");
14205        let mut writer =
14206            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
14207                .expect("new file");
14208        writer.append(&sample_ids()).expect("rows");
14209        writer.finish().expect("commit");
14210        let before = fs::metadata(&path).expect("the file is there").len();
14211        Writer::restate(&path, &[sample_view("v"), sample_view("w")]).expect("two views");
14212        let catalog = Catalog::open(&path).expect("reopen");
14213        assert_eq!(catalog.views().count(), 2);
14214        assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
14215        // A catalog on the end and nothing else, so what it grew by is the size of a catalog rather
14216        // than the size of the table.
14217        let after = fs::metadata(&path).expect("the file is there").len();
14218        assert!(after > before, "a generation was written");
14219        assert!(after - before < before, "the table was not written again");
14220        // The rows are still readable through the new generation, which is the part that would go
14221        // wrong if the catalog carried the wrong directory pointers forward.
14222        let reader = Catalog::open(&path).expect("reopen").table("items").expect("the table");
14223        assert_eq!(reader.table().rows, 3);
14224        // And a restate over a restate keeps working, because each one reads the slot that
14225        // checksummed rather than the highest number in the header.
14226        Writer::restate(&path, &[]).expect("no views at all");
14227        assert_eq!(Catalog::open(&path).expect("reopen").views().count(), 0);
14228        fs::remove_file(&path).expect("clean up");
14229    }
14230
14231    /// Two entries under one name is a catalog no lookup can answer, whichever two they are.
14232    #[test]
14233    fn a_view_named_after_a_table_is_refused_when_the_catalog_is_read() {
14234        let bytes = encode_catalog(
14235            &[Entry {
14236                name: "items".to_string(),
14237                fields: vec![Field::required("id", LogicalType::Integer)],
14238                rows: 1,
14239                directory: Page { offset: HEADER, length: 8, hash: 0 },
14240                nonzero: vec![None],
14241                aggregates: vec![None],
14242                distincts: vec![None],
14243                extremes: vec![None],
14244                frequencies: vec![None],
14245            }],
14246            &[sample_view("items")],
14247            None,
14248        )
14249        .expect("it encodes, because encoding does not look");
14250        let error = decode_catalog(&bytes, HEADER + 8).expect_err("and decoding does");
14251        assert!(error.to_string().contains("same name"), "{error}");
14252    }
14253
14254    /// A compressed text page read at some rows is those rows of the page read whole, nulls and
14255    /// all, and a row past the end or rows out of order are refused rather than guessed at.
14256    #[test]
14257    fn a_compressed_text_page_read_at_some_rows_is_those_rows_of_the_whole() {
14258        let rows: usize = 300;
14259        let text: Vec<String> =
14260            (0..rows).map(|row| format!("a street named after number {}", row * 7)).collect();
14261        let values: Vec<&[u8]> = text.iter().map(String::as_bytes).collect();
14262        let mut page = vec![6, 2];
14263        page.extend((0..rows.div_ceil(8)).map(|byte| {
14264            (0..8).filter(|bit| (byte * 8 + bit) % 5 != 3).fold(0_u8, |mask, bit| mask | 1 << bit)
14265        }));
14266        let compressed = string::encode_only(string::Kind::Fsst, &values)
14267            .expect("encoded")
14268            .expect("text this repetitive compresses");
14269        page.extend_from_slice(&compressed);
14270        let whole = decode(&LogicalType::Varchar, rows, &page, None).expect("the whole page");
14271        let positions = [0_u32, 3, 8, 13, 200, 299];
14272        let some =
14273            decode_at(&LogicalType::Varchar, rows, &page, None, &positions).expect("some rows");
14274        assert_eq!(some.len(), positions.len());
14275        for (at, &row) in positions.iter().enumerate() {
14276            assert_eq!(some.value_at(at), whole.value_at(row as usize), "row {row}");
14277        }
14278        assert_eq!(some.value_at(1), Value::Null, "row 3 is null");
14279        assert!(decode_at(&LogicalType::Varchar, rows, &page, None, &[300]).is_err());
14280        assert!(decode_at(&LogicalType::Varchar, rows, &page, None, &[8, 3]).is_err());
14281    }
14282
14283    /// Every column of a part read at some rows is the part read whole and gathered, whatever the
14284    /// page holds.
14285    #[test]
14286    fn a_part_read_at_some_rows_is_the_part_read_whole_and_gathered() {
14287        let path = path("rows");
14288        let mut writer = Writer::create(
14289            &path,
14290            "items",
14291            vec![
14292                Field::required("id", LogicalType::Integer),
14293                Field::new("text", LogicalType::Varchar),
14294            ],
14295        )
14296        .expect("new file");
14297        let rows = 2_000;
14298        let chunk = Chunk::new(vec![
14299            Vector::from_values(
14300                LogicalType::Integer,
14301                &(0..rows).map(Value::Integer).collect::<Vec<_>>(),
14302            )
14303            .expect("integers"),
14304            Vector::from_values(
14305                LogicalType::Varchar,
14306                &(0..rows)
14307                    .map(|row| {
14308                        if row % 7 == 2 {
14309                            Value::Null
14310                        } else {
14311                            Value::Varchar(format!("a comment about order {}", row * 13))
14312                        }
14313                    })
14314                    .collect::<Vec<_>>(),
14315            )
14316            .expect("strings"),
14317        ])
14318        .expect("matching rows");
14319        writer.append(&chunk).expect("one part");
14320        writer.finish().expect("commit");
14321        let reader = Reader::open(&path).expect("reopen from disk");
14322        let positions = [1_u32, 2, 9, 1_000, 1_999];
14323        for whole in [true, false] {
14324            let some = reader.read_rows(0, &[0, 1], &positions, whole).expect("some rows");
14325            let all = reader.read(0, &[0, 1]).expect("the whole part");
14326            assert_eq!(some.len(), positions.len());
14327            for column in 0..2 {
14328                for (at, &row) in positions.iter().enumerate() {
14329                    assert_eq!(some.value_at(at, column), all.value_at(row as usize, column));
14330                }
14331            }
14332        }
14333        assert!(reader.read_rows(0, &[1], &[2_000], true).is_err());
14334    }
14335
14336    #[test]
14337    fn committed_file_reopens_and_reads_only_requested_columns() {
14338        let path = path("reopen");
14339        let mut writer = Writer::create(
14340            &path,
14341            "items",
14342            vec![
14343                Field::required("id", LogicalType::Integer),
14344                Field::new("text", LogicalType::Varchar),
14345            ],
14346        )
14347        .expect("new file");
14348        writer.append(&sample()).expect("first part");
14349        writer.append(&sample()).expect("second part");
14350        writer.finish().expect("commit");
14351        let reader = Reader::open(&path).expect("reopen from disk");
14352        assert_eq!(reader.table().rows(), 6);
14353        // Two appends below the stripe bound are two parts of one stripe, which is the whole point
14354        // of the split: the directory describes the stripe and the scan still reads a part.
14355        assert_eq!(reader.table().stripes().len(), 1);
14356        assert_eq!(reader.parts(), 2);
14357        assert_eq!(reader.part_rows(0), 3);
14358        assert_eq!(reader.part_rows(1), 3);
14359        let text = reader.read(1, &[1]).expect("only text page");
14360        assert_eq!(text.width(), 1);
14361        assert_eq!(text.value_at(1, 0), Value::Null);
14362        assert_eq!(text.value_at(2, 0), Value::Varchar("long text after a slash".into()));
14363        let sparse = reader.read_sparse(1, &[1]).expect("one part without its whole page");
14364        assert_eq!(sparse.width(), 1);
14365        assert_eq!(sparse.value_at(1, 0), Value::Null);
14366        assert_eq!(sparse.value_at(2, 0), Value::Varchar("long text after a slash".into()));
14367        assert!(!reader.skips_codes(0, 1, &[0]).expect("alpha is in the stripe"));
14368        assert!(!reader.skips_codes(0, 1, &[2]).expect("long text is in the stripe"));
14369        assert!(reader.skips_codes(0, 1, &[3]).expect("unknown code is absent"));
14370        let count = reader.read(0, &[]).expect("no page is needed for count");
14371        assert_eq!(count.len(), 3);
14372        assert!(reader.skips(0, &[Probe { column: 0, op: Op::Greater, value: Bound::Int(100) }]));
14373        assert!(!reader.skips(0, &[Probe { column: 0, op: Op::Greater, value: Bound::Int(0) }]));
14374        assert_eq!(reader.top_frequencies(0, 1).expect("valid integer synopsis"), None);
14375        let integers = reader.frequency_prefix(0).expect("valid integer synopsis").expect("kept");
14376        assert_eq!(integers.entries, vec![(Value::Integer(-2), 2), (Value::Integer(4), 2)]);
14377        assert_eq!(integers.omitted_max, 2);
14378        let strings = reader.top_frequencies(1, 1).expect("valid string synopsis").expect("kept");
14379        assert_eq!(strings.len(), 3);
14380        assert!(strings.contains(&(Value::Null, 2)));
14381        assert!(strings.contains(&(Value::Varchar("alpha".into()), 2)));
14382        assert!(strings.contains(&(Value::Varchar("long text after a slash".into()), 2)));
14383        fs::remove_file(path).expect("remove scratch file");
14384    }
14385
14386    /// Two pipeline instances handing over whole runs, which is what makes the native sink safe to
14387    /// instance.
14388    ///
14389    /// The runs arrive in the order the instances finished reading them rather than in source
14390    /// order, and the second one to finish is the one that read the earlier rows. Each run is still
14391    /// a stripe of its own and the table still reads back in source order, which is the whole of
14392    /// what the writer promises about ordering.
14393    #[test]
14394    fn runs_handed_over_out_of_order_still_read_back_in_source_order() {
14395        let path = path("interleaved-runs");
14396        let mut writer =
14397            Writer::create(&path, "interleaved", vec![Field::new("v", LogicalType::BigInt)])
14398                .expect("new file");
14399        for morsel in [2_u64, 0, 3, 1] {
14400            let parts = (0..4_u64)
14401                .map(|chunk| {
14402                    let first = i64::try_from(morsel * 32 + chunk * 8).expect("small");
14403                    let values =
14404                        (0..8_i64).map(|row| Value::BigInt(first + row)).collect::<Vec<_>>();
14405                    let column =
14406                        Vector::from_values(LogicalType::BigInt, &values).expect("a column");
14407                    ((morsel, chunk), Chunk::new(vec![column]).expect("one column"))
14408                })
14409                .collect::<Vec<_>>();
14410            writer.append_stripe(parts).expect("a stripe");
14411        }
14412        writer.finish().expect("commit");
14413
14414        let reader = Reader::open(&path).expect("valid directory");
14415        assert_eq!(reader.table().stripes().len(), 4, "a run is a stripe of its own");
14416        assert_eq!(reader.table().rows(), 128);
14417        for part in 0..16_usize {
14418            let read = reader.read(part, &[0]).expect("a part back");
14419            for row in 0..8_usize {
14420                let want = i64::try_from(part * 8 + row).expect("small");
14421                assert_eq!(read.value_at(row, 0), Value::BigInt(want), "part {part} row {row}");
14422            }
14423        }
14424        fs::remove_file(path).expect("remove scratch file");
14425    }
14426
14427    /// Runs from different callers may interleave and may not overlap, and the commit is what
14428    /// catches an overlap.
14429    #[test]
14430    fn runs_that_overlap_each_other_are_refused_at_commit() {
14431        let path = path("overlapping-runs");
14432        let mut writer =
14433            Writer::create(&path, "overlapping", vec![Field::new("v", LogicalType::BigInt)])
14434                .expect("new file");
14435        let one = |order: (u64, u64)| {
14436            let column =
14437                Vector::from_values(LogicalType::BigInt, &[Value::BigInt(1)]).expect("a column");
14438            (order, Chunk::new(vec![column]).expect("one column"))
14439        };
14440        // The second run sits inside the first rather than after it, which is a thing no instance
14441        // holding its own contiguous run can produce and a thing the file cannot represent.
14442        writer.append_stripe(vec![one((0, 0)), one((0, 2))]).expect("a stripe");
14443        writer.append_stripe(vec![one((0, 1))]).expect("a stripe");
14444        let error = writer.finish().expect_err("the runs overlap");
14445        assert!(error.message().contains("source order"), "{error}");
14446        fs::remove_file(path).expect("remove scratch file");
14447    }
14448
14449    /// A stripe holds [`STRIPE_PARTS`] parts, so a run longer than that is a caller bug rather than
14450    /// something to split, and the writer says so at the door instead of quietly cutting it in two.
14451    #[test]
14452    fn a_run_longer_than_a_stripe_is_refused() {
14453        let path = path("overlong-run");
14454        let mut writer =
14455            Writer::create(&path, "overlong", vec![Field::new("v", LogicalType::BigInt)])
14456                .expect("new file");
14457        let parts = (0..=STRIPE_PARTS)
14458            .map(|at| {
14459                let column = Vector::from_values(LogicalType::BigInt, &[Value::BigInt(1)])
14460                    .expect("a column");
14461                let chunk = Chunk::new(vec![column]).expect("one column");
14462                ((0, u64::try_from(at).expect("small")), chunk)
14463            })
14464            .collect::<Vec<_>>();
14465        let error = writer.append_stripe(parts).expect_err("one part too many");
14466        assert!(error.message().contains("more parts than it holds"), "{error}");
14467        fs::remove_file(path).expect("remove scratch file");
14468    }
14469
14470    /// Parts past the stripe bound start a new stripe, and every part stays addressable on its own.
14471    ///
14472    /// This is the shape the format exists for, so both ends of the split are checked here. The
14473    /// directory holds three stripes rather than a hundred and thirty one, and a read of any one
14474    /// part still answers with that part's rows rather than with its whole stripe's.
14475    #[test]
14476    fn parts_past_the_stripe_bound_start_a_new_stripe() {
14477        let path = path("stripe-bound");
14478        let mut writer = Writer::create(
14479            &path,
14480            "items",
14481            vec![
14482                Field::required("id", LogicalType::Integer),
14483                Field::new("text", LogicalType::Varchar),
14484            ],
14485        )
14486        .expect("new file");
14487        let parts = STRIPE_PARTS * 2 + 3;
14488        for part in 0..parts {
14489            let id = part as i32;
14490            let chunk = Chunk::new(vec![
14491                Vector::from_values(
14492                    LogicalType::Integer,
14493                    &[Value::Integer(id), Value::Integer(-id)],
14494                )
14495                .expect("integers"),
14496                Vector::from_values(
14497                    LogicalType::Varchar,
14498                    &[Value::Varchar(format!("value {part}")), Value::Null],
14499                )
14500                .expect("strings"),
14501            ])
14502            .expect("matching rows");
14503            writer.append(&chunk).expect("one part");
14504        }
14505        writer.finish().expect("commit");
14506
14507        let reader = Reader::open(&path).expect("reopen from disk");
14508        assert_eq!(reader.parts(), parts);
14509        assert_eq!(reader.table().rows(), parts * 2);
14510        assert_eq!(reader.table().stripes().len(), parts.div_ceil(STRIPE_PARTS));
14511        assert_eq!(reader.table().stripes()[0].parts(), STRIPE_PARTS);
14512        assert_eq!(reader.table().stripes()[0].rows(), STRIPE_PARTS * 2);
14513        assert_eq!(reader.table().stripes()[2].parts(), 3);
14514        // Backwards on purpose. The reader keeps four stripes a column, so a scan that walks the
14515        // table the other way is what catches a cache that only ever holds what it just read.
14516        for part in (0..parts).rev() {
14517            let dense = reader.read(part, &[0, 1]).expect("a whole page read");
14518            let sparse = reader.read_sparse(part, &[0, 1]).expect("one part read");
14519            for chunk in [&dense, &sparse] {
14520                assert_eq!(chunk.len(), 2, "part {part} has its own row count");
14521                assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
14522                assert_eq!(chunk.value_at(1, 0), Value::Integer(-(part as i32)));
14523                assert_eq!(chunk.value_at(0, 1), Value::Varchar(format!("value {part}")));
14524                assert_eq!(chunk.value_at(1, 1), Value::Null);
14525            }
14526        }
14527        // The bounds are merged over the stripe, so they answer for the range the whole stripe
14528        // covers and not for the part that was asked about.
14529        let above = [Probe { column: 0, op: Op::Greater, value: Bound::Int(100) }];
14530        assert!(reader.skips(0, &above), "the first stripe stops at 63");
14531        assert!(!reader.skips(STRIPE_PARTS * 2, &above), "the third stripe reaches 130");
14532        fs::remove_file(path).expect("remove scratch file");
14533    }
14534
14535    /// A scattered value in the column that decides `WHERE UserID = ?`.
14536    fn scattered(n: i64) -> i64 {
14537        n.wrapping_mul(-7_046_029_254_386_353_131)
14538    }
14539
14540    /// A part whose sieve does not hold the constant is skipped, and a range would skip none of them.
14541    ///
14542    /// This is ClickBench query 19 in miniature. The values are spread over the whole of `BIGINT`, so
14543    /// every stripe's bounds cover nearly all of it and rule out nothing, and the part that really
14544    /// holds the value is the only one a scan has to read.
14545    #[test]
14546    fn a_part_is_skipped_when_its_sieve_does_not_hold_the_constant() {
14547        let path = path("sieve-skip");
14548        let mut writer =
14549            Writer::create(&path, "hits", vec![Field::required("id", LogicalType::BigInt)])
14550                .expect("new file");
14551        let parts = STRIPE_PARTS + 3;
14552        // Big enough that the filter is worth its bytes. A part of eight numbers packs to under a
14553        // hundred bytes and the smallest filter there is is sixty nine, so a filter over a part
14554        // that small costs about as much to read as the rows do and is no longer written.
14555        let per_part = 128;
14556        for part in 0..parts {
14557            let held: Vec<Value> = (0..per_part)
14558                .map(|row| Value::BigInt(scattered((part * per_part + row) as i64)))
14559                .collect();
14560            let chunk =
14561                Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
14562                    .expect("one column");
14563            writer.append(&chunk).expect("one part");
14564        }
14565        writer.finish().expect("commit");
14566
14567        let reader = Reader::open(&path).expect("reopen from disk");
14568        let probe = |value: i64| Probe {
14569            column: 0,
14570            op: Op::Equal,
14571            value: Bound::Int(i128::from(scattered(value))),
14572        };
14573        for wanted in [0_i64, (per_part + 1) as i64, (parts * per_part - 1) as i64] {
14574            let tests = [probe(wanted)];
14575            let kept: Vec<usize> = (0..parts).filter(|&part| !reader.skips(part, &tests)).collect();
14576            let home = wanted as usize / per_part;
14577            assert!(kept.contains(&home), "the part holding {wanted} is read");
14578            // A filter answers maybe, so a part it keeps need not hold the value. Sixty seven parts
14579            // of a hundred and twenty eight numbers each, at a dozen bits a value, is about one
14580            // stray part across the whole file and that is what this leaves room for.
14581            assert!(kept.len() <= 2, "{wanted} keeps {kept:?}, which is more than one stray part");
14582        }
14583        let absent = [probe((parts * per_part) as i64 + 1)];
14584        let kept = (0..parts).filter(|&part| !reader.skips(part, &absent)).count();
14585        assert!(kept <= 1, "{kept} parts of {parts} kept a value no part holds");
14586        // The same probes against the bounds alone, which is what this replaces. A column of
14587        // scattered numbers has a range per stripe that covers nearly the whole type.
14588        let tests = [probe(0)];
14589        assert!(
14590            reader.table().stripes().iter().all(|stripe| !stripe.zone.skips(&tests)),
14591            "the bounds rule out no stripe at all"
14592        );
14593        fs::remove_file(path).expect("remove scratch file");
14594    }
14595
14596    /// A part whose own bounds rule out an ordered comparison is skipped where the stripe's keep it.
14597    ///
14598    /// This is the shape of ClickBench 24. Each part covers a narrow stretch of the column and the
14599    /// stripe covers all sixty four of them at once, so a comparison that lands inside the stripe
14600    /// rules out none of it and rules out all but a few parts.
14601    #[test]
14602    fn a_part_is_skipped_when_its_own_bounds_rule_out_a_comparison_the_stripe_keeps() {
14603        let path = path("part-range-skip");
14604        let mut writer =
14605            Writer::create(&path, "hits", vec![Field::required("at", LogicalType::BigInt)])
14606                .expect("new file");
14607        let parts = STRIPE_PARTS + 3;
14608        let per_part = 128;
14609        for part in 0..parts {
14610            // Scattered inside the part's own band rather than a run, because a run of
14611            // consecutive numbers encodes to a stride of a few bytes and then the page of ranges
14612            // costs more than reading the column it indexes, which is the case the writer declines.
14613            let held: Vec<Value> = (0..per_part)
14614                .map(|row| {
14615                    Value::BigInt((part * 1_000) as i64 + (scattered(row as i64).rem_euclid(900)))
14616                })
14617                .collect();
14618            let chunk =
14619                Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
14620                    .expect("one column");
14621            writer.append(&chunk).expect("one part");
14622        }
14623        writer.finish().expect("commit");
14624
14625        let reader = Reader::open(&path).expect("reopen from disk");
14626        let under = [Probe { column: 0, op: Op::Less, value: Bound::Int(3_000) }];
14627        let kept: Vec<usize> = (0..parts).filter(|&part| !reader.skips(part, &under)).collect();
14628        assert_eq!(kept, vec![0, 1, 2], "only the three parts that start under three thousand");
14629        // The same question asked of the stripe alone, which is what this replaces.
14630        assert!(!reader.stripe_skips(0, &under), "the stripe reaches from zero and keeps itself");
14631        fs::remove_file(path).expect("remove scratch file");
14632    }
14633
14634    /// The other half of the same page. A part whose own bounds put every row of it inside the
14635    /// filter is waved through, so the comparison never runs on it, where the stripe's bounds reach
14636    /// across every part and can prove nothing.
14637    #[test]
14638    fn a_part_is_waved_through_when_its_own_bounds_pass_a_comparison_the_stripe_cannot() {
14639        let path = path("part-range-certain");
14640        let mut writer =
14641            Writer::create(&path, "hits", vec![Field::required("at", LogicalType::BigInt)])
14642                .expect("new file");
14643        let parts = STRIPE_PARTS + 3;
14644        let per_part = 128;
14645        for part in 0..parts {
14646            let held: Vec<Value> = (0..per_part)
14647                .map(|row| {
14648                    Value::BigInt((part * 1_000) as i64 + (scattered(row as i64).rem_euclid(900)))
14649                })
14650                .collect();
14651            let chunk =
14652                Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
14653                    .expect("one column");
14654            writer.append(&chunk).expect("one part");
14655        }
14656        writer.finish().expect("commit");
14657
14658        let reader = Reader::open(&path).expect("reopen from disk");
14659        let under = [Probe { column: 0, op: Op::Less, value: Bound::Int(3_000) }];
14660        let waved: Vec<usize> = (0..parts).filter(|&part| reader.certain(part, &under)).collect();
14661        assert_eq!(waved, vec![0, 1, 2], "the three parts that end under three thousand");
14662        // The first stripe reaches from zero to past sixty thousand, so it straddles three thousand
14663        // and settles nothing either way. The three yeses above are the parts' own ends talking.
14664        assert!(!reader.stripe_skips(0, &under), "the stripe straddles the comparison");
14665        fs::remove_file(path).expect("remove scratch file");
14666    }
14667
14668    /// The page is worth its bytes on a column with parts to tell apart and is not written on one
14669    /// that has a single part, where the stripe bounds already are the part's.
14670    #[test]
14671    fn a_stripe_of_one_part_writes_no_range_page_and_a_stripe_of_many_does() {
14672        for (parts, wanted) in [(1_usize, false), (STRIPE_PARTS, true)] {
14673            let path = path("part-range-page");
14674            let mut writer =
14675                Writer::create(&path, "hits", vec![Field::required("at", LogicalType::BigInt)])
14676                    .expect("new file");
14677            for part in 0..parts {
14678                let held: Vec<Value> = (0..128)
14679                    .map(|row| {
14680                        Value::BigInt((part * 1_000) as i64 + scattered(row as i64).rem_euclid(900))
14681                    })
14682                    .collect();
14683                let chunk = Chunk::new(vec![
14684                    Vector::from_values(LogicalType::BigInt, &held).expect("numbers"),
14685                ])
14686                .expect("one column");
14687                writer.append(&chunk).expect("one part");
14688            }
14689            writer.finish().expect("commit");
14690            let reader = Reader::open(&path).expect("reopen from disk");
14691            let bytes = reader.layout().columns[0].part_ranges;
14692            assert_eq!(bytes > 0, wanted, "{parts} parts wrote {bytes} bytes of ranges");
14693            fs::remove_file(path).expect("remove scratch file");
14694        }
14695    }
14696
14697    /// A cut down string end is still an end on the side it was, which is the only thing that keeps
14698    /// a shortened bound from turning a skip into a wrong answer.
14699    #[test]
14700    fn a_string_end_that_is_cut_down_still_covers_the_value_it_came_from() {
14701        let long = vec![b'a'; PART_BOUND_BYTES * 2];
14702        let low = shortened(Some(Bound::Bytes(long.clone())), false).expect("a low end");
14703        let high = shortened(Some(Bound::Bytes(long.clone())), true).expect("a high end");
14704        let Bound::Bytes(low) = low else { panic!("a string stays a string") };
14705        let Bound::Bytes(high) = high else { panic!("a string stays a string") };
14706        assert!(low.len() <= PART_BOUND_BYTES && high.len() <= PART_BOUND_BYTES);
14707        assert!(low.as_slice() <= long.as_slice(), "the low end is at or under the value");
14708        assert!(high.as_slice() >= long.as_slice(), "the high end is at or over the value");
14709    }
14710
14711    /// A string of nothing but the largest byte has no prefix that can be stepped up, so the high
14712    /// end is given up rather than claimed too small. No end keeps the part, which is always safe.
14713    #[test]
14714    fn a_string_end_with_no_room_to_step_up_gives_up_the_bound() {
14715        let long = vec![u8::MAX; PART_BOUND_BYTES * 2];
14716        assert_eq!(shortened(Some(Bound::Bytes(long.clone())), true), None);
14717        let low = shortened(Some(Bound::Bytes(long)), false).expect("a low end is still a prefix");
14718        assert_eq!(low, Bound::Bytes(vec![u8::MAX; PART_BOUND_BYTES]));
14719    }
14720
14721    /// What a column is stored as, asked of two files holding the same rows in a different order.
14722    ///
14723    /// This is the question the report exists to answer and it is the one the directory cannot. The
14724    /// two files have the same rows, the same schema and the same number of parts, and the column
14725    /// comes out four times smaller in one of them, because ascending keys delta encode to a few
14726    /// bits a row and shuffled ones do not. Nothing about the file's shape says so. The page header
14727    /// says so, and reading it is what this does.
14728    ///
14729    /// It is q18 on TPC-H in miniature: clustering lineitem by ship date leaves `l_orderkey`
14730    /// ascending inside a partition but sparse, its deltas go from six bits to twelve, and the scan
14731    /// pays for the wider ones.
14732    #[test]
14733    fn what_a_column_is_stored_as_follows_the_order_the_rows_were_written_in() {
14734        let parts = 4;
14735        let per_part = 1024;
14736        let rows = parts * per_part;
14737        let written = |name: &str, keys: &[i64]| {
14738            let path = path(name);
14739            let fields = vec![Field::required("key", LogicalType::BigInt)];
14740            let mut writer = Writer::create(&path, "keys", fields).expect("new file");
14741            for part in 0..parts {
14742                let values: Vec<Value> = keys[part * per_part..(part + 1) * per_part]
14743                    .iter()
14744                    .map(|key| Value::BigInt(*key))
14745                    .collect();
14746                let chunk = Chunk::new(vec![
14747                    Vector::from_values(LogicalType::BigInt, &values).expect("numbers"),
14748                ])
14749                .expect("one column");
14750                writer.append(&chunk).expect("one part");
14751            }
14752            writer.finish().expect("commit");
14753            path
14754        };
14755        // Ascending with a small irregular step, which is what a key column in arrival order looks
14756        // like: an order has one to seven line items, so the key repeats and then moves on by one.
14757        let climbing = |step: &dyn Fn(usize) -> i64| {
14758            let mut key = 0;
14759            (0..rows)
14760                .map(|row| {
14761                    key += step(row);
14762                    key
14763                })
14764                .collect::<Vec<i64>>()
14765        };
14766        let ascending = climbing(&|row| (row % 3) as i64);
14767        // The same rows in the same direction over a range a thousand times wider, which is what a
14768        // partition of a clustered table holds: still ascending, and far enough apart that the
14769        // deltas no longer fit in a handful of bits.
14770        let sparse = climbing(&|row| ((row * 2_654_435_761) % 4096) as i64);
14771        let near_path = written("stored-near", &ascending);
14772        let far_path = written("stored-far", &sparse);
14773
14774        let one = Reader::open(&near_path).expect("reopen from disk");
14775        let other = Reader::open(&far_path).expect("reopen from disk");
14776        let near = one.stored(0).expect("the column is stored");
14777        let far = other.stored(0).expect("the column is stored");
14778        assert_eq!(near.len(), parts, "one row per part");
14779        assert_eq!(far.len(), parts);
14780        // The bytes are the same bytes the directory totals, which is the check that this is
14781        // reading the pages the file really holds rather than some other pages.
14782        let total = |stored: &[StoredPart]| stored.iter().map(|part| part.bytes).sum::<u64>();
14783        assert_eq!(total(&near), one.layout().columns[0].pages);
14784        assert_eq!(total(&far), other.layout().columns[0].pages);
14785        assert!(
14786            total(&near) * 2 < total(&far),
14787            "the sparse keys cost more, {} against {}",
14788            total(&far),
14789            total(&near)
14790        );
14791        // Every part accounted for, in order, with the row it starts at following the one before.
14792        for (at, part) in near.iter().enumerate() {
14793            assert_eq!(part.part, at);
14794            assert_eq!(part.row, at * per_part);
14795            assert_eq!(part.rows, per_part);
14796            let held = &ascending[at * per_part..(at + 1) * per_part];
14797            assert_eq!(part.low, Some(Value::BigInt(held[0])));
14798            assert_eq!(part.high, Some(Value::BigInt(held[per_part - 1])));
14799            assert_eq!(part.nulls, Some(0));
14800        }
14801        // And the encoding is a line of text that names what the encoder chose, which is the whole
14802        // point. Both are a cascade over deltas and the widths inside them are what differ.
14803        assert!(near[0].encoding.contains("DELTA"), "{}", near[0].encoding);
14804        assert!(far[0].encoding.contains("DELTA"), "{}", far[0].encoding);
14805        assert_ne!(near[0].encoding, far[0].encoding);
14806        fs::remove_file(near_path).expect("remove scratch file");
14807        fs::remove_file(far_path).expect("remove scratch file");
14808    }
14809
14810    /// A sieve bigger than the part it indexes is not written, and one smaller than it still is.
14811    ///
14812    /// Both columns hold values spread over the whole of `BIGINT`, so neither gets a bitmap and both
14813    /// reach the filter. They differ in what the part costs to read. `spread` is a thousand distinct
14814    /// numbers and packs to eight kilobytes, so a filter of about thirteen hundred bytes is a good
14815    /// trade. `repeated` is the same thousand rows over four numbers in runs and encodes to
14816    /// almost nothing, but the filter is sized for the rows rather than the values it turns out to
14817    /// hold, so it comes out larger than the data. Reading it to decide whether to read the part spends more than
14818    /// the part, every time, and that is the case this drops.
14819    #[test]
14820    fn a_sieve_larger_than_the_part_it_indexes_is_not_written() {
14821        let path = path("sieve-pays");
14822        let fields = vec![
14823            Field::required("spread", LogicalType::BigInt),
14824            Field::required("repeated", LogicalType::BigInt),
14825        ];
14826        let mut writer = Writer::create(&path, "hits", fields).expect("new file");
14827        let parts = 3;
14828        let per_part = 1024;
14829        for part in 0..parts {
14830            let base = (part * per_part) as i64;
14831            let spread: Vec<Value> =
14832                (0..per_part).map(|row| Value::BigInt(scattered(base + row as i64))).collect();
14833            let repeated: Vec<Value> =
14834                (0..per_part).map(|row| Value::BigInt(scattered((row / 256) as i64))).collect();
14835            let chunk = Chunk::new(vec![
14836                Vector::from_values(LogicalType::BigInt, &spread).expect("numbers"),
14837                Vector::from_values(LogicalType::BigInt, &repeated).expect("numbers"),
14838            ])
14839            .expect("two columns");
14840            writer.append(&chunk).expect("one part");
14841        }
14842        writer.finish().expect("commit");
14843
14844        let reader = Reader::open(&path).expect("reopen from disk");
14845        let layout = reader.layout();
14846        let spread = &layout.columns[0];
14847        let repeated = &layout.columns[1];
14848        assert!(spread.sieves > 0, "a column whose parts are worth a filter keeps one");
14849        assert_eq!(
14850            repeated.sieves, 0,
14851            "a column whose filter costs more than its parts keeps none"
14852        );
14853        // Per part this is the rule itself, so it holds over the column as well: a part without a
14854        // sieve adds to one side of this and to nothing on the other.
14855        for column in &layout.columns {
14856            assert!(
14857                column.sieves < column.pages,
14858                "{} spends {} on sieves over {} of data",
14859                column.name,
14860                column.sieves,
14861                column.pages
14862            );
14863        }
14864        // The filter that was kept still does what it is for.
14865        let absent = [Probe {
14866            column: 0,
14867            op: Op::Equal,
14868            value: Bound::Int(i128::from(scattered((parts * per_part) as i64 + 1))),
14869        }];
14870        assert!((0..parts).all(|part| reader.skips(part, &absent)), "no part holds it");
14871        fs::remove_file(path).expect("remove scratch file");
14872    }
14873
14874    /// A damaged sieve page is a part that gets read, not a query that fails.
14875    ///
14876    /// A sieve is an index over rows that are still there and still correct, so losing one costs
14877    /// time and costs no answers. That is the opposite of the membership index beside it, which is
14878    /// the only thing standing between a string page and a wrong answer.
14879    #[test]
14880    fn a_damaged_sieve_page_is_read_through_rather_than_refused() {
14881        let path = path("sieve-damaged");
14882        let mut writer =
14883            Writer::create(&path, "hits", vec![Field::required("id", LogicalType::BigInt)])
14884                .expect("new file");
14885        let rows = 128;
14886        let held: Vec<Value> = (0..rows).map(|row| Value::BigInt(scattered(row))).collect();
14887        let chunk =
14888            Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
14889                .expect("one column");
14890        writer.append(&chunk).expect("one part");
14891        writer.finish().expect("commit");
14892
14893        let page = Reader::open(&path).expect("reopen").table.stripes[0]
14894            .sieves
14895            .get(0)
14896            .expect("a sieve page");
14897        let mut file = OpenOptions::new().write(true).open(&path).expect("open the sieve page");
14898        file.seek(SeekFrom::Start(page.offset + u64::from(page.length) - 1)).expect("seek");
14899        file.write_all(&[0xff]).expect("damage one byte");
14900        drop(file);
14901
14902        let reader = Reader::open(&path).expect("reopen the damaged file");
14903        let absent =
14904            [Probe { column: 0, op: Op::Equal, value: Bound::Int(i128::from(scattered(99))) }];
14905        assert!(!reader.skips(0, &absent), "a sieve that cannot be read skips nothing");
14906        assert_eq!(
14907            reader.read(0, &[0]).expect("the rows are untouched").len(),
14908            usize::try_from(rows).expect("a small count")
14909        );
14910        fs::remove_file(path).expect("remove scratch file");
14911    }
14912
14913    /// A scan that asks for each part twice reads each page whole once and keeps only the floor.
14914    ///
14915    /// This is ClickBench 21's shape: a `LIKE` asks a compressed text part whether it can answer and
14916    /// then reads the part. Counting parts rather than asks is what stops the second ask of every
14917    /// part from looking like a second scan, which would pool every page of the column.
14918    #[test]
14919    fn a_part_asked_for_twice_in_one_scan_keeps_its_page_only_to_the_floor() {
14920        let path = path("asked-twice");
14921        let parts = STRIPE_PARTS * (CACHED_STRIPES_PER_COLUMN + 2);
14922        let mut writer =
14923            Writer::create(&path, "a", vec![Field::required("id", LogicalType::Integer)])
14924                .expect("new file");
14925        for part in 0..parts {
14926            let chunk = Chunk::new(vec![
14927                Vector::from_values(LogicalType::Integer, &[Value::Integer(part as i32)])
14928                    .expect("integers"),
14929            ])
14930            .expect("matching rows");
14931            writer.append(&chunk).expect("one part");
14932        }
14933        writer.finish().expect("commit");
14934
14935        let pool = PagePool::new(usize::MAX);
14936        let catalog = Catalog::open_in(&path, &pool).expect("the file opens");
14937        let a = catalog.table("a").expect("a");
14938        let stripes = a.table().stripes().len();
14939        for part in 0..parts {
14940            for _ in 0..2 {
14941                let chunk = a.read(part, &[0]).expect("a part");
14942                assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
14943            }
14944        }
14945        assert_eq!(
14946            a.pages.load(Atomic::Relaxed),
14947            stripes,
14948            "a page a stripe, read on the second ask"
14949        );
14950        assert_eq!(pool.bytes(), 0, "one scan puts nothing in the pool");
14951        let column = a.cache.columns[0].lock().expect("the column");
14952        assert_eq!(column.pages.iter().flatten().count(), CACHED_STRIPES_PER_COLUMN);
14953        drop(column);
14954        drop((a, catalog));
14955        fs::remove_file(path).expect("remove scratch file");
14956    }
14957
14958    /// Eight workers over one stripe read it once between them.
14959    ///
14960    /// This is the shape a scan actually has. Parts are handed out in order, so every worker on a
14961    /// column crosses into a stripe within a few parts of the others, and before [`Reader::held`]
14962    /// started sharing the read every one of them read the whole page. On the full ClickBench file
14963    /// that was a `MIN(EventDate), MAX(EventDate)` moving 3.2 GB off the disk to look at 400 MB of
14964    /// column, which is most of what a first touch costs.
14965    ///
14966    /// The workers that lose the race still answer, out of the part reads they do instead, which is
14967    /// what the values below are checking.
14968    #[test]
14969    fn workers_that_want_the_same_stripe_read_it_once() {
14970        let path = path("single-flight");
14971        let mut writer =
14972            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
14973                .expect("new file");
14974        for part in 0..STRIPE_PARTS {
14975            let id = part as i32;
14976            let chunk = Chunk::new(vec![
14977                Vector::from_values(
14978                    LogicalType::Integer,
14979                    &[Value::Integer(id), Value::Integer(-id)],
14980                )
14981                .expect("integers"),
14982            ])
14983            .expect("matching rows");
14984            writer.append(&chunk).expect("one part");
14985        }
14986        writer.finish().expect("commit");
14987
14988        let reader = Reader::open(&path).expect("reopen from disk");
14989        assert_eq!(reader.table().stripes().len(), 1, "one stripe is the point of the test");
14990        // Once through a part at a time first, since a stripe's page is only read whole the second
14991        // time a scan comes to it.
14992        for part in 0..STRIPE_PARTS {
14993            reader.read(part, &[0]).expect("a part");
14994        }
14995        assert_eq!(reader.pages.load(Atomic::Relaxed), 0, "the first pass reads no page whole");
14996        let barrier = std::sync::Barrier::new(8);
14997        std::thread::scope(|scope| {
14998            for worker in 0..8 {
14999                let reader = &reader;
15000                let barrier = &barrier;
15001                scope.spawn(move || {
15002                    barrier.wait();
15003                    for part in (worker..STRIPE_PARTS).step_by(8) {
15004                        let chunk = reader.read(part, &[0]).expect("a whole page read");
15005                        assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
15006                        assert_eq!(chunk.value_at(1, 0), Value::Integer(-(part as i32)));
15007                    }
15008                });
15009            }
15010        });
15011        assert_eq!(reader.pages.load(Atomic::Relaxed), 1, "one stripe, one page read, whoever won");
15012        fs::remove_file(path).expect("remove scratch file");
15013    }
15014
15015    /// Opening a file reads the header and the directory, and nothing that depends on the rows.
15016    ///
15017    /// `spec/stats/04-in-memory.md` section 4.2. There are no statistics in the file yet, so this
15018    /// holds today by not having anything to load, and that is exactly why it is worth pinning now.
15019    /// The change that breaks it is the reasonable looking one: summaries are a few hundred bytes,
15020    /// the next query will want them, so read them on the way past. A process that opened the
15021    /// database to run one trivial query pays for all of it and gets nothing.
15022    ///
15023    /// Two files of the same shape and a thousand times the rows in one of them, opened, and the
15024    /// two openings cost the same. The stripe count is held equal so that the directory is the same
15025    /// size in both, which leaves the rows as the only thing that changed. Anything read out of the
15026    /// data would show up here.
15027    #[test]
15028    fn opening_costs_the_same_over_a_thousand_times_the_rows() {
15029        let opened = |label: &str, rows_per_part: i32| {
15030            let path = path(label);
15031            let mut writer =
15032                Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
15033                    .expect("new file");
15034            for part in 0..STRIPE_PARTS * 3 {
15035                // Scrambled rather than sequential, so that the fat file is actually fatter. A run
15036                // of consecutive integers encodes to almost nothing and would leave the two files
15037                // the same size, which would make this test pass for the wrong reason.
15038                let values = (0..rows_per_part)
15039                    .map(|row| {
15040                        Value::Integer((part as i32 * rows_per_part + row).wrapping_mul(2_654_435))
15041                    })
15042                    .collect::<Vec<_>>();
15043                let chunk = Chunk::new(vec![
15044                    Vector::from_values(LogicalType::Integer, &values).expect("integers"),
15045                ])
15046                .expect("matching rows");
15047                writer.append(&chunk).expect("one part");
15048            }
15049            writer.finish().expect("commit");
15050            let reader = Reader::open(&path).expect("reopen from disk");
15051            let size = fs::metadata(&path).expect("the file is there").len();
15052            let out = (reader.reads(), reader.table().stripes().len(), size);
15053            fs::remove_file(path).expect("remove scratch file");
15054            out
15055        };
15056
15057        let (thin, thin_stripes, thin_size) = opened("open-thin", 1);
15058        let (fat, fat_stripes, fat_size) = opened("open-fat", 1000);
15059        assert_eq!(
15060            thin_stripes, fat_stripes,
15061            "the same stripe count is what makes this a fair ask"
15062        );
15063        assert!(
15064            fat_size > thin_size * 50,
15065            "the fat file has to actually be larger, and it is {fat_size} against {thin_size}"
15066        );
15067
15068        assert_eq!(thin.opening.reads, fat.opening.reads, "the same reads either way");
15069        assert_eq!(thin.pages, 0, "opening read a page");
15070        assert_eq!(fat.pages, 0, "opening read a page");
15071        assert_eq!(thin.indexes, 0, "opening read an index");
15072        assert_eq!(fat.indexes, 0, "opening read an index");
15073        // Not exactly equal, because a directory holds offsets and a larger file has larger ones,
15074        // and a handful of bytes of varint is not somebody loading statistics. A factor is.
15075        assert!(
15076            fat.opening.bytes < thin.opening.bytes * 2,
15077            "opening the thin file read {} bytes and the fat one read {}",
15078            thin.opening.bytes,
15079            fat.opening.bytes
15080        );
15081    }
15082
15083    /// The reads a file costs to open are fixed by its shape and not by what ran before.
15084    ///
15085    /// `spec/stats/04-in-memory.md` section 4.3, which is the rule that keeps a plan reproducible:
15086    /// the plan is a function of the data, the generation and the settings, and never of what
15087    /// happened to be in cache. Opening the same file twice in the same process has to cost the
15088    /// same, because a second open that read less would be an open that was about to plan
15089    /// differently.
15090    #[test]
15091    fn two_opens_of_one_file_cost_the_same_and_the_second_is_not_cheaper() {
15092        let path = path("open-twice");
15093        let mut writer =
15094            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
15095                .expect("new file");
15096        for part in 0..STRIPE_PARTS * 3 {
15097            let chunk = Chunk::new(vec![
15098                Vector::from_values(LogicalType::Integer, &[Value::Integer(part as i32)])
15099                    .expect("integers"),
15100            ])
15101            .expect("matching rows");
15102            writer.append(&chunk).expect("one part");
15103        }
15104        writer.finish().expect("commit");
15105
15106        let first = Reader::open(&path).expect("open");
15107        // A whole scan in between, so the operating system's page cache is as warm as it gets and
15108        // anything that consulted it would show up in the second open.
15109        for part in 0..first.parts() {
15110            first.read(part, &[0]).expect("a part");
15111        }
15112        assert!(first.reads().indexes > 0, "the scan has to have read something");
15113        let second = Reader::open(&path).expect("open again");
15114
15115        assert_eq!(first.reads().opening, second.reads().opening);
15116        assert_eq!(
15117            second.reads().pages,
15118            0,
15119            "the second open read a page off the back of the first"
15120        );
15121        assert_eq!(second.reads().indexes, 0, "the second open read an index it inherited");
15122        fs::remove_file(path).expect("remove scratch file");
15123    }
15124
15125    /// A scan reads a stripe's index once for the whole scan, not once per part that misses.
15126    ///
15127    /// The page cache holds four stripes and an index used to ride inside it, so a table with more
15128    /// stripes than that read the index again every time a stripe came back around. The index is a
15129    /// few hundred bytes and the page is a quarter of a megabyte, which is why they are now under
15130    /// different budgets. This is the test that keeps them there, since the saving is small enough
15131    /// that nothing in a benchmark would notice it going away again.
15132    #[test]
15133    fn an_index_is_read_once_per_stripe_however_often_the_page_is_evicted() {
15134        let path = path("index-cache");
15135        let mut writer =
15136            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
15137                .expect("new file");
15138        let parts = STRIPE_PARTS * (CACHED_STRIPES_PER_COLUMN + 2);
15139        for part in 0..parts {
15140            let id = part as i32;
15141            let chunk = Chunk::new(vec![
15142                Vector::from_values(LogicalType::Integer, &[Value::Integer(id)]).expect("integers"),
15143            ])
15144            .expect("matching rows");
15145            writer.append(&chunk).expect("one part");
15146        }
15147        writer.finish().expect("commit");
15148
15149        let reader = Reader::open(&path).expect("reopen from disk");
15150        let stripes = reader.table().stripes().len();
15151        assert!(stripes > CACHED_STRIPES_PER_COLUMN, "the page cache has to be too small for this");
15152        // Three times over. The first pass reads a part at a time, the second reads the pages, and
15153        // the third finds every page evicted and every index kept.
15154        for _ in 0..3 {
15155            for part in 0..parts {
15156                let chunk = reader.read(part, &[0]).expect("a part");
15157                assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
15158            }
15159        }
15160        assert_eq!(reader.indexes.load(Atomic::Relaxed), stripes, "one index read per stripe");
15161        assert!(
15162            reader.pages.load(Atomic::Relaxed) > stripes,
15163            "the pages are the ones that get read again, which is what makes the index count mean \
15164             something"
15165        );
15166        fs::remove_file(path).expect("remove scratch file");
15167    }
15168
15169    /// A page stays in memory from one scan to the next while the pool has room for it, and a
15170    /// table that is being read takes room from one that is not, down to the floor and no further.
15171    ///
15172    /// This is what the pool is for. Each reader lives as long as its database, so a second query
15173    /// over the same table should find every page it read the first time, and before the pool it
15174    /// found four stripes a column and read the rest off the file again.
15175    #[test]
15176    fn a_pool_keeps_pages_between_scans_and_gives_them_to_the_table_being_read() {
15177        let path = path("page-pool");
15178        let parts = STRIPE_PARTS * (CACHED_STRIPES_PER_COLUMN * 2 + 2);
15179        let fields = || vec![Field::required("id", LogicalType::Integer)];
15180        let mut writer = Writer::create(&path, "a", fields()).expect("new file");
15181        for table in ["a", "b"] {
15182            if table == "b" {
15183                writer = writer.next("b".to_string(), fields()).expect("a second table");
15184            }
15185            for part in 0..parts {
15186                let chunk = Chunk::new(vec![
15187                    Vector::from_values(LogicalType::Integer, &[Value::Integer(part as i32)])
15188                        .expect("integers"),
15189                ])
15190                .expect("matching rows");
15191                writer.append(&chunk).expect("one part");
15192            }
15193        }
15194        writer.finish().expect("commit");
15195
15196        let pool = PagePool::new(usize::MAX);
15197        let catalog = Catalog::open_in(&path, &pool).expect("the file opens");
15198        let (a, b) = (catalog.table("a").expect("a"), catalog.table("b").expect("b"));
15199        let stripes = a.table().stripes().len();
15200        assert!(
15201            stripes > CACHED_STRIPES_PER_COLUMN * 2,
15202            "the floor has to be smaller than a table"
15203        );
15204        let scan = |reader: &Reader| {
15205            for part in 0..parts {
15206                let chunk = reader.read(part, &[0]).expect("a part");
15207                assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
15208            }
15209        };
15210        // The first scan reads a part at a time and keeps no page, the second reads every page and
15211        // keeps it, and the third reads nothing.
15212        scan(&a);
15213        assert_eq!(a.pages.load(Atomic::Relaxed), 0, "the first scan reads no page whole");
15214        assert_eq!(pool.bytes(), 0, "a stripe read once is not the pool's");
15215        scan(&a);
15216        assert_eq!(a.pages.load(Atomic::Relaxed), stripes, "the second scan reads every page");
15217        scan(&a);
15218        assert_eq!(a.pages.load(Atomic::Relaxed), stripes, "the third scan reads nothing");
15219        let one = pool.bytes();
15220        assert!(one > 0, "the pool counts what the reader holds");
15221
15222        // Room for one table. Reading the other takes the first one's pages down to its floor.
15223        pool.budget.store(one, Atomic::Relaxed);
15224        scan(&b);
15225        scan(&b);
15226        assert_eq!(b.pages.load(Atomic::Relaxed), stripes, "a page is never let go while in use");
15227        assert_eq!(a.cache.held[0].load(Atomic::Relaxed), CACHED_STRIPES_PER_COLUMN);
15228        let column = a.cache.columns[0].lock().expect("the column");
15229        let held = column.pages.iter().flatten().count();
15230        assert_eq!(held, CACHED_STRIPES_PER_COLUMN, "the count and the slots agree");
15231        drop(column);
15232
15233        // A reader that goes takes its pages out of the count with it.
15234        drop((a, b, catalog));
15235        let c = Catalog::open_in(&path, &pool).expect("again").table("a").expect("a");
15236        scan(&c);
15237        scan(&c);
15238        assert!(pool.bytes() <= one, "only what the live reader holds is counted");
15239        fs::remove_file(path).expect("remove scratch file");
15240    }
15241
15242    /// A worker per stripe reads its stripe once, once the cache has been told how many there are.
15243    ///
15244    /// This is the shape a scan has when it hands out a whole stripe per morsel rather than a part.
15245    /// Nobody races for a page any more, but every worker holds a different one for the length of a
15246    /// stripe, so a cache that keeps four pages while eight workers are in eight stripes evicts
15247    /// every one of them before its owner has finished with it, and the owner reads a quarter of a
15248    /// megabyte again for the next part. The barrier is what makes that certain rather than likely:
15249    /// without it a worker can run a whole stripe before the next one starts and never collide.
15250    #[test]
15251    fn a_worker_per_stripe_reads_its_page_once_when_the_cache_was_told_to_expect_it() {
15252        let workers = CACHED_STRIPES_PER_COLUMN + 4;
15253        let path = path("stripe-per-worker");
15254        let mut writer =
15255            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
15256                .expect("new file");
15257        for part in 0..STRIPE_PARTS * workers {
15258            let chunk = Chunk::new(vec![
15259                Vector::from_values(LogicalType::Integer, &[Value::Integer(part as i32)])
15260                    .expect("integers"),
15261            ])
15262            .expect("matching rows");
15263            writer.append(&chunk).expect("one part");
15264        }
15265        writer.finish().expect("commit");
15266
15267        let read = |told: bool| {
15268            let reader = Reader::open(&path).expect("reopen from disk");
15269            assert_eq!(reader.table().stripes().len(), workers, "a stripe per worker");
15270            if told {
15271                reader.keep_stripes(workers);
15272            }
15273            // Through once a part at a time, so that the pass below is the one that reads pages.
15274            for part in 0..reader.parts() {
15275                reader.read(part, &[0]).expect("a part");
15276            }
15277            let barrier = std::sync::Barrier::new(workers);
15278            std::thread::scope(|scope| {
15279                for (worker, run) in reader.stripe_parts().into_iter().enumerate() {
15280                    let reader = &reader;
15281                    let barrier = &barrier;
15282                    scope.spawn(move || {
15283                        for part in run {
15284                            barrier.wait();
15285                            let chunk = reader.read(part, &[0]).expect("a part of my own stripe");
15286                            assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
15287                        }
15288                        assert!(worker < workers);
15289                    });
15290                }
15291            });
15292            reader.pages.load(Atomic::Relaxed)
15293        };
15294
15295        assert_eq!(read(true), workers, "one page read per stripe and no more");
15296        assert!(read(false) > workers, "a cache that small is read again on every part");
15297        fs::remove_file(path).expect("remove scratch file");
15298    }
15299
15300    /// A damaged index page is caught before anything decodes a part out of it.
15301    ///
15302    /// The index is the one structure a reader trusts to find bytes with, so it carries a checksum
15303    /// per column section rather than one for the page, and this is what says that check runs.
15304    #[test]
15305    fn a_damaged_index_page_is_an_error() {
15306        let path = path("damaged-index");
15307        let mut writer =
15308            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
15309                .expect("new file");
15310        writer.append(&sample_ids()).expect("first part");
15311        writer.append(&sample_ids()).expect("second part");
15312        writer.finish().expect("commit");
15313
15314        let reader = Reader::open(&path).expect("valid directory");
15315        let index = reader.table.stripes[0].index;
15316        let mut byte = [0; 1];
15317        read_at(&reader.file, index.offset, &mut byte).expect("the first part length");
15318        let mut file = OpenOptions::new().write(true).open(&path).expect("open index page");
15319        file.seek(SeekFrom::Start(index.offset)).expect("index start");
15320        file.write_all(&[!byte[0]]).expect("damage the first part length");
15321        let error = reader.read(1, &[0]).expect_err("a damaged index must not be used");
15322        assert!(error.message().contains("index page section checksum differs"), "{error}");
15323        fs::remove_file(path).expect("remove scratch file");
15324    }
15325
15326    /// Every integer width the format knows about, written and read back.
15327    ///
15328    /// The unsigned ones are the reason ClickBench can be stored at all: `hits` types `EventDate`
15329    /// as `USMALLINT`, and one unsupported column meant the whole table was refused. The extremes
15330    /// are in here on purpose, because a width that round trips through the wrong signedness only
15331    /// goes wrong at the end of its range.
15332    #[test]
15333    fn every_integer_width_round_trips_through_a_page() {
15334        let path = path("integer-widths");
15335        let columns = [
15336            (LogicalType::TinyInt, vec![Value::TinyInt(i8::MIN), Value::TinyInt(i8::MAX)]),
15337            (LogicalType::UTinyInt, vec![Value::UTinyInt(0), Value::UTinyInt(u8::MAX)]),
15338            (LogicalType::SmallInt, vec![Value::SmallInt(i16::MIN), Value::SmallInt(i16::MAX)]),
15339            (LogicalType::USmallInt, vec![Value::USmallInt(0), Value::USmallInt(u16::MAX)]),
15340            (LogicalType::Integer, vec![Value::Integer(i32::MIN), Value::Integer(i32::MAX)]),
15341            (LogicalType::UInteger, vec![Value::UInteger(0), Value::UInteger(u32::MAX)]),
15342            (LogicalType::BigInt, vec![Value::BigInt(i64::MIN), Value::BigInt(i64::MAX)]),
15343            (LogicalType::UBigInt, vec![Value::UBigInt(0), Value::UBigInt(u64::MAX)]),
15344        ];
15345        let fields = columns
15346            .iter()
15347            .enumerate()
15348            .map(|(at, (ty, _))| Field::required(format!("c{at}"), ty.clone()))
15349            .collect::<Vec<_>>();
15350        let vectors = columns
15351            .iter()
15352            .map(|(ty, values)| Vector::from_values(ty.clone(), values).expect("a vector"))
15353            .collect::<Vec<_>>();
15354        let mut writer = Writer::create(&path, "widths", fields).expect("new file");
15355        writer.append(&Chunk::new(vectors).expect("matching rows")).expect("one stripe");
15356        writer.finish().expect("commit");
15357
15358        let reader = Reader::open(&path).expect("reopen from disk");
15359        let wanted = (0..columns.len()).collect::<Vec<_>>();
15360        let read = reader.read(0, &wanted).expect("every column");
15361        assert_eq!(read.len(), 2);
15362        // row at a time: each column has its own type and its own pair of extremes.
15363        for (at, (ty, values)) in columns.iter().enumerate() {
15364            assert_eq!(read.value_at(0, at), values[0], "the low end of {ty}");
15365            assert_eq!(read.value_at(1, at), values[1], "the high end of {ty}");
15366        }
15367        fs::remove_file(path).expect("remove scratch file");
15368    }
15369
15370    /// The rest of the fixed width types, and the byte strings, written and read back.
15371    ///
15372    /// The extremes again, and for a float that means more than the ends of the range. Negative
15373    /// zero and a NaN are the two values that go through an encoder unnoticed and come back
15374    /// different, so they are here on purpose, and the NaN is compared by its bits rather than by
15375    /// `==`, which a NaN fails against itself.
15376    ///
15377    /// A blob is here beside them because it is the same round trip asked of bytes that are not
15378    /// text. The value in it is not UTF-8, so a path that reads a payload as a string on the way
15379    /// past turns this test red rather than turning a user's column into nulls.
15380    #[test]
15381    fn every_other_type_the_format_knows_round_trips_through_a_page() {
15382        let path = path("other-types");
15383        let columns = [
15384            (LogicalType::Float, vec![Value::Float(f32::MIN), Value::Float(-0.0)]),
15385            (LogicalType::Double, vec![Value::Double(f64::MIN), Value::Double(f64::MAX)]),
15386            (LogicalType::HugeInt, vec![Value::HugeInt(i128::MIN), Value::HugeInt(i128::MAX)]),
15387            (LogicalType::UHugeInt, vec![Value::UHugeInt(0), Value::UHugeInt(u128::MAX)]),
15388            (LogicalType::Time, vec![Value::Time(0), Value::Time(86_399_999_999)]),
15389            (LogicalType::TimeTz, vec![Value::TimeTz(-50_400_000_000), Value::TimeTz(0)]),
15390            (
15391                LogicalType::TimestampTz,
15392                vec![Value::TimestampTz(i64::MIN + 1), Value::TimestampTz(i64::MAX)],
15393            ),
15394            (
15395                LogicalType::Interval,
15396                vec![
15397                    Value::Interval { months: i32::MIN, days: i32::MAX, micros: i64::MIN },
15398                    Value::Interval { months: 13, days: -1, micros: 1 },
15399                ],
15400            ),
15401            (
15402                LogicalType::Blob,
15403                vec![Value::Blob(vec![0, 0xff, 0x80, 0xfe]), Value::Blob(Vec::new())],
15404            ),
15405        ];
15406        let fields = columns
15407            .iter()
15408            .enumerate()
15409            .map(|(at, (ty, _))| Field::required(format!("c{at}"), ty.clone()))
15410            .collect::<Vec<_>>();
15411        let vectors = columns
15412            .iter()
15413            .map(|(ty, values)| Vector::from_values(ty.clone(), values).expect("a vector"))
15414            .collect::<Vec<_>>();
15415        let mut writer = Writer::create(&path, "others", fields).expect("new file");
15416        writer.append(&Chunk::new(vectors).expect("matching rows")).expect("one stripe");
15417        writer.finish().expect("commit");
15418
15419        let reader = Reader::open(&path).expect("reopen from disk");
15420        let wanted = (0..columns.len()).collect::<Vec<_>>();
15421        let read = reader.read(0, &wanted).expect("every column");
15422        assert_eq!(read.len(), 2);
15423        for (at, (ty, values)) in columns.iter().enumerate() {
15424            assert_eq!(read.value_at(0, at), values[0], "the low end of {ty}");
15425            assert_eq!(read.value_at(1, at), values[1], "the high end of {ty}");
15426        }
15427        // A float keeps its sign through a zero, which `==` says nothing about because negative
15428        // zero and zero compare equal.
15429        let Value::Float(zero) = read.value_at(1, 0) else { panic!("a float stays a float") };
15430        assert!(zero.is_sign_negative(), "a negative zero came back as {zero}");
15431
15432        fs::remove_file(path).expect("remove scratch file");
15433    }
15434
15435    /// A NaN is still a NaN after a trip through a page.
15436    ///
15437    /// Apart from the other floats because it cannot be asserted the same way. A NaN is not equal
15438    /// to itself, so a comparison against the value that was written passes for every NaN and for
15439    /// nothing else, which is the one assertion that would not catch a page that lost it.
15440    #[test]
15441    fn a_nan_survives_being_written_down() {
15442        let path = path("nan");
15443        let nan = Vector::from_values(LogicalType::Double, &[Value::Double(f64::NAN)])
15444            .expect("a NaN vector");
15445        let mut writer =
15446            Writer::create(&path, "nan", vec![Field::required("d", LogicalType::Double)])
15447                .expect("new file");
15448        writer.append(&Chunk::new(vec![nan]).expect("one column")).expect("one stripe");
15449        writer.finish().expect("commit");
15450        let read = Reader::open(&path).expect("reopen").read(0, &[0]).expect("the column");
15451        let Value::Double(back) = read.value_at(0, 0) else { panic!("a double stays a double") };
15452        assert!(back.is_nan(), "a NaN came back as {back}");
15453        fs::remove_file(path).expect("remove scratch file");
15454    }
15455
15456    /// A uuid and a bit string, which have no `Value` arm of their own and are checked as bits.
15457    ///
15458    /// A uuid is the 128 bit lane and a bit string is bytes, and neither of them reads back as
15459    /// anything in `Value` today, so asking for a value here would compare two nulls and pass
15460    /// whatever the file held. The data underneath is what the storage promise is about, so that is
15461    /// what this reads.
15462    #[test]
15463    fn a_uuid_and_a_bit_string_come_back_as_the_bits_that_went_in() {
15464        let path = path("uuid-and-bit");
15465        let uuids = vec![0_i128, i128::MIN, -1];
15466        let mut bits = StringColumn::new();
15467        for value in [&b"\x02\xff"[..], &b""[..], &b"\x00\x01\x02\x03\x04\x05"[..]] {
15468            bits.push_bytes(value);
15469        }
15470        let expected = bits.clone();
15471        let fields =
15472            vec![Field::required("u", LogicalType::Uuid), Field::required("b", LogicalType::Bit)];
15473        let vectors = vec![
15474            Vector::flat(LogicalType::Uuid, Data::Int128(uuids.clone().into())).expect("uuids"),
15475            Vector::flat(LogicalType::Bit, Data::Varlen(bits)).expect("bit strings"),
15476        ];
15477        let mut writer = Writer::create(&path, "ids", fields).expect("new file");
15478        writer.append(&Chunk::new(vectors).expect("matching rows")).expect("one stripe");
15479        writer.finish().expect("commit");
15480
15481        let reader = Reader::open(&path).expect("reopen from disk");
15482        let read = reader.read(0, &[0, 1]).expect("both columns").flatten().expect("flat");
15483        let Some(Data::Int128(back)) = read.column(0).expect("the uuids").data() else {
15484            panic!("a uuid column is the 128 bit lane")
15485        };
15486        assert_eq!(back.as_slice(), uuids.as_slice());
15487        let Some(Data::Varlen(back)) = read.column(1).expect("the bits").data() else {
15488            panic!("a bit column is bytes")
15489        };
15490        for row in 0..expected.len() {
15491            assert_eq!(back.bytes(row), expected.bytes(row), "row {row} of the bit column");
15492        }
15493        fs::remove_file(path).expect("remove scratch file");
15494    }
15495
15496    /// Counting a run at once has to leave the candidate table exactly where counting its rows one
15497    /// at a time would, including once the table is full and a run is turned away row by row.
15498    #[test]
15499    fn a_run_counted_at_once_leaves_the_candidates_a_row_at_a_time_would() {
15500        let mut rows: Vec<Option<u64>> = Vec::new();
15501        let mut state = 0x2545_f491_4f6c_dd1d_u64;
15502        for index in 0..400_000_u64 {
15503            state ^= state << 13;
15504            state ^= state >> 7;
15505            state ^= state << 17;
15506            let times = 1 + (state % 7) as usize;
15507            let bits = match state % 11 {
15508                0 => None,
15509                1..=3 => Some(state % 16),
15510                _ => Some(index.wrapping_mul(0x9e37_79b9_7f4a_7c15)),
15511            };
15512            rows.extend(std::iter::repeat_n(bits, times));
15513        }
15514        let mut by_row = Candidates::default();
15515        for &bits in &rows {
15516            by_row.add(bits, 1);
15517        }
15518        let mut by_run = Candidates::default();
15519        let mut run = Run::default();
15520        let mut runs = 0_usize;
15521        for &bits in &rows {
15522            if let Some((bits, times)) = run.push(bits) {
15523                by_run.add(bits, times);
15524                runs += 1;
15525            }
15526        }
15527        if let Some((bits, times)) = run.take() {
15528            by_run.add(bits, times);
15529        }
15530        assert!(runs < rows.len() / 2, "the rows came in runs");
15531        assert!(by_row.decrements > 0, "the table filled and turned values away");
15532        assert_eq!(sorted_candidates(&by_run), sorted_candidates(&by_row));
15533        assert_eq!(by_run.nulls, by_row.nulls);
15534        assert_eq!(by_run.decrements, by_row.decrements);
15535    }
15536
15537    fn sorted_candidates(candidates: &Candidates) -> Vec<(u64, u32)> {
15538        let mut pairs = candidates.pairs().collect::<Vec<_>>();
15539        pairs.sort_unstable();
15540        assert_eq!(pairs.len(), candidates.held, "the count of held slots drifted");
15541        pairs
15542    }
15543
15544    /// The Misra-Gries table as it was written over a `HashMap`, kept as the oracle the open
15545    /// addressed one has to agree with.
15546    #[derive(Default)]
15547    struct MapCandidates {
15548        counts: HashMap<u64, u32>,
15549        nulls: u32,
15550        decrements: u64,
15551    }
15552
15553    impl MapCandidates {
15554        fn add(&mut self, bits: Option<u64>, mut times: u32) {
15555            while times > 0 {
15556                let held = match bits {
15557                    Some(bits) => self.counts.get_mut(&bits),
15558                    None if self.nulls != 0 => Some(&mut self.nulls),
15559                    None => None,
15560                };
15561                if let Some(count) = held {
15562                    *count = count.saturating_add(times);
15563                    return;
15564                }
15565                if self.counts.len() + usize::from(self.nulls != 0) < FREQUENCY_CANDIDATES {
15566                    match bits {
15567                        Some(bits) => {
15568                            self.counts.insert(bits, times);
15569                        }
15570                        None => self.nulls = times,
15571                    }
15572                    return;
15573                }
15574                self.counts.retain(|_, count| {
15575                    *count -= 1;
15576                    *count != 0
15577                });
15578                self.nulls = self.nulls.saturating_sub(1);
15579                self.decrements = self.decrements.saturating_add(1);
15580                times -= 1;
15581            }
15582        }
15583    }
15584
15585    /// Near unique values, a few heavy ones, nulls, and runs, through enough rows that the table
15586    /// fills, grows through every size and is decremented many times over. Both tables have to hold
15587    /// the same candidates with the same counts at the end, and at points along the way.
15588    #[test]
15589    fn the_open_addressed_candidates_agree_with_the_map_they_replaced() {
15590        for seed in [0x2545_f491_4f6c_dd1d_u64, 0x9e37_79b9_7f4a_7c15, 7] {
15591            let mut table = Candidates::default();
15592            let mut oracle = MapCandidates::default();
15593            let mut state = seed;
15594            for index in 0..300_000_u64 {
15595                state ^= state << 13;
15596                state ^= state >> 7;
15597                state ^= state << 17;
15598                let bits = match state % 13 {
15599                    0 => None,
15600                    1..=4 => Some(state % 40),
15601                    5 => Some((index % 1000) * 1_000_000),
15602                    _ => Some(state),
15603                };
15604                let times = 1 + (state >> 60) as u32 % 3;
15605                table.add(bits, times);
15606                oracle.add(bits, times);
15607                if index % 50_000 == 0 {
15608                    let mut expected =
15609                        oracle.counts.iter().map(|(&b, &c)| (b, c)).collect::<Vec<_>>();
15610                    expected.sort_unstable();
15611                    assert_eq!(sorted_candidates(&table), expected, "seed {seed} row {index}");
15612                }
15613            }
15614            let mut expected = oracle.counts.iter().map(|(&b, &c)| (b, c)).collect::<Vec<_>>();
15615            expected.sort_unstable();
15616            assert_eq!(sorted_candidates(&table), expected, "seed {seed}");
15617            assert_eq!(table.nulls, oracle.nulls, "seed {seed}");
15618            assert_eq!(table.decrements, oracle.decrements, "seed {seed}");
15619            assert!(table.decrements > 0, "seed {seed} never filled the table");
15620            for &(bits, _) in &expected {
15621                assert!(table.position(bits).is_some(), "seed {seed} lost {bits}");
15622            }
15623        }
15624    }
15625
15626    #[test]
15627    fn numeric_frequency_candidates_keep_bounded_row_ordinals() {
15628        let path = path("frequency-ordinals");
15629        let mut writer =
15630            Writer::create(&path, "items", vec![Field::required("id", LogicalType::BigInt)])
15631                .expect("new file");
15632        let mut values = Vec::new();
15633        for leader in 0..10_i64 {
15634            values.extend(std::iter::repeat_n(leader, 100));
15635        }
15636        values.extend(1_000_i64..41_000);
15637        for part in values.chunks(1_024) {
15638            let vector = Vector::flat(LogicalType::BigInt, Data::Int64(part.to_vec().into()))
15639                .expect("big integers");
15640            writer.append(&Chunk::new(vec![vector]).expect("one column")).expect("one stripe");
15641        }
15642        writer.finish().expect("commit");
15643
15644        let reader = Reader::open(&path).expect("reopen from disk");
15645        let occurrences =
15646            reader.frequency_occurrences(0).expect("valid metadata").expect("bounded ordinals");
15647        assert!(occurrences.omitted_max < 100);
15648        assert!(occurrences.ordinals.len() <= FREQUENCY_ORDINALS);
15649        assert_eq!(occurrences.anchor_indices.len(), occurrences.ordinals.len());
15650        assert!(occurrences.ordinals.windows(2).all(|pair| pair[0] < pair[1]));
15651        assert_eq!(&occurrences.ordinals[..1_000], &(0_u64..1_000).collect::<Vec<_>>());
15652        assert_eq!(
15653            &occurrences.anchor_indices[..1_000]
15654                .iter()
15655                .map(|&entry| occurrences.anchors[entry as usize].clone())
15656                .collect::<Vec<_>>(),
15657            &(0_i64..10)
15658                .flat_map(|leader| std::iter::repeat_n(Value::BigInt(leader), 100))
15659                .collect::<Vec<_>>()
15660        );
15661        fs::remove_file(path).expect("remove scratch file");
15662    }
15663
15664    #[test]
15665    fn numeric_frequencies_count_nulls_and_values_past_the_top_of_bigint() {
15666        // Ten leaders, then more unique values than the candidate table holds, so the first pass
15667        // has to decrement and the counts come from the recount. The unsigned leaders sit above
15668        // `i64::MAX`, where reading the bits as signed would give a different value, and the signed
15669        // ones are negative, where reading them as unsigned would.
15670        let path = path("frequency-bits");
15671        let mut writer = Writer::create(
15672            &path,
15673            "items",
15674            vec![Field::new("u", LogicalType::UBigInt), Field::new("s", LogicalType::BigInt)],
15675        )
15676        .expect("new file");
15677        let mut rows = Vec::new();
15678        let mut leaders = Vec::new();
15679        for leader in 0..10_u64 {
15680            let count = 300 - leader * 10;
15681            let (unsigned, signed) = if leader == 0 {
15682                (Value::Null, Value::Null)
15683            } else {
15684                (Value::UBigInt(u64::MAX - leader), Value::BigInt(-(leader as i64)))
15685            };
15686            rows.extend(std::iter::repeat_n((unsigned.clone(), signed.clone()), count as usize));
15687            leaders.push(((unsigned, count), (signed, count)));
15688        }
15689        rows.extend((1_000..41_000_u64).map(|id| (Value::UBigInt(id), Value::BigInt(id as i64))));
15690        for part in rows.chunks(1_024) {
15691            let unsigned = part.iter().map(|(value, _)| value.clone()).collect::<Vec<_>>();
15692            let signed = part.iter().map(|(_, value)| value.clone()).collect::<Vec<_>>();
15693            let chunk = Chunk::new(vec![
15694                Vector::from_values(LogicalType::UBigInt, &unsigned).expect("unsigned"),
15695                Vector::from_values(LogicalType::BigInt, &signed).expect("signed"),
15696            ])
15697            .expect("matching columns");
15698            writer.append(&chunk).expect("rows");
15699        }
15700        writer.finish().expect("commit");
15701
15702        let reader = Reader::open(&path).expect("reopen from disk");
15703        for column in 0..2 {
15704            let prefix =
15705                reader.frequency_prefix(column).expect("valid metadata").expect("a synopsis");
15706            let wanted = leaders
15707                .iter()
15708                .map(|(unsigned, signed)| if column == 0 { unsigned } else { signed })
15709                .cloned()
15710                .collect::<Vec<_>>();
15711            assert_eq!(&prefix.entries[..10], &wanted[..], "column {column}");
15712            assert!(prefix.omitted_max < 210, "column {column}");
15713            assert_eq!(
15714                reader.distinct_values(column).expect("valid metadata"),
15715                Some(9 + 40_000),
15716                "column {column}"
15717            );
15718        }
15719        fs::remove_file(path).expect("remove scratch file");
15720    }
15721
15722    #[test]
15723    fn a_narrow_column_takes_its_frequencies_from_the_tally_and_they_match_the_rows() {
15724        // Every column here has fewer distinct values than the tally holds, so the close takes its
15725        // counts from the gather rather than reading the pages back. The types are the ones whose
15726        // bits could come out wrong on that road: a negative tiny integer that has to be sign
15727        // extended, an unsigned one past the top of `INTEGER`, a date and a timestamp. A null every
15728        // thirteenth row checks that the nulls come from the pass and not from the list.
15729        let path = path("frequency-tally");
15730        let types = [
15731            LogicalType::TinyInt,
15732            LogicalType::UInteger,
15733            LogicalType::Date,
15734            LogicalType::Timestamp,
15735        ];
15736        let value = |ty: &LogicalType, at: i64| match ty {
15737            LogicalType::TinyInt => Value::TinyInt((at % 250 - 125) as i8),
15738            LogicalType::UInteger => Value::UInteger(u32::MAX - at as u32),
15739            LogicalType::Date => Value::Date(19_000 - at as i32),
15740            _ => Value::Timestamp(1_700_000_000_000_000 - at * 1_000_003),
15741        };
15742        let fields = types
15743            .iter()
15744            .enumerate()
15745            .map(|(at, ty)| Field::new(format!("c{at}"), ty.clone()))
15746            .collect::<Vec<_>>();
15747        let mut writer = Writer::create(&path, "items", fields).expect("new file");
15748        let mut rows = Vec::new();
15749        for at in 0..250_i64 {
15750            for _ in 0..=(at % 37) {
15751                rows.push(if rows.len() % 13 == 0 { None } else { Some(at) });
15752            }
15753        }
15754        for part in rows.chunks(1_000) {
15755            let columns = types
15756                .iter()
15757                .map(|ty| {
15758                    let values = part
15759                        .iter()
15760                        .map(|row| row.map_or(Value::Null, |at| value(ty, at)))
15761                        .collect::<Vec<_>>();
15762                    Vector::from_values(ty.clone(), &values).expect("a column")
15763                })
15764                .collect();
15765            writer.append(&Chunk::new(columns).expect("matching columns")).expect("rows");
15766        }
15767        writer.finish().expect("commit");
15768
15769        let reader = Reader::open(&path).expect("reopen from disk");
15770        for (column, ty) in types.iter().enumerate() {
15771            let mut counts = HashMap::<Option<i64>, u64>::new();
15772            for row in &rows {
15773                *counts.entry(*row).or_default() += 1;
15774            }
15775            let wanted = counts
15776                .into_iter()
15777                .map(|(row, count)| (row.map_or(Value::Null, |at| value(ty, at)), count))
15778                .collect::<Vec<_>>();
15779            let prefix =
15780                reader.frequency_prefix(column).expect("valid metadata").expect("a synopsis");
15781            assert_eq!(prefix.entries.len(), 2, "column {column}");
15782            assert!(prefix.omitted_max > 0, "column {column}");
15783            for (value, count) in &prefix.entries {
15784                let held =
15785                    wanted.iter().find(|(wanted, _)| wanted == value).map(|(_, count)| count);
15786                assert_eq!(held, Some(count), "column {column} value {value:?}");
15787            }
15788            assert!(prefix.entries.windows(2).all(|pair| pair[0].1 >= pair[1].1));
15789            assert_eq!(
15790                reader.distinct_values(column).expect("valid metadata"),
15791                Some(wanted.len() as u64 - 1),
15792                "column {column}"
15793            );
15794        }
15795        fs::remove_file(path).expect("remove scratch file");
15796    }
15797
15798    #[test]
15799    fn distinct_counts_are_exact_either_side_of_a_full_candidate_table() {
15800        // The count comes from the candidate table while it has room and from the set once it
15801        // fills, so the sizes around the fill, with and without a null taking a place, are where a
15802        // value could be counted twice or missed. Zero is in every column because the set keeps it
15803        // apart from the other values, and every value comes back later to be counted again.
15804        let edge = FREQUENCY_CANDIDATES as i64;
15805        for distinct in [0, 1, 7, edge - 2, edge - 1, edge, edge + 1, edge + 2, 3 * edge] {
15806            for with_null in [false, true] {
15807                let path = path("distinct-edge");
15808                let mut writer =
15809                    Writer::create(&path, "items", vec![Field::new("id", LogicalType::BigInt)])
15810                        .expect("new file");
15811                let mut values = Vec::new();
15812                for round in 0..2 {
15813                    for value in 0..distinct {
15814                        let repeat = if round == 0 { 1 + (value % 3) as usize } else { 1 };
15815                        values.extend(std::iter::repeat_n(
15816                            Value::BigInt(value * 7_919 % distinct),
15817                            repeat,
15818                        ));
15819                        if with_null && value % 1_000 == 0 {
15820                            values.push(Value::Null);
15821                        }
15822                    }
15823                }
15824                if with_null {
15825                    values.push(Value::Null);
15826                }
15827                for part in values.chunks(1_024) {
15828                    let chunk = Chunk::new(vec![
15829                        Vector::from_values(LogicalType::BigInt, part).expect("ids"),
15830                    ])
15831                    .expect("one column");
15832                    writer.append(&chunk).expect("rows");
15833                }
15834                writer.finish().expect("commit");
15835                let reader = Reader::open(&path).expect("reopen from disk");
15836                assert_eq!(
15837                    reader.distinct_values(0).expect("valid metadata"),
15838                    Some(distinct as u64),
15839                    "{distinct} values, null {with_null}"
15840                );
15841                fs::remove_file(path).expect("remove scratch file");
15842            }
15843        }
15844    }
15845
15846    #[test]
15847    fn narrow_nonzero_count_matches_the_full_reader_across_stripes() {
15848        let path = path("quick-nonzero");
15849        let mut writer = Writer::create(
15850            &path,
15851            "items",
15852            vec![Field::new("label", LogicalType::Varchar), Field::new("id", LogicalType::Integer)],
15853        )
15854        .expect("create");
15855        for ids in [
15856            &[Value::Integer(0), Value::Null, Value::Integer(3)][..],
15857            &[Value::Integer(0), Value::Integer(7), Value::Null][..],
15858        ] {
15859            let labels = vec![Value::Varchar("same".into()); ids.len()];
15860            writer
15861                .append(
15862                    &Chunk::new(vec![
15863                        Vector::from_values(LogicalType::Varchar, &labels).expect("labels"),
15864                        Vector::from_values(LogicalType::Integer, ids).expect("ids"),
15865                    ])
15866                    .expect("chunk"),
15867                )
15868                .expect("append");
15869        }
15870        writer.finish().expect("finish");
15871        let catalog = Catalog::open(&path).expect("catalog");
15872        assert_eq!(catalog.entries[0].nonzero, vec![None, None]);
15873        assert_eq!(catalog.entries[0].aggregates, vec![None, Some((10, 4))]);
15874        assert_eq!(catalog.entries[0].distincts, vec![Some(1), Some(3)]);
15875        assert_eq!(catalog.exact_numeric_frequencies("items", 1).expect("frequencies"), None);
15876        let prefix = catalog
15877            .table("items")
15878            .expect("reader")
15879            .frequency_prefix(1)
15880            .expect("valid metadata")
15881            .expect("partial frequencies");
15882        assert_eq!(prefix.entries, vec![(Value::Null, 2), (Value::Integer(0), 2)]);
15883        assert_eq!(prefix.omitted_max, 1);
15884        assert_eq!(catalog.distinct_count("items", 1).expect("distinct count"), Some(3));
15885        assert_eq!(
15886            catalog.integer_extremes("items", 1).expect("extremes"),
15887            Some(IntegerExtremes::Values { low: 0, high: 7 })
15888        );
15889        assert_eq!(
15890            catalog.aggregate_sums("items", &[1]).expect("catalog sums"),
15891            Some(CertifiedSums { columns: vec![(10, 4)], rows: 6 })
15892        );
15893        assert_eq!(catalog.nonzero_count("items", 1).expect("quick count"), Some(2));
15894        let mut legacy = catalog.clone();
15895        Arc::make_mut(&mut legacy.entries)[0].nonzero[1] = Some(999);
15896        assert_eq!(legacy.nonzero_count("items", 1).expect("ignore legacy count"), Some(2));
15897        Arc::make_mut(&mut legacy.entries)[0].frequencies[1] = None;
15898        assert_eq!(legacy.nonzero_count("items", 1).expect("directory fallback"), Some(2));
15899        Writer::certify_counts(&path).expect("recertify");
15900        assert_eq!(
15901            Catalog::open(&path).expect("reopen").nonzero_count("items", 1).expect("count"),
15902            Some(2)
15903        );
15904        assert_eq!(
15905            Catalog::open(&path).expect("reopen").aggregate_sums("items", &[1]).expect("sums"),
15906            Some(CertifiedSums { columns: vec![(10, 4)], rows: 6 })
15907        );
15908        assert_eq!(
15909            Catalog::open(&path).expect("reopen").distinct_count("items", 1).expect("distinct"),
15910            Some(3)
15911        );
15912        assert_eq!(
15913            Catalog::open(&path).expect("reopen").integer_extremes("items", 1).expect("ends"),
15914            Some(IntegerExtremes::Values { low: 0, high: 7 })
15915        );
15916        assert_eq!(
15917            Catalog::open(&path)
15918                .expect("reopen")
15919                .exact_numeric_frequencies("items", 1)
15920                .expect("frequencies"),
15921            None
15922        );
15923        assert_eq!(catalog.table("items").expect("reader").null_count(1).expect("nulls"), 2);
15924        fs::remove_file(path).expect("remove scratch file");
15925    }
15926
15927    #[test]
15928    fn numeric_string_pair_leaders_are_certified_in_the_directory() {
15929        let path = path("pair-frequencies");
15930        let mut pairs = Vec::new();
15931        pairs.extend(std::iter::repeat_n((1_i64, "alpha".to_string()), 100));
15932        pairs.extend(std::iter::repeat_n((1_i64, "beta".to_string()), 50));
15933        pairs.extend(std::iter::repeat_n((2_i64, "gamma".to_string()), 40));
15934        pairs.extend((1_000_i64..1_600).map(|id| (id, format!("tail {id}"))));
15935        let mut writer = Writer::create(
15936            &path,
15937            "items",
15938            vec![
15939                Field::required("id", LogicalType::BigInt),
15940                Field::required("phrase", LogicalType::Varchar),
15941            ],
15942        )
15943        .expect("new file");
15944        for part in pairs.chunks(1_024) {
15945            let ids = part.iter().map(|(id, _)| Value::BigInt(*id)).collect::<Vec<_>>();
15946            let phrases =
15947                part.iter().map(|(_, phrase)| Value::Varchar(phrase.clone())).collect::<Vec<_>>();
15948            writer
15949                .append(
15950                    &Chunk::new(vec![
15951                        Vector::from_values(LogicalType::BigInt, &ids).expect("ids"),
15952                        Vector::from_values(LogicalType::Varchar, &phrases).expect("phrases"),
15953                    ])
15954                    .expect("matching columns"),
15955                )
15956                .expect("rows");
15957        }
15958        writer.finish().expect("commit");
15959
15960        let reader = Reader::open(&path).expect("reopen from disk");
15961        assert!(
15962            reader.table.pair_frequencies.is_empty(),
15963            "no query-specific pair result is stored"
15964        );
15965        fs::remove_file(path).expect("remove scratch file");
15966    }
15967
15968    #[test]
15969    fn legacy_group_answers_are_ignored() {
15970        let path = path("legacy-group-answers");
15971        let mut writer = Writer::create(
15972            &path,
15973            "items",
15974            vec![
15975                Field::required("id", LogicalType::BigInt),
15976                Field::required("text", LogicalType::Varchar),
15977            ],
15978        )
15979        .expect("new file");
15980        writer
15981            .append(
15982                &Chunk::new(vec![
15983                    Vector::from_values(LogicalType::BigInt, &[Value::BigInt(1)]).expect("id"),
15984                    Vector::from_values(LogicalType::Varchar, &[Value::Varchar("x".into())])
15985                        .expect("text"),
15986                ])
15987                .expect("row"),
15988            )
15989            .expect("append");
15990        writer.finish().expect("commit");
15991        let mut reader = Reader::open(&path).expect("reopen");
15992        let table = Arc::make_mut(&mut reader.table);
15993        table.pair_frequencies.push(PairFrequencySummary {
15994            first: 0,
15995            second: 1,
15996            entries: vec![PairFrequencyEntry { first_entry: 0, second: Some(0), count: 999 }],
15997            omitted_max: 0,
15998        });
15999        table.host_groups = Some(host::HostSummary {
16000            column: 1,
16001            omitted_max: 0,
16002            entries: vec![host::HostEntry {
16003                host: "fake.test".into(),
16004                count: 999,
16005                bytes_sum: 999,
16006                minimum: "x".into(),
16007            }],
16008        });
16009        assert_eq!(reader.top_pair_frequencies(0, 1, 1).expect("legacy pair"), None);
16010        assert_eq!(reader.host_groups(1, 1).expect("legacy host"), None);
16011        fs::remove_file(path).expect("remove scratch file");
16012    }
16013
16014    /// The bug this is here for cost a 43 GB ClickBench table and an hour of reloading it. The
16015    /// format went from 11 to 12, every binary built after that said "magic or major version is
16016    /// unsupported" about the file, and there was no way to tell from the message whether the path
16017    /// was wrong, the file was truncated, or it was ours and simply older. The number this build
16018    /// wants is the whole answer and it was the one thing the message did not carry.
16019    #[test]
16020    fn a_file_from_another_format_says_which_format_it_is() {
16021        let older = path("older-format");
16022        let mut writer =
16023            Writer::create(&older, "items", vec![Field::new("id", LogicalType::Integer)])
16024                .expect("new file");
16025        let chunk = Chunk::new(vec![
16026            Vector::flat(LogicalType::Integer, Data::Int32(vec![1, 2, 3].into()))
16027                .expect("integers"),
16028        ])
16029        .expect("chunk");
16030        writer.append(&chunk).expect("page written");
16031        writer.finish().expect("commit");
16032
16033        // A format below the whole readable set, rather than `FORMAT - 1`, because the set has
16034        // more than one member now: format 22 is deliberately still readable, so the version that
16035        // has to be refused is the one under the oldest one accepted.
16036        let unreadable =
16037            READABLE.iter().copied().min().expect("at least one format is readable") - 1;
16038        let mut file = OpenOptions::new().write(true).open(&older).expect("open for the header");
16039        file.seek(SeekFrom::Start(8)).expect("the version follows the magic");
16040        file.write_all(&unreadable.to_le_bytes()).expect("write an older version");
16041        drop(file);
16042        let complaint = Reader::open(&older).expect_err("an older format is refused").to_string();
16043        assert!(complaint.contains(&format!("format {unreadable}")), "{complaint}");
16044        assert!(complaint.contains(&format!("format {FORMAT}")), "{complaint}");
16045
16046        let mut file = OpenOptions::new().write(true).open(&older).expect("open for the header");
16047        file.seek(SeekFrom::Start(0)).expect("the magic is first");
16048        file.write_all(b"NOTRUDB!").expect("write another engine's magic");
16049        drop(file);
16050        let complaint = Reader::open(&older).expect_err("a foreign file is refused").to_string();
16051        assert!(complaint.contains("magic"), "{complaint}");
16052        assert!(!complaint.contains("format"), "a version has nothing to do with it: {complaint}");
16053        fs::remove_file(older).expect("remove scratch file");
16054    }
16055
16056    #[test]
16057    fn an_unfinished_or_damaged_file_does_not_answer_with_partial_rows() {
16058        let unfinished = path("unfinished");
16059        let mut writer =
16060            Writer::create(&unfinished, "items", vec![Field::new("id", LogicalType::Integer)])
16061                .expect("new file");
16062        let chunk = Chunk::new(vec![
16063            Vector::flat(LogicalType::Integer, Data::Int32(vec![1, 2, 3].into()))
16064                .expect("integers"),
16065        ])
16066        .expect("chunk");
16067        writer.append(&chunk).expect("page written");
16068        drop(writer);
16069        assert!(Reader::open(&unfinished).is_err(), "no directory was committed");
16070        fs::remove_file(unfinished).expect("remove scratch file");
16071
16072        let damaged = path("damaged");
16073        let mut writer =
16074            Writer::create(&damaged, "items", vec![Field::new("id", LogicalType::Integer)])
16075                .expect("new file");
16076        writer.append(&chunk).expect("page written");
16077        writer.finish().expect("commit");
16078        let reader = Reader::open(&damaged).expect("valid directory");
16079        let mut file =
16080            OpenOptions::new().write(true).open(&damaged).expect("open for a damaged page");
16081        file.seek(SeekFrom::Start(HEADER + 1)).expect("inside first page");
16082        file.write_all(&[255]).expect("damage one byte");
16083        assert!(reader.read(0, &[0]).is_err(), "page checksum rejects corruption");
16084        fs::remove_file(damaged).expect("remove scratch file");
16085    }
16086
16087    #[test]
16088    fn damaged_lazy_dictionary_payload_is_an_error() {
16089        let path = path("damaged-dictionary");
16090        let mut writer = Writer::create(
16091            &path,
16092            "items",
16093            vec![
16094                Field::required("id", LogicalType::Integer),
16095                Field::new("text", LogicalType::Varchar),
16096            ],
16097        )
16098        .expect("new file");
16099        writer.append(&sample()).expect("stripe written");
16100        writer.finish().expect("commit");
16101
16102        let reader = Reader::open(&path).expect("valid directory");
16103        let dictionary = reader.table.dictionaries[1].expect("string dictionary page");
16104        // Read the count out of the page rather than writing it here, so that adding something
16105        // else to the index does not silently turn this into a test that damages the index.
16106        let mut header = [0; DICTIONARY_HEADER];
16107        read_at(&reader.file, dictionary.offset, &mut header).expect("dictionary header");
16108        // The first block's start is the first word after the offsets, since the blocks are written
16109        // during the load and are wherever the writer was when each was encoded.
16110        let count = u32::from_le_bytes(header[0..4].try_into().expect("four bytes")) as usize;
16111        let width = u32::from_le_bytes(header[12..16].try_into().expect("four bytes"));
16112        assert_ne!(width & DICTIONARY_SCATTERED, 0, "the blocks say where they are");
16113        let bits = (width & !DICTIONARY_FLAGS) as usize;
16114        let mut start = [0; 8];
16115        let at = dictionary.offset + (DICTIONARY_HEADER + offset_bytes(count, bits)) as u64;
16116        read_at(&reader.file, at, &mut start).expect("the first block's start");
16117        let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
16118        file.seek(SeekFrom::Start(u64::from_le_bytes(start))).expect("inside dictionary payload");
16119        file.write_all(&[255]).expect("damage dictionary payload");
16120
16121        let chunk = reader.read(0, &[1]).expect("code page and dictionary index remain valid");
16122        let error =
16123            chunk.validate_external().expect_err("payload corruption must reach the caller");
16124        assert!(error.message().contains("payload checksum differs"), "{error}");
16125        fs::remove_file(path).expect("remove scratch file");
16126    }
16127
16128    /// A column whose values are all different is written without a dictionary, and one whose
16129    /// values repeat keeps it.
16130    ///
16131    /// The two columns go in the same table and hold the same number of rows, so the only thing
16132    /// separating them is how much of the first stripe was a value it had not seen before. Both have
16133    /// to read back the values that were written, because the decision is about cost and nothing
16134    /// else. The file size is the other half of it: a column written without a dictionary goes
16135    /// through the string cascade instead, so dropping the dictionary must not turn into storing the
16136    /// column raw.
16137    #[test]
16138    fn a_column_of_all_different_values_is_written_without_a_dictionary() {
16139        let path = path("dictionary-decide");
16140        let rows = 20_000;
16141        // Long enough that storing it raw would show, and different in every row.
16142        let unique =
16143            |row: usize| format!("{row:09} a value that appears exactly once in the table");
16144        // The same values in the same shape, each one used forty times over.
16145        let repeated = |row: usize| unique(row / 40);
16146        let mut writer = Writer::create(
16147            &path,
16148            "items",
16149            vec![
16150                Field::required("unique", LogicalType::Varchar),
16151                Field::required("repeated", LogicalType::Varchar),
16152            ],
16153        )
16154        .expect("new file");
16155        for part in (0..rows).step_by(1_000) {
16156            let span = part..(part + 1_000).min(rows);
16157            let left = span.clone().map(|row| Value::Varchar(unique(row))).collect::<Vec<_>>();
16158            let right = span.map(|row| Value::Varchar(repeated(row))).collect::<Vec<_>>();
16159            writer
16160                .append(
16161                    &Chunk::new(vec![
16162                        Vector::from_values(LogicalType::Varchar, &left).expect("strings"),
16163                        Vector::from_values(LogicalType::Varchar, &right).expect("strings"),
16164                    ])
16165                    .expect("two columns"),
16166                )
16167                .expect("a part");
16168        }
16169        writer.finish().expect("commit");
16170
16171        let reader = Reader::open(&path).expect("reopen from disk");
16172        assert!(
16173            reader.table.dictionaries[0].is_none(),
16174            "a column with no repeats has nothing to say twice"
16175        );
16176        assert!(
16177            reader.table.dictionaries[1].is_some(),
16178            "a column whose values come round again keeps its dictionary"
16179        );
16180        let mut first = 0;
16181        for part in 0..reader.parts() {
16182            let chunk = reader.read(part, &[0, 1]).expect("a part");
16183            for row in 0..chunk.len() {
16184                assert_eq!(chunk.value_at(row, 0), Value::Varchar(unique(first + row)));
16185                assert_eq!(chunk.value_at(row, 1), Value::Varchar(repeated(first + row)));
16186            }
16187            first += chunk.len();
16188        }
16189        assert_eq!(first, rows, "every row was read back");
16190        let raw = (0..rows).map(|row| unique(row).len()).sum::<usize>();
16191        let size = fs::metadata(&path).expect("the file is there").len() as usize;
16192        assert!(size < raw, "a column without a dictionary is still encoded: {size} against {raw}");
16193        fs::remove_file(path).expect("remove scratch file");
16194    }
16195
16196    /// A payload of many blocks reads and checks every block of it.
16197    ///
16198    /// The test above has a dictionary of three values, which is one block, so it says nothing
16199    /// about a reader finding the right block among many. This one has thirty two thousand values,
16200    /// which is thirty two blocks, and it reads a value out of the first block and a value out of
16201    /// the last and then damages the last and asks for it again.
16202    ///
16203    /// Forty thousand rows over those thirty two thousand values, because a column the writer finds
16204    /// to be all distinct does not get a dictionary at all and there would be nothing here to test.
16205    /// Four rows in five holding a value the stripe has not seen before is a column that keeps one.
16206    /// The repeats are put at the front so that the values still arrive in order after them, which
16207    /// is what keeps the last part of the table on the last block of the payload.
16208    #[test]
16209    fn a_dictionary_over_many_blocks_checks_every_block_of_it() {
16210        let path = path("dictionary-blocks");
16211        let value = |row: usize| {
16212            let row = row.saturating_sub(8_000);
16213            format!("{row:07} a value long enough to be worth a payload block")
16214        };
16215        let parts = 40;
16216        let per_part = 1000;
16217        let mut writer =
16218            Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
16219                .expect("new file");
16220        for part in 0..parts {
16221            let values = (0..per_part)
16222                .map(|row| Value::Varchar(value(part * per_part + row)))
16223                .collect::<Vec<_>>();
16224            let chunk = Chunk::new(vec![
16225                Vector::from_values(LogicalType::Varchar, &values).expect("strings"),
16226            ])
16227            .expect("matching rows");
16228            writer.append(&chunk).expect("a part");
16229        }
16230        writer.finish().expect("commit");
16231
16232        let reader = Reader::open(&path).expect("reopen from disk");
16233        let dictionary = reader.table.dictionaries[0].expect("string dictionary page");
16234        assert!(
16235            parts * per_part > TEXT_PAYLOAD_VALUES * 4,
16236            "the dictionary has to be several blocks for this to be testing anything"
16237        );
16238        for part in [0, parts - 1] {
16239            let chunk = reader.read(part, &[0]).expect("a part");
16240            chunk.validate_external().expect("every payload block checks out");
16241            assert_eq!(chunk.value_at(0, 0), Value::Varchar(value(part * per_part)));
16242        }
16243
16244        // The last block is wherever the writer was when it was encoded, which the index says.
16245        let mut header = [0; DICTIONARY_HEADER];
16246        read_at(&reader.file, dictionary.offset, &mut header).expect("dictionary header");
16247        let count = u32::from_le_bytes(header[0..4].try_into().expect("four bytes")) as usize;
16248        let blocks = u32::from_le_bytes(header[8..12].try_into().expect("four bytes")) as usize;
16249        let width = u32::from_le_bytes(header[12..16].try_into().expect("four bytes"));
16250        let bits = (width & !DICTIONARY_FLAGS) as usize;
16251        let mut place = [0; 16];
16252        let at = DICTIONARY_HEADER + offset_bytes(count, bits) + (blocks - 1) * 16;
16253        read_at(&reader.file, dictionary.offset + at as u64, &mut place).expect("its place");
16254        let start = u64::from_le_bytes(place[..8].try_into().expect("eight bytes"));
16255        let length = u64::from_le_bytes(place[8..].try_into().expect("eight bytes"));
16256        let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
16257        file.seek(SeekFrom::Start(start + length - 4)).expect("the last bytes of the last block");
16258        file.write_all(&[255]).expect("damage the last payload block");
16259        let reader = Reader::open(&path).expect("the directory and the index are untouched");
16260        let chunk = reader.read(parts - 1, &[0]).expect("the code page remains valid");
16261        let error = chunk.validate_external().expect_err("the damage must reach the caller");
16262        assert!(error.message().contains("payload checksum differs"), "{error}");
16263        fs::remove_file(path).expect("remove scratch file");
16264    }
16265
16266    /// Values of different lengths read back where the offsets say they do.
16267    ///
16268    /// The offsets are packed at one width for the column, they are relative to the payload block a
16269    /// value lands in, and they go in runs of half a block, so there are two boundaries where the
16270    /// arithmetic could be off by one and neither shows up on values that are all the same length.
16271    /// This writes 5,000 values whose lengths cycle through a wide range and reads every one back,
16272    /// so the first value of a block, the last value of a run and the last value of a block are all
16273    /// covered several times over. An empty value is in the cycle because a zero length span is the
16274    /// case the reader short circuits.
16275    ///
16276    /// Six thousand rows over those 5,000 values, because a column the writer finds to be all
16277    /// distinct is written without a dictionary and then there are no packed offsets to be off by
16278    /// one in.
16279    #[test]
16280    fn values_of_different_lengths_read_back_out_of_packed_offsets() {
16281        let path = path("dictionary-offsets");
16282        let value = |row: usize| {
16283            let row = row % 5_000;
16284            if row % 511 == 3 { String::new() } else { "x".repeat(row % 97) + &format!("{row:05}") }
16285        };
16286        let rows = 6_000;
16287        let mut writer =
16288            Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
16289                .expect("new file");
16290        let values = (0..rows).map(|row| Value::Varchar(value(row))).collect::<Vec<_>>();
16291        for part in values.chunks(1_000) {
16292            let chunk =
16293                Chunk::new(vec![Vector::from_values(LogicalType::Varchar, part).expect("strings")])
16294                    .expect("matching rows");
16295            writer.append(&chunk).expect("a part");
16296        }
16297        writer.finish().expect("commit");
16298
16299        let reader = Reader::open(&path).expect("reopen from disk");
16300        assert!(
16301            rows > TEXT_PAYLOAD_VALUES * 4,
16302            "the dictionary has to be several blocks for this to be testing anything"
16303        );
16304        for part in 0..rows / 1_000 {
16305            let chunk = reader.read(part, &[0]).expect("a part");
16306            for row in 0..1_000 {
16307                let row = part * 1_000 + row;
16308                assert_eq!(
16309                    chunk.value_at(row % 1_000, 0),
16310                    Value::Varchar(value(row)),
16311                    "value {row}"
16312                );
16313            }
16314        }
16315        // The lengths a vector at a time, twice over, because the first pass is what makes the
16316        // table of ends worth building and the second is read out of the lengths worked out of it.
16317        for _ in 0..2 {
16318            for part in 0..rows / 1_000 {
16319                let chunk = reader.read(part, &[0]).expect("a part");
16320                let mut lens = vec![0_i64; 1_000];
16321                let column = chunk.column(0).expect("one column");
16322                assert!(column.try_bytes_lens(&mut lens).expect("lengths"), "a stored column");
16323                for (row, &len) in lens.iter().enumerate() {
16324                    let row = part * 1_000 + row;
16325                    assert_eq!(len as usize, value(row).len(), "the length of value {row}");
16326                }
16327            }
16328        }
16329        fs::remove_file(path).expect("remove scratch file");
16330    }
16331
16332    /// Lengths start again at every block, and ends that go backwards inside one give no table.
16333    #[test]
16334    fn lengths_restart_at_each_block_and_refuse_ends_that_go_backwards() {
16335        let mut ends: Vec<u32> = (1..=TEXT_PAYLOAD_VALUES as u32).map(|at| at * 2).collect();
16336        ends.extend([3, 3, 10]);
16337        let Some(Lengths::Narrow(lens)) = lengths_of(&ends) else { panic!("short ordered ends") };
16338        assert!(lens[..TEXT_PAYLOAD_VALUES].iter().all(|&len| len == 2));
16339        assert_eq!(&lens[TEXT_PAYLOAD_VALUES..], &[3, 0, 7]);
16340        // One value longer than sixteen bits keeps every length at four bytes.
16341        let long = [5, 70_005, 70_006];
16342        let Some(Lengths::Wide(lens)) = lengths_of(&long) else { panic!("long ordered ends") };
16343        assert_eq!(lens, [5, 70_000, 1]);
16344        let mut read = Vec::new();
16345        Lengths::Wide(lens).extend_at(&[1, 9, 0], &mut read);
16346        assert_eq!(read, [70_000, 0, 5], "a position past the end is no length");
16347        ends.push(9);
16348        assert!(lengths_of(&ends).is_none());
16349    }
16350
16351    /// Every worker of a scan wants the dictionary at the same moment and one of them fetches it.
16352    ///
16353    /// Asking a `OnceLock` whether it holds something answers the question a worker that already has
16354    /// the dictionary is asking and not the one a worker without it is asking, which is whether
16355    /// somebody is already on their way with it. Sixteen workers that all miss will all read the
16356    /// page, all verify it and all decode it, and fifteen will drop the result. Nothing about that
16357    /// is incorrect, which is why it went unnoticed, and it showed up as ClickBench 38 getting
16358    /// slower when the scan in front of it got faster and stopped staggering the arrivals.
16359    ///
16360    /// The barrier is what makes the test about that rather than about luck. Without it the first
16361    /// thread is usually finished before the last one starts and the count is one either way.
16362    #[test]
16363    fn a_global_dictionary_is_opened_once_however_many_workers_ask_at_once() {
16364        let path = path("dictionary-once");
16365        let parts = 8;
16366        let per_part = 500;
16367        let value =
16368            |row: usize| format!("{row:07} a value long enough to be worth a payload block");
16369        let mut writer =
16370            Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
16371                .expect("new file");
16372        for part in 0..parts {
16373            let values = (0..per_part)
16374                .map(|row| Value::Varchar(value(part * per_part + row)))
16375                .collect::<Vec<_>>();
16376            let chunk = Chunk::new(vec![
16377                Vector::from_values(LogicalType::Varchar, &values).expect("strings"),
16378            ])
16379            .expect("matching rows");
16380            writer.append(&chunk).expect("a part");
16381        }
16382        writer.finish().expect("commit");
16383
16384        let reader = Reader::open(&path).expect("reopen from disk");
16385        assert!(reader.table.dictionaries[0].is_some(), "the column has to have one to share");
16386        assert_eq!(reader.reads().dictionaries, 0, "opening the file does not open a dictionary");
16387
16388        let workers = 16;
16389        let gate = std::sync::Barrier::new(workers);
16390        std::thread::scope(|scope| {
16391            for worker in 0..workers {
16392                let reader = reader.clone();
16393                let gate = &gate;
16394                scope.spawn(move || {
16395                    gate.wait();
16396                    let chunk = reader.read(worker % parts, &[0]).expect("a part");
16397                    assert_eq!(
16398                        chunk.value_at(0, 0),
16399                        Value::Varchar(value((worker % parts) * per_part))
16400                    );
16401                });
16402            }
16403        });
16404
16405        assert_eq!(reader.reads().dictionaries, 1, "sixteen workers, one dictionary, one open");
16406        fs::remove_file(path).expect("remove scratch file");
16407    }
16408
16409    /// The sorted order sits outside the index the page checksum covers, because a query that
16410    /// never searches a dictionary should not read it, so it carries its own checksums and this is
16411    /// what says they are checked. A search that trusted a damaged order would give a wrong answer
16412    /// rather than a slow one.
16413    #[test]
16414    fn a_damaged_sorted_order_is_an_error() {
16415        let path = path("damaged-order");
16416        let mut writer = Writer::create(
16417            &path,
16418            "items",
16419            vec![
16420                Field::required("id", LogicalType::Integer),
16421                Field::new("text", LogicalType::Varchar),
16422            ],
16423        )
16424        .expect("new file");
16425        writer.append(&sample()).expect("stripe written");
16426        writer.finish().expect("commit");
16427
16428        let reader = Reader::open(&path).expect("valid directory");
16429        let page = reader.table.dictionaries[1].expect("string dictionary page");
16430        let mut header = [0; DICTIONARY_HEADER];
16431        read_at(&reader.file, page.offset, &mut header).expect("dictionary header");
16432        let index_len = dictionary_index_len(&header);
16433        let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
16434        file.seek(SeekFrom::Start(page.offset + index_len)).expect("the first head");
16435        file.write_all(&[255]).expect("damage the order");
16436
16437        let dictionary = reader.dictionary(1).expect("read").expect("a string column has one");
16438        let error = dictionary.compare_rank(0, b"anything").expect_err("a damaged order is caught");
16439        assert!(error.message().contains("rank checksum differs"), "{error}");
16440        fs::remove_file(path).expect("remove scratch file");
16441    }
16442
16443    /// Codes stay in first appearance order and the sorted order is written beside them, so a
16444    /// reader can put the values back in order without the writer having had to know them all
16445    /// before it handed out the first code.
16446    #[test]
16447    fn a_global_dictionary_carries_the_sorted_order_of_its_values() {
16448        // Chosen so the sort cannot be decided on the first eight bytes alone. Three values share
16449        // a nine byte prefix, one is a prefix of another, and one is empty.
16450        let spellings = ["overlong1z", "b", "", "overlong1a", "overlong", "ab", "a", "overlong1"];
16451        let path = path("dictionary-order");
16452        let mut writer =
16453            Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
16454                .expect("new file");
16455        writer
16456            .append(
16457                &Chunk::new(vec![
16458                    Vector::from_values(
16459                        LogicalType::Varchar,
16460                        &spellings.map(|text| Value::Varchar(text.into())),
16461                    )
16462                    .expect("strings"),
16463                ])
16464                .expect("one column"),
16465            )
16466            .expect("stripe written");
16467        writer.finish().expect("commit");
16468
16469        let reader = Reader::open(&path).expect("valid directory");
16470        let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
16471        let count = dictionary.ranks().expect("a v10 file stores one");
16472        assert_eq!(count, spellings.len(), "every distinct value has a rank");
16473        let order = (0..count)
16474            .map(|rank| dictionary.code_at_rank(rank).expect("a code"))
16475            .collect::<Vec<_>>();
16476        let mut seen = order.clone();
16477        seen.sort_unstable();
16478        assert_eq!(seen, (0..spellings.len() as u32).collect::<Vec<_>>(), "a permutation of codes");
16479
16480        let ranked = order
16481            .iter()
16482            .map(|&code| {
16483                dictionary.try_bytes_at(code as usize).expect("read").expect("a value").to_vec()
16484            })
16485            .collect::<Vec<_>>();
16486        let mut expected = spellings.map(|text| text.as_bytes().to_vec()).to_vec();
16487        expected.sort();
16488        assert_eq!(ranked, expected, "rank order is value order");
16489
16490        // What a search asks, on the values themselves rather than through a kernel, so that a
16491        // file whose heads disagree with its bytes is caught here rather than as a wrong answer.
16492        for (rank, value) in expected.iter().enumerate() {
16493            assert_eq!(
16494                dictionary.compare_rank(rank, value).expect("compare"),
16495                Ordering::Equal,
16496                "rank {rank} is its own value"
16497            );
16498            if rank > 0 {
16499                assert_eq!(
16500                    dictionary.compare_rank(rank - 1, value).expect("compare"),
16501                    Ordering::Less,
16502                    "rank {rank} follows the one before it"
16503                );
16504            }
16505        }
16506        fs::remove_file(path).expect("remove scratch file");
16507    }
16508
16509    /// Five text columns of different sizes close at the same time, and each comes back with its
16510    /// own values in its own order.
16511    ///
16512    /// The sizes differ so that the columns are taken in an order that is not the column order, and
16513    /// the values of each column are spelled with its number so that one column's page written in
16514    /// another's place would read back as the wrong strings rather than the right ones by chance.
16515    #[test]
16516    fn text_columns_closed_at_once_each_keep_their_own_dictionary() {
16517        let sizes = [300_usize, 5_000, 40, 2_000, 1_200];
16518        let path = path("dictionaries-at-once");
16519        let fields = (0..sizes.len())
16520            .map(|column| Field::new(format!("text{column}"), LogicalType::Varchar))
16521            .collect::<Vec<_>>();
16522        let mut writer = Writer::create(&path, "items", fields).expect("new file");
16523        let rows = 10_000_usize;
16524        for start in (0..rows).step_by(1_024) {
16525            let columns = sizes
16526                .iter()
16527                .enumerate()
16528                .map(|(column, &size)| {
16529                    let values = (start..(start + 1_024).min(rows))
16530                        .map(|row| Value::Varchar(format!("c{column}-{:05}", (row * 7919) % size)))
16531                        .collect::<Vec<_>>();
16532                    Vector::from_values(LogicalType::Varchar, &values).expect("strings")
16533                })
16534                .collect::<Vec<_>>();
16535            writer.append(&Chunk::new(columns).expect("five columns")).expect("stripe written");
16536        }
16537        writer.finish().expect("commit");
16538
16539        let reader = Reader::open(&path).expect("valid directory");
16540        for (column, &size) in sizes.iter().enumerate() {
16541            let dictionary =
16542                reader.dictionary(column).expect("read").expect("a string column has one");
16543            let count = dictionary.ranks().expect("a v10 file stores one");
16544            assert_eq!(count, size, "column {column} has its own distinct count");
16545            let ranked = (0..count)
16546                .map(|rank| {
16547                    let code = dictionary.code_at_rank(rank).expect("a code");
16548                    dictionary.try_bytes_at(code as usize).expect("read").expect("a value").to_vec()
16549                })
16550                .collect::<Vec<_>>();
16551            let expected = (0..size)
16552                .map(|value| format!("c{column}-{value:05}").into_bytes())
16553                .collect::<Vec<_>>();
16554            assert_eq!(ranked, expected, "column {column} ranks its own values in order");
16555        }
16556        fs::remove_file(path).expect("remove scratch file");
16557    }
16558
16559    /// A dictionary large enough to be decoded and sorted on several threads ranks the way one small
16560    /// enough for one thread does.
16561    ///
16562    /// Seventy thousand values over sixty nine blocks, in no order and each four times over so the
16563    /// column is worth a dictionary, written and ranked in the close.
16564    /// Some share a long prefix and some differ only in the last byte, so the buckets of the sort cut
16565    /// through runs of values that agree for a long way.
16566    #[test]
16567    fn a_large_dictionary_ranks_in_value_order() {
16568        let path = path("dictionary-large-rank");
16569        let value = |row: u64| {
16570            let mixed = row.wrapping_mul(0x9e37_79b9_7f4a_7c15) >> 40;
16571            match row % 3 {
16572                0 => format!("https://example.com/a/long/shared/path/{mixed:08}"),
16573                1 => format!("{mixed}"),
16574                _ => format!("x{}", row % 1000).repeat(1 + (row % 4) as usize) + &row.to_string(),
16575            }
16576        };
16577        let distinct = 70_000;
16578        let parts = 4 * distinct / 1000;
16579        let per_part = 1000;
16580        let mut writer =
16581            Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
16582                .expect("new file");
16583        for part in 0..parts {
16584            let values = (0..per_part)
16585                .map(|row| Value::Varchar(value((part * per_part + row) / 4)))
16586                .collect::<Vec<_>>();
16587            let chunk = Chunk::new(vec![
16588                Vector::from_values(LogicalType::Varchar, &values).expect("strings"),
16589            ])
16590            .expect("matching rows");
16591            writer.append(&chunk).expect("a part");
16592        }
16593        writer.finish().expect("commit");
16594
16595        let reader = Reader::open(&path).expect("reopen from disk");
16596        let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
16597        let count = dictionary.ranks().expect("a ranked dictionary");
16598        assert_eq!(count, distinct as usize, "every distinct value has a rank");
16599        assert!(count >= PARALLEL_SORT_MIN, "too few values to be sorted on more than one thread");
16600        let ranked = (0..count)
16601            .map(|rank| {
16602                let code = dictionary.code_at_rank(rank).expect("a code");
16603                dictionary.try_bytes_at(code as usize).expect("read").expect("a value").to_vec()
16604            })
16605            .collect::<Vec<_>>();
16606        let mut expected = (0..distinct).map(|row| value(row).into_bytes()).collect::<Vec<_>>();
16607        expected.sort();
16608        assert_eq!(ranked, expected, "rank order is value order");
16609        fs::remove_file(path).expect("remove scratch file");
16610    }
16611
16612    /// A string column's synopsis is turned into values without keeping the blocks it went through.
16613    ///
16614    /// Three thousand values, every fifth of them four times over, so the synopsis is a prefix of
16615    /// five hundred and twelve codes spread over all three payload blocks. Reading it used to leave
16616    /// all three decoded for as long as the reader lived. It leaves none of them now, and the second
16617    /// read answers out of what the first remembered.
16618    /// A directory read out of the file a window at a time is the directory read whole.
16619    ///
16620    /// The windows here are far smaller than any field is long, so every kind of field is split
16621    /// across a refill somewhere, and a bound is offered to its codec short more than once. The
16622    /// synopses are left in the file, and each one read back from where it was left is the one the
16623    /// whole read decoded.
16624    #[test]
16625    fn a_directory_read_a_window_at_a_time_is_the_directory_read_whole() {
16626        let path = path("windowed-directory");
16627        let fields = vec![
16628            Field::required("id", LogicalType::BigInt),
16629            Field::required("word", LogicalType::Varchar),
16630            Field::new("score", LogicalType::Double),
16631        ];
16632        let mut writer = Writer::create(&path, "items", fields).expect("new file");
16633        for part in 0..70_i64 {
16634            let ids = (0..100).map(|row| Value::BigInt(part * 100 + row % 7)).collect::<Vec<_>>();
16635            let words = (0..100)
16636                .map(|row| Value::Varchar(format!("word {}", row % 13)))
16637                .collect::<Vec<_>>();
16638            let scores = (0..100)
16639                .map(|row| if row % 4 == 0 { Value::Null } else { Value::Double(row as f64) })
16640                .collect::<Vec<_>>();
16641            let chunk = Chunk::new(vec![
16642                Vector::from_values(LogicalType::BigInt, &ids).expect("integers"),
16643                Vector::from_values(LogicalType::Varchar, &words).expect("strings"),
16644                Vector::from_values(LogicalType::Double, &scores).expect("doubles"),
16645            ])
16646            .expect("three columns");
16647            writer.append(&chunk).expect("a part");
16648        }
16649        writer.finish().expect("commit");
16650
16651        let catalog = Catalog::open(&path).expect("reopen");
16652        let entry = catalog.entries.first().expect("one table").directory;
16653        let (offset, length) = (entry.offset, entry.length as usize);
16654        let mut bytes = vec![0; length];
16655        read_at(&catalog.file, offset, &mut bytes).expect("the directory");
16656        assert_eq!(file_checksum(&catalog.file, offset, length).expect("checksum"), entry.hash);
16657        let whole = decode_directory(&bytes, catalog.size).expect("whole");
16658        assert!(whole.stripes.len() > 1, "the table should span stripes");
16659        for size in [1, 7, 33, 4_096] {
16660            let mut cursor = Cursor::over(&catalog.file, offset, length);
16661            cursor.window.as_mut().expect("a window").size = size;
16662            let windowed = read_directory(cursor, catalog.size, Some(offset)).expect("windowed");
16663            assert_eq!(format!("{:?}", windowed.stripes), format!("{:?}", whole.stripes));
16664            assert_eq!(format!("{:?}", windowed.fields), format!("{:?}", whole.fields));
16665            let mut stored = 0;
16666            for (column, (left, held)) in
16667                windowed.frequencies.iter().zip(&whole.frequencies).enumerate()
16668            {
16669                match (left, held) {
16670                    (None, None) => {}
16671                    (
16672                        Some(super::Frequencies::Stored { span, values, entries }),
16673                        Some(super::Frequencies::Held(summary)),
16674                    ) => {
16675                        let mut one = vec![0; span.length as usize];
16676                        read_at(&catalog.file, span.offset, &mut one).expect("a synopsis");
16677                        let read = decode_summary(
16678                            &mut Cursor::new(&one),
16679                            &whole.fields[column],
16680                            whole.rows,
16681                            *values,
16682                        )
16683                        .expect("a valid synopsis")
16684                        .expect("one is there");
16685                        assert_eq!(*entries, read.entries.len());
16686                        assert_eq!(format!("{read:?}"), format!("{summary:?}"));
16687                        stored += 1;
16688                    }
16689                    other => panic!("column {column} came back as {other:?}"),
16690                }
16691            }
16692            assert!(stored >= 2, "only {stored} synopses were left in the file");
16693        }
16694        let reader = catalog.table("items").expect("the table");
16695        assert!(reader.frequency_summaries[1].get().is_none());
16696        assert!(reader.top_frequencies(1, 1).expect("a readable synopsis").is_some());
16697        let first = reader.frequency_summaries[1].get().expect("decoded synopsis");
16698        let clone = reader.clone();
16699        assert!(clone.top_frequencies(1, 1).expect("cached synopsis").is_some());
16700        assert!(Arc::ptr_eq(first, clone.frequency_summaries[1].get().expect("same synopsis")));
16701        fs::remove_file(path).expect("remove scratch file");
16702    }
16703
16704    #[test]
16705    fn a_checksum_carried_across_reads_is_the_checksum_of_the_whole() {
16706        let path = path("file-checksum");
16707        let bytes = (0..200_000_u32)
16708            .map(|at| (at.wrapping_mul(2_654_435_761) >> 13) as u8)
16709            .collect::<Vec<_>>();
16710        fs::write(&path, &bytes).expect("scratch file");
16711        let file = File::open(&path).expect("open");
16712        for (offset, length) in [
16713            (0, 0),
16714            (3, 1),
16715            (5, 31),
16716            (0, 32),
16717            (9, 33),
16718            (1, 65_536),
16719            (7, 65_567),
16720            (0, 200_000),
16721            (11, 131_101),
16722        ] {
16723            let whole = checksum(&bytes[offset..offset + length]);
16724            assert_eq!(
16725                file_checksum(&file, offset as u64, length).expect("read"),
16726                whole,
16727                "{offset} {length}"
16728            );
16729        }
16730        fs::remove_file(path).expect("remove scratch file");
16731    }
16732
16733    #[test]
16734    fn a_string_synopsis_is_read_without_keeping_the_dictionary_blocks() {
16735        let path = path("synopsis-keeps-no-block");
16736        let spelled = |index: usize| Value::Varchar(format!("phrase {index:05}"));
16737        let mut values = (0..3_000).map(spelled).collect::<Vec<_>>();
16738        for _ in 0..3 {
16739            values.extend((0..3_000).step_by(5).map(spelled));
16740        }
16741        let mut writer =
16742            Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
16743                .expect("new file");
16744        for part in values.chunks(1_024) {
16745            writer
16746                .append(
16747                    &Chunk::new(vec![
16748                        Vector::from_values(LogicalType::Varchar, part).expect("strings"),
16749                    ])
16750                    .expect("one column"),
16751                )
16752                .expect("a part");
16753        }
16754        writer.finish().expect("commit");
16755
16756        let reader = Reader::open(&path).expect("reopen from disk");
16757        let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
16758        let resting = dictionary.footprint();
16759        let prefix = reader.frequency_prefix(0).expect("a readable synopsis").expect("one");
16760        assert_eq!(prefix.entries.len(), 512);
16761        for (value, count) in &prefix.entries {
16762            let Value::Varchar(text) = value else { panic!("a string column gave {value:?}") };
16763            let index = text["phrase ".len()..].parse::<usize>().expect("a spelled number");
16764            assert_eq!((index % 5, *count), (0, 4), "{text} came back with {count}");
16765        }
16766        assert_eq!(dictionary.footprint(), resting, "reading the synopsis kept a decoded block");
16767        let again = reader.frequency_prefix(0).expect("a readable synopsis").expect("one");
16768        assert_eq!(again.entries, prefix.entries);
16769        fs::remove_file(path).expect("remove scratch file");
16770    }
16771
16772    /// `length` over a stored column keeps a count a value rather than the blocks it counted.
16773    ///
16774    /// Reading the bytes a row at a time keeps every block it touches, so a scan of `length` over a
16775    /// whole column used to end up holding the column decoded. The counts are what is kept now, and
16776    /// they have to be the counts of characters rather than bytes, which is why the values here are
16777    /// not ASCII.
16778    #[test]
16779    fn character_lengths_are_counted_without_keeping_the_dictionary_blocks() {
16780        let path = path("character-lengths");
16781        let spellings = (0..2_500)
16782            .map(|index| Value::Varchar(format!("héllo {index:05} {}", "ü".repeat(index % 30))))
16783            .collect::<Vec<_>>();
16784        let mut writer =
16785            Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
16786                .expect("new file");
16787        for part in spellings.chunks(1_024) {
16788            writer
16789                .append(
16790                    &Chunk::new(vec![
16791                        Vector::from_values(LogicalType::Varchar, part).expect("strings"),
16792                    ])
16793                    .expect("one column"),
16794                )
16795                .expect("a part");
16796        }
16797        writer.finish().expect("commit");
16798
16799        let reader = Reader::open(&path).expect("reopen from disk");
16800        let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
16801        let resting = dictionary.footprint();
16802        let mut lens = Vec::new();
16803        assert!(dictionary.try_chars_lens(&mut lens).expect("counted"), "a stored source counts");
16804        let counted = dictionary.footprint() - resting;
16805        let blocks = dictionary.len().div_ceil(TEXT_PAYLOAD_VALUES);
16806        assert!(
16807            counted <= blocks * TEXT_PAYLOAD_VALUES * size_of::<u32>(),
16808            "counting kept {counted} bytes, more than a count a value"
16809        );
16810        let expected = (0..dictionary.len())
16811            .map(|code| {
16812                let bytes = dictionary.try_bytes_at(code).expect("read").expect("a value");
16813                i64::try_from(std::str::from_utf8(bytes).expect("utf-8").chars().count())
16814                    .expect("small")
16815            })
16816            .collect::<Vec<_>>();
16817        assert_eq!(lens, expected, "a count is the number of characters, not of bytes");
16818        let mut again = Vec::new();
16819        assert!(dictionary.try_chars_lens(&mut again).expect("counted"));
16820        assert_eq!(again, lens, "the kept counts answer the second time");
16821        fs::remove_file(path).expect("remove scratch file");
16822    }
16823
16824    /// Writes one column of strings whose code is where they sit in `spellings`, and reopens it.
16825    fn stored_spellings(label: &str, spellings: &[String]) -> (PathBuf, Reader) {
16826        let path = path(label);
16827        let values = spellings.iter().map(|text| Value::Varchar(text.clone())).collect::<Vec<_>>();
16828        let mut writer =
16829            Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
16830                .expect("new file");
16831        for part in values.chunks(1_024) {
16832            writer
16833                .append(
16834                    &Chunk::new(vec![
16835                        Vector::from_values(LogicalType::Varchar, part).expect("strings"),
16836                    ])
16837                    .expect("one column"),
16838                )
16839                .expect("a part");
16840        }
16841        writer.finish().expect("commit");
16842        let reader = Reader::open(&path).expect("reopen from disk");
16843        (path, reader)
16844    }
16845
16846    /// Codes that go all over a dictionary of `len` values, and every seventh row null.
16847    ///
16848    /// The shape of a vector a scan hands out: its codes are in row order, which lands them in
16849    /// every block of the dictionary in no order at all, so a read of the whole vector has to put
16850    /// them in block order itself to read each block once.
16851    fn scattered_rows(len: usize) -> (Vec<u32>, Vec<bool>) {
16852        let codes = (0..len)
16853            .map(|row| u32::try_from(row * 7_919 % len).expect("a small dictionary"))
16854            .collect::<Vec<_>>();
16855        let valid = (0..len).map(|row| row % 7 != 3).collect::<Vec<_>>();
16856        (codes, valid)
16857    }
16858
16859    /// `length` over a vector with nulls keeps the counts and not the blocks, the same as over one
16860    /// without.
16861    ///
16862    /// The whole vector count used to be taken only when no row was null, and every other vector
16863    /// went a row at a time through the bytes, which keeps every block it reads. A column with a
16864    /// null in each vector was held decoded after one `length` over it.
16865    #[test]
16866    fn character_lengths_with_nulls_are_counted_without_keeping_the_dictionary_blocks() {
16867        let spellings = (0..2_500)
16868            .map(|index| format!("héllo {index:05} {}", "ü".repeat(index % 30)))
16869            .collect::<Vec<_>>();
16870        let (path, reader) = stored_spellings("character-lengths-nulls", &spellings);
16871        let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
16872        let (codes, valid) = scattered_rows(spellings.len());
16873        let rows = Vector::dictionary_over(codes.clone(), Arc::clone(&dictionary))
16874            .expect("every code is inside")
16875            .with_validity(Validity::from_run(&valid));
16876
16877        let resting = dictionary.footprint();
16878        let lens = rudb_kernels::call("length", &[&rows], &LogicalType::BigInt, None)
16879            .expect("length reads");
16880        let counted = dictionary.footprint() - resting;
16881        let blocks = dictionary.len().div_ceil(TEXT_PAYLOAD_VALUES);
16882        assert!(
16883            counted <= blocks * TEXT_PAYLOAD_VALUES * size_of::<u32>(),
16884            "length over a vector with nulls kept {counted} bytes, more than a count a value"
16885        );
16886        let expected = (0..rows.len())
16887            .map(|row| match valid[row] {
16888                true => Value::BigInt(
16889                    i64::try_from(spellings[codes[row] as usize].chars().count()).expect("small"),
16890                ),
16891                false => Value::Null,
16892            })
16893            .collect::<Vec<_>>();
16894        let answers = (0..lens.len()).map(|row| lens.value_at(row)).collect::<Vec<_>>();
16895        assert_eq!(answers, expected, "a count of characters where a row has one, null elsewhere");
16896        fs::remove_file(path).expect("remove scratch file");
16897    }
16898
16899    /// `lower`, `upper` and `substring` read a stored dictionary a block at a time and keep none of
16900    /// it while the column is at its budget, until reading without keeping stops being cheap.
16901    ///
16902    /// The three used to read a row at a time through the bytes, which keeps every block a row lands
16903    /// in for as long as the table is open. They read the whole vector in one visit now, and the
16904    /// dictionary here is opened with a budget of zero so that what a visit would keep under the
16905    /// budget of a running database is what the test sees dropped. After a column's worth of blocks
16906    /// has been decoded and dropped the visit keeps what it reads, which is what bounds its cost on
16907    /// a scan whose codes keep coming back to every block, and the end of the test holds it to that.
16908    #[test]
16909    fn string_kernels_read_a_stored_dictionary_without_keeping_its_blocks() {
16910        let spellings = (0..2_500)
16911            .map(|index| format!("HéLLo {index:05} {}", "Üß".repeat(index % 30)))
16912            .collect::<Vec<_>>();
16913        let (path, reader) = stored_spellings("string-kernels", &spellings);
16914        let page = reader.table.dictionaries[0].expect("a string column has one");
16915        let starved =
16916            open_global_dictionary(Arc::clone(&reader.file), page, &LogicalType::Varchar, 0)
16917                .expect("a dictionary opens whatever it may keep");
16918        let starved = Arc::new(starved);
16919        let (codes, valid) = scattered_rows(spellings.len());
16920        let rows = Vector::dictionary_over(codes.clone(), Arc::clone(&starved))
16921            .expect("every code is inside")
16922            .with_validity(Validity::from_run(&valid));
16923        let expected = |each: &dyn Fn(&str) -> String| {
16924            (0..rows.len())
16925                .map(|row| match valid[row] {
16926                    true => Value::Varchar(each(&spellings[codes[row] as usize])),
16927                    false => Value::Null,
16928                })
16929                .collect::<Vec<_>>()
16930        };
16931        let answers =
16932            |vector: &Vector| (0..vector.len()).map(|row| vector.value_at(row)).collect::<Vec<_>>();
16933
16934        // What a visit may add is the table of where every value ends, four bytes a value, which
16935        // reading every value this often makes worth building. A block is tens of bytes a value.
16936        let resting = starved.footprint();
16937        let ends = spellings.len() * size_of::<u32>();
16938        let lowered = rudb_kernels::call("lower", &[&rows], &LogicalType::Varchar, None)
16939            .expect("lower reads");
16940        assert_eq!(answers(&lowered), expected(&|text| text.to_lowercase()), "lower");
16941        assert!(starved.footprint() <= resting + ends, "lower kept a block it read");
16942
16943        let start = Vector::constant(LogicalType::BigInt, Value::BigInt(3), rows.len());
16944        let length = Vector::constant(LogicalType::BigInt, Value::BigInt(9), rows.len());
16945        let cut =
16946            rudb_kernels::call("substring", &[&rows, &start, &length], &LogicalType::Varchar, None)
16947                .expect("substring reads");
16948        let cut_of = |text: &str| text.chars().skip(2).take(9).collect::<String>();
16949        assert_eq!(answers(&cut), expected(&cut_of), "substring");
16950        assert!(starved.footprint() <= resting + ends, "substring kept a block it read");
16951
16952        // Every block has been read twice now and dropped the second time as well, which is a
16953        // column's worth dropped for want of a budget, so the next visit keeps what it reads.
16954        let raised = rudb_kernels::call("upper", &[&rows], &LogicalType::Varchar, None)
16955            .expect("upper reads");
16956        assert_eq!(answers(&raised), expected(&|text| text.to_uppercase()), "upper");
16957        let payload = spellings.iter().map(String::len).sum::<usize>();
16958        assert!(
16959            starved.footprint() >= resting + payload,
16960            "a visit that has dropped a column's worth of blocks keeps what it reads"
16961        );
16962        let again = rudb_kernels::call("upper", &[&rows], &LogicalType::Varchar, None)
16963            .expect("upper reads kept blocks");
16964        assert_eq!(answers(&again), answers(&raised), "the kept blocks answer the same");
16965        fs::remove_file(path).expect("remove scratch file");
16966    }
16967
16968    /// A sweep of the dictionary reads every value, and the second sweep keeps what it read, up to
16969    /// the budget.
16970    ///
16971    /// The point of the sweep is the resident size rather than the answer, so both are checked
16972    /// here. The first sweep keeps nothing, because a process that runs one statement never reads
16973    /// a block twice. A dictionary this small is well under [`TEXT_KEEP_BUDGET`], so the second
16974    /// sweep keeps everything and a third decodes nothing, which is what makes a session asking the
16975    /// same question again cost what it should. The ceiling is the other half of it and it has its own
16976    /// test below, because a ceiling that never binds is not a ceiling anybody checked.
16977    #[test]
16978    fn a_dictionary_sweep_reads_every_value_and_keeps_it_under_the_budget() {
16979        let path = path("dictionary-sweep");
16980        // Two thousand five hundred distinct values is two whole payload blocks and a part of a
16981        // third, so the sweep has to be called more than once and the last call has to stop short.
16982        let spellings = (0..2_500)
16983            .map(|index| Value::Varchar(format!("value {index:08} {}", "x".repeat(index % 40))))
16984            .collect::<Vec<_>>();
16985        let mut writer =
16986            Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
16987                .expect("new file");
16988        // A chunk is a part and a part is at most 1,024 rows, so the values go in three of them.
16989        // The dictionary is table wide and does not care where a value was written.
16990        for part in spellings.chunks(1_024) {
16991            writer
16992                .append(
16993                    &Chunk::new(vec![
16994                        Vector::from_values(LogicalType::Varchar, part).expect("strings"),
16995                    ])
16996                    .expect("one column"),
16997                )
16998                .expect("stripe written");
16999        }
17000        writer.finish().expect("commit");
17001
17002        let reader = Reader::open(&path).expect("valid directory");
17003        let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
17004        assert_eq!(dictionary.len(), spellings.len(), "every value is distinct");
17005        for first in [0, TEXT_PAYLOAD_VALUES, TEXT_PAYLOAD_VALUES * 2] {
17006            assert!(dictionary.text_block_might_contain(first, b"value").expect("signature"));
17007            assert!(!dictionary.text_block_might_contain(first, b"google").expect("signature"));
17008        }
17009
17010        let resting = dictionary.footprint();
17011        let sweep = || {
17012            let mut swept: Vec<Vec<u8>> = Vec::new();
17013            let mut at = 0;
17014            let mut calls = 0;
17015            while at < dictionary.len() {
17016                let stopped = dictionary
17017                    .sweep_text(at, dictionary.len(), &mut |index: usize, text: &[u8]| {
17018                        assert_eq!(index, swept.len(), "a sweep hands its values over in order");
17019                        swept.push(text.to_vec());
17020                        Ok(())
17021                    })
17022                    .expect("a sweep reads");
17023                assert!(stopped > at, "a sweep moves");
17024                at = stopped;
17025                calls += 1;
17026            }
17027            assert_eq!(calls, 3, "a sweep hands over one block at a time");
17028            swept
17029        };
17030        let swept = sweep();
17031        assert_eq!(dictionary.footprint(), resting, "a first sweep keeps nothing it decoded");
17032        assert_eq!(sweep(), swept, "a second sweep reads what the first did");
17033        let after = dictionary.footprint();
17034        assert!(after > resting, "a second sweep under the budget keeps what it decoded");
17035
17036        let read = (0..dictionary.len())
17037            .map(|code| dictionary.try_bytes_at(code).expect("read").expect("a value").to_vec())
17038            .collect::<Vec<_>>();
17039        assert_eq!(swept, read, "a sweep answers what a point read answers");
17040        // A read per value is about what makes the unpacked ends worth building, so whether they
17041        // are built here depends on how many reads the sweep made on the way. They are the one thing
17042        // allowed to grow, by four bytes a value, and nothing of the payload is.
17043        let grown = dictionary.footprint() - after;
17044        assert!(
17045            grown == 0 || grown == dictionary.len() * size_of::<u32>(),
17046            "a point read of a kept block decodes nothing, and {grown} bytes grew"
17047        );
17048        fs::remove_file(path).expect("remove scratch file");
17049    }
17050
17051    #[test]
17052    fn a_narrow_signature_of_an_older_file_answers_by_its_own_width() {
17053        let path = path("narrow-substring-signature");
17054        let blocks = [&b"https://google.com/"[..], b"https://example.org/", b"mail.google.com"];
17055        let mut grams = Vec::new();
17056        for text in blocks {
17057            let mut bits = vec![0_u8; NARROW_GRAM_BYTES];
17058            for gram in text.windows(4) {
17059                for bit in gram_bits(gram, NARROW_GRAM_BYTES) {
17060                    bits[bit / 8] |= 1 << (bit % 8);
17061                }
17062            }
17063            grams.extend(bits);
17064        }
17065        fs::write(&path, &grams).expect("scratch file");
17066        let file = File::open(&path).expect("open scratch file");
17067        let signatures = NativeGrams {
17068            start: 0,
17069            length: grams.len(),
17070            width: NARROW_GRAM_BYTES,
17071            hash: checksum(&grams),
17072            verdicts: Mutex::new(Vec::new()),
17073        };
17074        let verdict = signatures.verdicts(&file, b"google").expect("signatures read");
17075        assert_eq!(&verdict[..], &[true, false, true], "one verdict a block, at the narrow width");
17076        assert!(signatures.footprint() > 0, "a verdict is remembered");
17077        let again = signatures.verdicts(&file, b"google").expect("remembered");
17078        assert!(Arc::ptr_eq(&verdict, &again), "a second question about a literal reads nothing");
17079
17080        let damaged = NativeGrams {
17081            hash: signatures.hash ^ 1,
17082            verdicts: Mutex::new(Vec::new()),
17083            ..signatures
17084        };
17085        let error = damaged.verdicts(&file, b"google").expect_err("a damaged region is refused");
17086        assert!(error.to_string().contains("substring signatures checksum differs"), "{error}");
17087        fs::remove_file(path).expect("remove scratch file");
17088    }
17089
17090    #[test]
17091    fn a_damaged_substring_signature_is_checked_only_when_used() {
17092        let path = path("damaged-substring-signature");
17093        let mut writer =
17094            Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
17095                .expect("new file");
17096        let rows = [Value::Varchar("google".into()), Value::Varchar("example".into())];
17097        writer
17098            .append(
17099                &Chunk::new(vec![
17100                    Vector::from_values(LogicalType::Varchar, &rows).expect("strings"),
17101                ])
17102                .expect("one column"),
17103            )
17104            .expect("stripe written");
17105        writer.finish().expect("commit");
17106
17107        let reader = Reader::open(&path).expect("valid directory");
17108        let page = reader.table.dictionaries[0].expect("string dictionary page");
17109        let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
17110        file.seek(SeekFrom::Start(page.offset + u64::from(page.length) - 1))
17111            .expect("last signature byte");
17112        file.write_all(&[255]).expect("damage signature");
17113        let reader = Reader::open(&path).expect("the directory is still valid");
17114        let dictionary = reader.dictionary(0).expect("index is still valid").expect("dictionary");
17115        let error = dictionary
17116            .text_block_might_contain(0, b"goog")
17117            .expect_err("a used signature checks its own checksum");
17118        assert!(error.message().contains("substring signatures checksum differs"), "{error}");
17119        fs::remove_file(path).expect("remove scratch file");
17120    }
17121
17122    /// A sweep over a block whose second run of offsets is short reads the same values as a point
17123    /// read does.
17124    ///
17125    /// The sweep decodes the offsets of a whole run at a time rather than a value at a time, and a
17126    /// run holds half a block, so the count it asks for is the run length everywhere but at the end
17127    /// of the dictionary. Two thousand five hundred values, which is what the test above writes,
17128    /// never puts a short run second in its block: the last block there begins on a run boundary and
17129    /// holds one run. Two thousand eight hundred does, so the last block is a whole run of five
17130    /// hundred and twelve followed by two hundred and forty, and an off by one in either the count
17131    /// asked for or the slice taken out of the answer shows up as a wrong value or a refusal.
17132    #[test]
17133    fn a_sweep_over_a_block_with_a_short_second_run_reads_what_a_point_read_reads() {
17134        let path = path("dictionary-sweep-short-run");
17135        let spellings = (0..2_800)
17136            .map(|index| Value::Varchar(format!("value {index:08} {}", "x".repeat(index % 40))))
17137            .collect::<Vec<_>>();
17138        let mut writer =
17139            Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
17140                .expect("new file");
17141        for part in spellings.chunks(1_024) {
17142            writer
17143                .append(
17144                    &Chunk::new(vec![
17145                        Vector::from_values(LogicalType::Varchar, part).expect("strings"),
17146                    ])
17147                    .expect("one column"),
17148                )
17149                .expect("stripe written");
17150        }
17151        writer.finish().expect("commit");
17152
17153        let reader = Reader::open(&path).expect("valid directory");
17154        let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
17155        assert_eq!(dictionary.len(), spellings.len(), "every value is distinct");
17156        let last = dictionary.len() % TEXT_PAYLOAD_VALUES;
17157        assert!(last > TEXT_OFFSET_RUN, "the last block has to reach into a second run of offsets");
17158        assert!(last < TEXT_PAYLOAD_VALUES, "and that second run has to be short of a whole one");
17159
17160        let mut swept: Vec<Vec<u8>> = Vec::new();
17161        let mut at = 0;
17162        while at < dictionary.len() {
17163            let stopped = dictionary
17164                .sweep_text(at, dictionary.len(), &mut |index: usize, text: &[u8]| {
17165                    assert_eq!(index, swept.len(), "a sweep hands its values over in order");
17166                    swept.push(text.to_vec());
17167                    Ok(())
17168                })
17169                .expect("a sweep reads");
17170            assert!(stopped > at, "a sweep moves");
17171            at = stopped;
17172        }
17173        let read = (0..dictionary.len())
17174            .map(|code| dictionary.try_bytes_at(code).expect("read").expect("a value").to_vec())
17175            .collect::<Vec<_>>();
17176        assert_eq!(swept, read, "a sweep answers what a point read answers");
17177        fs::remove_file(path).expect("remove scratch file");
17178    }
17179
17180    /// The unpacked ends answer what the packed ends answer, on both sides of the switch.
17181    ///
17182    /// A column asked for one offset at a time reads them out of the packed form until the reads
17183    /// are worth a table and out of the table after that, so every value here is read twice and the
17184    /// two passes are compared against the spellings and against each other. Two thousand eight
17185    /// hundred values is two payload blocks and a bit, which puts the switch in the middle of the
17186    /// first pass and means the pass straddles a block boundary, where the start of a value is zero
17187    /// rather than the end of the value before it.
17188    #[test]
17189    fn the_unpacked_ends_answer_what_the_packed_ends_answer() {
17190        let path = path("dictionary-unpacked-ends");
17191        let spellings = (0..2_800)
17192            .map(|index| Value::Varchar(format!("value {index:08} {}", "x".repeat(index % 40))))
17193            .collect::<Vec<_>>();
17194        let mut writer =
17195            Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
17196                .expect("new file");
17197        for part in spellings.chunks(1_024) {
17198            writer
17199                .append(
17200                    &Chunk::new(vec![
17201                        Vector::from_values(LogicalType::Varchar, part).expect("strings"),
17202                    ])
17203                    .expect("one column"),
17204                )
17205                .expect("stripe written");
17206        }
17207        writer.finish().expect("commit");
17208
17209        let reader = Reader::open(&path).expect("valid directory");
17210        let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
17211        assert_eq!(dictionary.len(), spellings.len(), "every value is distinct");
17212        let wanted = (0..spellings.len())
17213            .map(|index| format!("value {index:08} {}", "x".repeat(index % 40)).into_bytes())
17214            .collect::<Vec<_>>();
17215
17216        let pass = |what: &str| {
17217            for (index, value) in wanted.iter().enumerate() {
17218                let len = dictionary.try_bytes_len_at(index).expect("read").expect("a value");
17219                assert_eq!(len, value.len(), "{what} has the wrong length at {index}");
17220                let bytes = dictionary.try_bytes_at(index).expect("read").expect("a value");
17221                assert_eq!(bytes, value.as_slice(), "{what} has the wrong value at {index}");
17222            }
17223        };
17224        pass("the first pass");
17225        pass("the second pass");
17226
17227        // The whole vector in one call, over the text and through codes into it, which is how a
17228        // scan of a stored column hands it out. The codes run backwards and repeat so that they are
17229        // neither the positions nor in order.
17230        let lens = wanted.iter().map(|value| value.len() as i64).collect::<Vec<_>>();
17231        let mut whole = vec![0i64; wanted.len()];
17232        assert!(dictionary.try_bytes_lens(&mut whole).expect("read"), "the text answers whole");
17233        assert_eq!(whole, lens, "a vector of lengths answers what a length at a time answers");
17234        let codes = (0..4_000_u32).map(|row| (7 * (4_000 - row)) % 2_800).collect::<Vec<_>>();
17235        let coded = Vector::dictionary_over(codes.clone(), dictionary).expect("codes in range");
17236        let mut through = vec![0i64; codes.len()];
17237        assert!(coded.try_bytes_lens(&mut through).expect("read"), "the codes answer whole");
17238        for (row, &code) in codes.iter().enumerate() {
17239            assert_eq!(through[row], lens[code as usize], "row {row} reads code {code}");
17240            let one = coded.try_bytes_len_at(row).expect("read").expect("a value");
17241            assert_eq!(through[row], one as i64, "row {row} a row at a time");
17242        }
17243
17244        // A handful of codes over a column nobody has read yet is short of the table, so the same
17245        // call answers out of the packed ends instead, and has to answer the same.
17246        let fresh = Reader::open(&path).expect("valid directory");
17247        let untouched = fresh.dictionary(0).expect("read").expect("a string column has one");
17248        let few = vec![2_799_u32, 0, 1_024, 1_023, 511, 512];
17249        let coded = Vector::dictionary_over(few.clone(), untouched).expect("in range");
17250        let mut short = vec![0i64; few.len()];
17251        assert!(coded.try_bytes_lens(&mut short).expect("read"), "the codes answer whole");
17252        let expected = few.iter().map(|&code| lens[code as usize]).collect::<Vec<_>>();
17253        assert_eq!(short, expected, "the packed ends answer what the table answers");
17254        fs::remove_file(path).expect("remove scratch file");
17255    }
17256
17257    /// All three block layouts come back as the same values in the same order.
17258    ///
17259    /// Blocks outside the page are what every file this build writes holds. Blocks that say where
17260    /// they are but sit inside the page behind the order are format 26, and blocks behind one
17261    /// another with only their ends recorded are older still. Nothing in the writer produces the
17262    /// last two any more, so the only way to find out whether the reader still understands those
17263    /// files is to write them here. The
17264    /// bytes go straight into a file with no directory around them, because what is under test is
17265    /// [`open_global_dictionary`], which is handed a page and a file and asks the directory for
17266    /// nothing.
17267    ///
17268    /// Three thousand values so that there are three payload blocks and a partial fourth, which is
17269    /// what makes the last block the one place where a length and an end disagree about what they
17270    /// are counting.
17271    #[test]
17272    fn a_dictionary_reads_the_same_whether_its_blocks_say_where_they_are() {
17273        let spellings = (0..3_000)
17274            .map(|index| format!("value {index:08} {}", "y".repeat(index % 40)))
17275            .collect::<Vec<_>>();
17276        let mut read = Vec::new();
17277        for layout in ["outside", "inside", "behind"] {
17278            let mut dictionary = GlobalDictionary::new();
17279            for text in &spellings {
17280                dictionary.code(text).expect("a code for every spelling");
17281            }
17282            dictionary.finish_blocks().expect("the last block encodes");
17283            let order = dictionary.ranked(None).expect("a sorted order");
17284            // Where the blocks go if they start at `from` and follow one another.
17285            let laid = |from: u64| {
17286                let mut at = from;
17287                dictionary
17288                    .blocks
17289                    .iter()
17290                    .map(|block| {
17291                        let place =
17292                            Placed { start: at, length: block.len() as u64, hash: checksum(block) };
17293                        at += block.len() as u64;
17294                        place
17295                    })
17296                    .collect::<Vec<_>>()
17297            };
17298            let payload = dictionary.blocks.concat();
17299            let scattered = layout != "behind";
17300            let (bytes, encoded, offset, length) = if layout == "outside" {
17301                let mut bytes = vec![0; HEADER as usize];
17302                bytes.extend_from_slice(&payload);
17303                let encoded = encode_global_dictionary(&dictionary, &order, &laid(HEADER), true)
17304                    .expect("an encoding");
17305                let offset = bytes.len() as u64;
17306                bytes.extend_from_slice(&encoded.index);
17307                bytes.extend_from_slice(&encoded.ranks);
17308                bytes.extend_from_slice(&encoded.grams);
17309                let length = encoded.index.len() + encoded.ranks.len() + encoded.grams.len();
17310                (bytes, encoded, offset, length)
17311            } else {
17312                // The index is the same length wherever the blocks are, so a first pass says where
17313                // the page ends and the second writes the places that follow it.
17314                let first = encode_global_dictionary(&dictionary, &order, &laid(0), scattered)
17315                    .expect("an encoding");
17316                let body = (first.index.len() + first.ranks.len() + first.grams.len()) as u64;
17317                let encoded = encode_global_dictionary(&dictionary, &order, &laid(body), scattered)
17318                    .expect("an encoding");
17319                let mut bytes = encoded.index.clone();
17320                bytes.extend_from_slice(&encoded.ranks);
17321                bytes.extend_from_slice(&encoded.grams);
17322                bytes.extend_from_slice(&payload);
17323                let length = bytes.len();
17324                (bytes, encoded, 0, length)
17325            };
17326            let path = path(&format!("blocks-{layout}"));
17327            fs::write(&path, &bytes).expect("the dictionary is written on its own");
17328            let file = Arc::new(File::open(&path).expect("it opens again"));
17329            let page = Page {
17330                offset,
17331                length: u32::try_from(length).expect("a test dictionary is small"),
17332                hash: checksum(&encoded.index),
17333            };
17334            let opened =
17335                open_global_dictionary(file, page, &LogicalType::Varchar, TEXT_KEEP_BUDGET)
17336                    .expect("a dictionary laid out either way opens");
17337            let mut swept: Vec<Vec<u8>> = Vec::new();
17338            let mut at = 0;
17339            while at < opened.len() {
17340                at = opened
17341                    .sweep_text(at, opened.len(), &mut |_index: usize, text: &[u8]| {
17342                        swept.push(text.to_vec());
17343                        Ok(())
17344                    })
17345                    .expect("a sweep reads");
17346            }
17347            fs::remove_file(&path).expect("clean up");
17348            read.push(swept);
17349        }
17350        let wanted =
17351            spellings.iter().map(|text| text.as_bytes().to_vec()).collect::<Vec<Vec<u8>>>();
17352        assert_eq!(read[0], wanted, "the blocks outside the page hold the values");
17353        assert_eq!(read[1], read[0], "the blocks inside the page hold the same values");
17354        assert_eq!(read[2], read[0], "the blocks behind one another hold the same values");
17355    }
17356
17357    /// A dictionary at its budget sweeps without keeping, and still answers what it answered.
17358    ///
17359    /// The budget is a quarter of a gigabyte in a running database, which is a fine size for a real
17360    /// column and no size at all for a test, so this opens the same dictionary a second time with a
17361    /// budget of zero. That is the shape of the hundred million row case: `URL` fills the budget
17362    /// somewhere in the middle of itself and everything past that point is read and dropped, which
17363    /// costs the decode again and holds none of it.
17364    #[test]
17365    fn a_dictionary_at_its_budget_sweeps_without_keeping() {
17366        let path = path("dictionary-budget");
17367        let spellings = (0..2_500)
17368            .map(|index| Value::Varchar(format!("value {index:08} {}", "y".repeat(index % 40))))
17369            .collect::<Vec<_>>();
17370        let mut writer =
17371            Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
17372                .expect("new file");
17373        for part in spellings.chunks(1_024) {
17374            writer
17375                .append(
17376                    &Chunk::new(vec![
17377                        Vector::from_values(LogicalType::Varchar, part).expect("strings"),
17378                    ])
17379                    .expect("one column"),
17380                )
17381                .expect("stripe written");
17382        }
17383        writer.finish().expect("commit");
17384
17385        let reader = Reader::open(&path).expect("valid directory");
17386        let page = reader.table.dictionaries[0].expect("a string column has one");
17387        let file = Arc::clone(&reader.file);
17388        let starved = open_global_dictionary(file, page, &LogicalType::Varchar, 0)
17389            .expect("a dictionary opens whatever it may keep");
17390
17391        let resting = starved.footprint();
17392        let mut swept: Vec<Vec<u8>> = Vec::new();
17393        let mut at = 0;
17394        while at < starved.len() {
17395            at = starved
17396                .sweep_text(at, starved.len(), &mut |_index: usize, text: &[u8]| {
17397                    swept.push(text.to_vec());
17398                    Ok(())
17399                })
17400                .expect("a sweep reads");
17401        }
17402        assert_eq!(swept.len(), spellings.len(), "a starved sweep still reads every value");
17403        assert_eq!(starved.footprint(), resting, "and keeps no block it decoded");
17404
17405        let generous = reader.dictionary(0).expect("read").expect("a string column has one");
17406        let read = (0..generous.len())
17407            .map(|code| generous.try_bytes_at(code).expect("read").expect("a value").to_vec())
17408            .collect::<Vec<_>>();
17409        assert_eq!(swept, read, "a starved sweep answers what a point read answers");
17410        fs::remove_file(path).expect("remove scratch file");
17411    }
17412
17413    /// A part is hashed the first time a reader reads it and not after, and a reader opened after
17414    /// the part was damaged still refuses it.
17415    #[test]
17416    fn a_part_is_checked_once_per_open_reader() {
17417        let path = path("checked-once");
17418        let mut writer = Writer::create(
17419            &path,
17420            "items",
17421            vec![
17422                Field::required("id", LogicalType::Integer),
17423                Field::new("text", LogicalType::Varchar),
17424            ],
17425        )
17426        .expect("new file");
17427        writer.append(&sample()).expect("stripe written");
17428        writer.finish().expect("commit");
17429
17430        let reader = Reader::open(&path).expect("valid directory");
17431        let first = reader.read_rows(0, &[0], &[0, 1], false).expect("checked and read");
17432        assert!(reader.is_verified(0), "the part is remembered as checked");
17433        let page = reader.table.stripes[0].pages[0];
17434        let mut file = OpenOptions::new().write(true).open(&path).expect("open column page");
17435        file.seek(SeekFrom::Start(page.offset + u64::from(page.length) - 1)).expect("page end");
17436        file.write_all(&[0xa5]).expect("damage page");
17437        if let Err(error) = reader.read_rows(0, &[0], &[0, 1], false) {
17438            assert!(!error.message().contains("checksum differs"), "not hashed again: {error}");
17439        }
17440        let fresh = Reader::open(&path).expect("valid directory");
17441        let error = fresh.read_rows(0, &[0], &[0, 1], false).expect_err("a new reader checks");
17442        assert!(error.message().contains("column page checksum differs"), "{error}");
17443        assert_eq!(first.len(), 2);
17444        fs::remove_file(path).expect("remove scratch file");
17445    }
17446
17447    #[test]
17448    fn damaged_membership_cannot_skip_a_string_page() {
17449        let path = path("damaged-membership");
17450        let mut writer = Writer::create(
17451            &path,
17452            "items",
17453            vec![
17454                Field::required("id", LogicalType::Integer),
17455                Field::new("text", LogicalType::Varchar),
17456            ],
17457        )
17458        .expect("new file");
17459        writer.append(&sample()).expect("stripe written");
17460        writer.finish().expect("commit");
17461
17462        let reader = Reader::open(&path).expect("valid directory");
17463        let membership = reader.table.stripes[0].memberships.get(1).expect("string membership");
17464        let mut file = OpenOptions::new().write(true).open(&path).expect("open membership page");
17465        file.seek(SeekFrom::Start(membership.offset)).expect("membership start");
17466        file.write_all(&[255]).expect("damage membership");
17467        let error = reader.skips_codes(0, 1, &[3]).expect_err("corruption must not skip rows");
17468        assert!(error.message().contains("membership page checksum differs"), "{error}");
17469        fs::remove_file(path).expect("remove scratch file");
17470    }
17471
17472    #[test]
17473    fn membership_delta_stream_is_sorted_exact_and_bounded() {
17474        let unique = unique_codes(&[900, 4, 4, 72, 9, u32::MAX]);
17475        assert_eq!(unique, [4, 9, 72, 900, u32::MAX]);
17476        let encoded = encode_membership(&unique);
17477        assert_eq!(
17478            decode_membership(&encoded).expect("valid membership"),
17479            [4, 9, 72, 900, u32::MAX]
17480        );
17481        // A stripe's index is the union of its parts', so a code in two of them is in it once and
17482        // the result is still one ascending run of deltas.
17483        let merged = merged_codes(vec![vec![4, 900], vec![9, 900, u32::MAX], vec![72]]);
17484        assert_eq!(merged, [4, 9, 72, 900, u32::MAX]);
17485        assert_eq!(
17486            decode_membership(&encode_membership(&merged)).expect("valid membership"),
17487            unique
17488        );
17489        assert!(decode_membership(&[1, 0x80]).is_err(), "a truncated varint is invalid");
17490        assert!(
17491            decode_membership(&[1, 0xff, 0xff, 0xff, 0xff, 0x10]).is_err(),
17492            "a value past u32 is invalid"
17493        );
17494    }
17495
17496    #[test]
17497    fn a_global_dictionary_may_be_larger_than_one_column_page() {
17498        let dictionary = Page {
17499            offset: HEADER,
17500            length: u32::try_from(MAX_PAGE + 1).expect("the page bound fits on disk"),
17501            hash: 0,
17502        };
17503        let table = Table {
17504            name: "items".to_owned(),
17505            fields: vec![Field::new("text", LogicalType::Varchar)],
17506            stripes: Vec::new(),
17507            rows: 0,
17508            dictionaries: vec![Some(dictionary)],
17509            dictionary_payloads: Vec::new(),
17510            demoted: Vec::new(),
17511            distincts: vec![None],
17512            frequencies: vec![None],
17513            pair_frequencies: Vec::new(),
17514            frequency_texts: Vec::new(),
17515            host_groups: None,
17516            clustering: None,
17517            constraints: Constraints::default(),
17518            generation: 1,
17519            sections: Vec::new(),
17520        };
17521        let directory = encode_directory(&table).expect("directory");
17522        let file_size = dictionary.offset + u64::from(dictionary.length) + 1;
17523
17524        let decoded = decode_directory(&directory, file_size).expect("large lazy dictionary");
17525        assert_eq!(decoded.dictionaries[0].expect("dictionary").length, dictionary.length);
17526    }
17527
17528    #[test]
17529    fn a_column_with_one_value_everywhere_costs_almost_nothing_a_row() {
17530        let path = path("constant-codes");
17531        let mut writer =
17532            Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
17533                .expect("new file");
17534        let empty = vec![Value::Varchar(String::new()); 1024];
17535        for _ in 0..4 {
17536            let column = Vector::from_values(LogicalType::Varchar, &empty).expect("strings");
17537            writer.append(&Chunk::new(vec![column]).expect("one column")).expect("a part");
17538        }
17539        writer.finish().expect("commit");
17540
17541        let reader = Reader::open(&path).expect("valid directory");
17542        let pages = reader.layout().columns.first().expect("one column").pages;
17543        // This column used to cost four bytes a row, 16,384 of them, the same as a column of four
17544        // thousand distinct URLs would. The cascade calls each part a constant, so what is left is
17545        // a tag, a count and the value, and the row count stops being what drives the number.
17546        assert!(pages < 256, "{pages} bytes of pages for 4,096 rows of one value");
17547        let read = reader.read(3, &[0]).expect("the last part back");
17548        assert_eq!(read.value_at(0, 0), Value::Varchar(String::new()));
17549        assert_eq!(read.value_at(1023, 0), Value::Varchar(String::new()));
17550        fs::remove_file(path).expect("remove scratch file");
17551    }
17552
17553    #[test]
17554    fn a_cascade_value_too_wide_for_its_column_is_refused_rather_than_cut() {
17555        // What a damaged page looks like from here: the cascade decoded, so the bytes are not
17556        // truncated, but the values do not belong to the column the directory says they do.
17557        let over = integer::encode(&[i64::from(i32::MAX) + 1]).expect("a chunk");
17558        let error = cascade(&LogicalType::Integer, &over, 1).expect_err("a page that disagrees");
17559        assert!(format!("{error}").contains("not of its type"), "{error}");
17560        let low = integer::encode(&[i64::MIN]).expect("a chunk");
17561        assert!(cascade(&LogicalType::BigInt, &low, 1).is_ok(), "bigint holds all of i64");
17562        let zero = integer::encode(&[0]).expect("a chunk");
17563        assert!(cascade(&LogicalType::Varchar, &zero, 1).is_err(), "strings are not integers");
17564    }
17565
17566    #[test]
17567    fn a_code_stream_the_cascade_cannot_shrink_is_left_alone() {
17568        // A shift register rather than a run, because an arithmetic run is the one wide shape the
17569        // cascade does shrink. This is what a column with tens of millions of distinct values hands
17570        // over: full width codes with no order to them.
17571        let mut state: u32 = 0x9e37_79b9;
17572        let spread: Vec<u32> = (0..1024)
17573            .map(|_| {
17574                state ^= state << 13;
17575                state ^= state >> 17;
17576                state ^= state << 5;
17577                state
17578            })
17579            .collect();
17580        assert_eq!(encoded_codes(&spread).expect("no failure"), None);
17581        let near: Vec<u32> = (0..1024).collect();
17582        let coded = encoded_codes(&near).expect("no failure").expect("counting up is packable");
17583        assert!(coded.len() < near.len() * 4, "{} bytes for a run of 1,024", coded.len());
17584    }
17585
17586    /// The columns of a stripe are encoded on whichever thread got to them, so the one thing that
17587    /// must not depend on which thread that was is the file. Two writes of the same rows are
17588    /// compared byte for byte rather than value for value, because a dictionary that two columns
17589    /// somehow shared would still read back correctly and would hand out its codes in the order the
17590    /// threads happened to run in, which is exactly what this is here to catch.
17591    #[test]
17592    fn two_writes_of_the_same_rows_give_the_same_bytes() {
17593        fn written(path: &PathBuf) {
17594            let fields = (0..40)
17595                .map(|column| {
17596                    let ty =
17597                        if column % 4 == 0 { LogicalType::Varchar } else { LogicalType::BigInt };
17598                    Field::new(format!("c{column}"), ty)
17599                })
17600                .collect::<Vec<_>>();
17601            let mut writer = Writer::create(path, "wide", fields).expect("new file");
17602            for part in 0..70_u64 {
17603                let columns = (0..40)
17604                    .map(|column| {
17605                        let values = (0..64_u64)
17606                            .map(|row| {
17607                                let seed = part.wrapping_mul(31).wrapping_add(row);
17608                                if column % 4 == 0 {
17609                                    Value::Varchar(format!("v{}", seed % 17))
17610                                } else {
17611                                    Value::BigInt(i64::try_from(seed % 97).expect("small"))
17612                                }
17613                            })
17614                            .collect::<Vec<_>>();
17615                        let ty = if column % 4 == 0 {
17616                            LogicalType::Varchar
17617                        } else {
17618                            LogicalType::BigInt
17619                        };
17620                        Vector::from_values(ty, &values).expect("a column")
17621                    })
17622                    .collect::<Vec<_>>();
17623                writer.append(&Chunk::new(columns).expect("forty columns")).expect("a part");
17624            }
17625            writer.finish().expect("commit");
17626        }
17627
17628        let first = path("repeatable-one");
17629        let second = path("repeatable-two");
17630        written(&first);
17631        written(&second);
17632        let left = fs::read(&first).expect("the first file");
17633        let right = fs::read(&second).expect("the second file");
17634        assert_eq!(left.len(), right.len(), "two writes of the same rows differ in length");
17635        assert!(left == right, "two writes of the same rows differ in their bytes");
17636
17637        // And the rows are still there, since a pair of identically wrong files would pass the
17638        // comparison above on its own.
17639        let reader = Reader::open(&first).expect("valid directory");
17640        assert_eq!(reader.table().rows(), 70 * 64);
17641        let read = reader.read(0, &[0, 1]).expect("the first part back");
17642        assert_eq!(read.value_at(0, 0), Value::Varchar("v0".to_owned()));
17643        assert_eq!(read.value_at(0, 1), Value::BigInt(0));
17644        fs::remove_file(first).expect("remove scratch file");
17645        fs::remove_file(second).expect("remove scratch file");
17646    }
17647
17648    /// Three tables of different shapes in one file, read back by name.
17649    fn three_tables(path: &PathBuf) {
17650        let writer = Writer::create(
17651            path,
17652            "region",
17653            vec![
17654                Field::new("r_key", LogicalType::Integer),
17655                Field::new("r_name", LogicalType::Varchar),
17656            ],
17657        )
17658        .expect("new file");
17659        let mut writer = writer;
17660        writer
17661            .append(
17662                &Chunk::new(vec![
17663                    Vector::from_values(
17664                        LogicalType::Integer,
17665                        &[Value::Integer(0), Value::Integer(1)],
17666                    )
17667                    .expect("keys"),
17668                    Vector::from_values(
17669                        LogicalType::Varchar,
17670                        &[Value::Varchar("AFRICA".to_owned()), Value::Varchar("ASIA".to_owned())],
17671                    )
17672                    .expect("names"),
17673                ])
17674                .expect("two columns"),
17675            )
17676            .expect("a part");
17677        let mut writer = writer
17678            .next("empty", vec![Field::new("nothing", LogicalType::BigInt)])
17679            .expect("a second table");
17680        writer
17681            .append(
17682                &Chunk::new(vec![
17683                    Vector::from_values(LogicalType::BigInt, &[Value::BigInt(7)]).expect("a row"),
17684                ])
17685                .expect("one column"),
17686            )
17687            .expect("a part");
17688        let mut writer =
17689            writer.next("wide", vec![Field::new("n", LogicalType::BigInt)]).expect("a third table");
17690        for part in 0..70_i64 {
17691            let values = (0..64).map(|row| Value::BigInt(part * 64 + row)).collect::<Vec<_>>();
17692            writer
17693                .append(
17694                    &Chunk::new(vec![
17695                        Vector::from_values(LogicalType::BigInt, &values).expect("a column"),
17696                    ])
17697                    .expect("one column"),
17698                )
17699                .expect("a part");
17700        }
17701        writer.finish().expect("commit");
17702    }
17703
17704    #[test]
17705    fn three_tables_in_one_file_read_back_by_name() {
17706        let file = path("three-tables");
17707        three_tables(&file);
17708        let catalog = Catalog::open(&file).expect("a committed catalog");
17709        assert_eq!(catalog.names().collect::<Vec<_>>(), ["region", "empty", "wide"]);
17710
17711        let region = catalog.table("region").expect("the first table");
17712        assert_eq!(region.table().rows(), 2);
17713        assert_eq!(
17714            region.read(0, &[1]).expect("names").value_at(1, 0),
17715            Value::Varchar("ASIA".to_owned())
17716        );
17717
17718        let wide = catalog.table("wide").expect("the third table");
17719        assert_eq!(wide.table().rows(), 70 * 64);
17720        assert_eq!(wide.read(0, &[0]).expect("the first part").value_at(0, 0), Value::BigInt(0));
17721
17722        // The middle table is reached without the one after it having been touched, which is what
17723        // a directory per table buys over one directory of everything.
17724        let empty = catalog.table("empty").expect("the second table");
17725        assert_eq!(empty.table().rows(), 1);
17726        assert_eq!(empty.read(0, &[0]).expect("the row").value_at(0, 0), Value::BigInt(7));
17727
17728        fs::remove_file(file).expect("remove scratch file");
17729    }
17730
17731    #[test]
17732    fn a_name_the_file_does_not_hold_is_an_error_rather_than_the_first_table() {
17733        let file = path("three-tables-missing");
17734        three_tables(&file);
17735        let catalog = Catalog::open(&file).expect("a committed catalog");
17736        let error = catalog.table("nation").expect_err("no such table");
17737        assert!(error.message().contains("nation"), "{}", error.message());
17738        fs::remove_file(file).expect("remove scratch file");
17739    }
17740
17741    #[test]
17742    fn a_file_of_three_tables_will_not_open_as_one() {
17743        let file = path("three-tables-unnamed");
17744        three_tables(&file);
17745        let error = Reader::open(&file).expect_err("more than one table");
17746        assert!(error.message().contains("more than one table"), "{}", error.message());
17747        fs::remove_file(file).expect("remove scratch file");
17748    }
17749
17750    /// One column per storage width, because the width is what decides how many bytes a row costs.
17751    #[test]
17752    fn decimals_of_every_storage_width_round_trip() {
17753        let file = path("decimals");
17754        let widths = [(4_u8, 2_u8), (9, 2), (18, 4), (38, 6)];
17755        let fields = widths
17756            .iter()
17757            .enumerate()
17758            .map(|(index, (width, scale))| {
17759                Field::new(
17760                    format!("d{index}"),
17761                    LogicalType::decimal(*width, *scale).expect("a decimal type"),
17762                )
17763            })
17764            .collect::<Vec<_>>();
17765        let mut writer = Writer::create(&file, "money", fields).expect("new file");
17766        let rows: [i128; 3] = [-1234, 0, 999];
17767        let columns = widths
17768            .iter()
17769            .map(|(width, scale)| {
17770                let values = rows
17771                    .iter()
17772                    .map(|unscaled| Value::Decimal {
17773                        unscaled: *unscaled,
17774                        width: *width,
17775                        scale: *scale,
17776                    })
17777                    .collect::<Vec<_>>();
17778                Vector::from_values(
17779                    LogicalType::decimal(*width, *scale).expect("a decimal type"),
17780                    &values,
17781                )
17782                .expect("a decimal column")
17783            })
17784            .collect::<Vec<_>>();
17785        writer.append(&Chunk::new(columns).expect("four columns")).expect("a part");
17786        writer.finish().expect("commit");
17787
17788        let reader = Reader::open(&file).expect("a committed file");
17789        for (index, (width, scale)) in widths.iter().enumerate() {
17790            assert_eq!(
17791                reader.table().fields()[index].ty,
17792                LogicalType::decimal(*width, *scale).expect("a decimal type"),
17793                "column {index} came back as another type"
17794            );
17795            let column = reader.read(0, &[index]).expect("the column");
17796            for (row, unscaled) in rows.iter().enumerate() {
17797                assert_eq!(
17798                    column.value_at(row, 0),
17799                    Value::Decimal { unscaled: *unscaled, width: *width, scale: *scale },
17800                    "column {index} row {row}"
17801                );
17802            }
17803        }
17804        fs::remove_file(file).expect("remove scratch file");
17805    }
17806
17807    #[test]
17808    fn two_tables_of_one_name_are_refused_before_anything_is_committed() {
17809        let file = path("two-of-a-name");
17810        let writer = Writer::create(&file, "t", vec![Field::new("a", LogicalType::BigInt)])
17811            .expect("new file");
17812        let error = writer
17813            .next("t", vec![Field::new("a", LogicalType::BigInt)])
17814            .expect_err("the same name twice");
17815        assert!(error.message().contains("same name"), "{}", error.message());
17816        fs::remove_file(file).expect("remove scratch file");
17817    }
17818
17819    #[test]
17820    fn integer_tally_counts_encoded_rows_and_declines_null_parts() {
17821        let file = path("integer-tally");
17822        let mut writer =
17823            Writer::create(&file, "events", vec![Field::new("source", LogicalType::SmallInt)])
17824                .expect("new file");
17825        let mut values = vec![Value::SmallInt(0); 1024];
17826        values[7] = Value::SmallInt(3);
17827        values[99] = Value::SmallInt(-2);
17828        values[1001] = Value::SmallInt(3);
17829        let column = Vector::from_values(LogicalType::SmallInt, &values).expect("integer values");
17830        writer.append(&Chunk::new(vec![column]).expect("one column")).expect("first part");
17831        values[0] = Value::Null;
17832        let column = Vector::from_values(LogicalType::SmallInt, &values).expect("nullable values");
17833        writer.append(&Chunk::new(vec![column]).expect("one column")).expect("second part");
17834        writer.finish().expect("commit");
17835
17836        let reader = Reader::open(&file).expect("read file");
17837        assert_eq!(
17838            reader.integer_tally(0, 0).expect("valid part"),
17839            Some(vec![(-2, 1), (0, 1021), (3, 2)])
17840        );
17841        assert!(reader.integer_tally(1, 0).expect("valid null part").is_none());
17842        let catalog = Catalog::open(&file).expect("catalog");
17843        assert_eq!(
17844            catalog.integer_tally("events", 0).expect("nullable column"),
17845            Some(vec![(-2, 2), (0, 2041), (3, 4)])
17846        );
17847        fs::remove_file(file).expect("remove scratch file");
17848    }
17849
17850    #[test]
17851    fn catalog_tallies_one_integer_column_without_opening_the_whole_table() {
17852        let file = path("catalog-integer-tally");
17853        let mut writer = Writer::create(
17854            &file,
17855            "events",
17856            vec![
17857                Field::new("noise", LogicalType::SmallInt),
17858                Field::new("source", LogicalType::SmallInt),
17859            ],
17860        )
17861        .expect("new file");
17862        let noise = vec![Value::SmallInt(9); 1024];
17863        let mut source = vec![Value::SmallInt(0); 1024];
17864        source[7] = Value::SmallInt(3);
17865        source[99] = Value::SmallInt(-2);
17866        let chunk = Chunk::new(vec![
17867            Vector::from_values(LogicalType::SmallInt, &noise).expect("noise"),
17868            Vector::from_values(LogicalType::SmallInt, &source).expect("source"),
17869        ])
17870        .expect("two columns");
17871        writer.append(&chunk).expect("append");
17872        writer.finish().expect("commit");
17873
17874        let catalog = Catalog::open(&file).expect("catalog");
17875        assert_eq!(
17876            catalog.integer_tally("events", 1).expect("selected column"),
17877            Some(vec![(-2, 1), (0, 1022), (3, 1)])
17878        );
17879        assert_eq!(
17880            catalog.integer_tally("events", 0).expect("other column"),
17881            Some(vec![(9, 1024)])
17882        );
17883        fs::remove_file(file).expect("remove scratch file");
17884    }
17885
17886    #[test]
17887    fn opening_the_catalog_reads_no_table_directory() {
17888        let file = path("catalog-only");
17889        three_tables(&file);
17890        let catalog = Catalog::open(&file).expect("a committed catalog");
17891        // The header and one slot, and nothing under it. The third table's directory covers seventy
17892        // stripes and reading it here would be the whole point of the two levels thrown away.
17893        assert_eq!(catalog.opening.reads, 2, "opening the catalog read more than the slot");
17894        assert_eq!(catalog.names().len(), 3);
17895        fs::remove_file(file).expect("remove scratch file");
17896    }
17897
17898    /// The checksum answers what it has always answered, at every length its branches split on.
17899    ///
17900    /// This is a compatibility test rather than a correctness one. Nothing about the hash has to be
17901    /// any particular function, but a file already on disk carries the answers the version that
17902    /// wrote it gave, so a change here is a change that makes every stored file fail to verify. The
17903    /// lengths are the ones the code makes decisions about: nothing, under a block, a block exactly,
17904    /// a block and a word, a word and a half word, and a half word and a byte.
17905    ///
17906    /// The empty answer is the published xxHash64 vector for an empty input at seed zero, which is
17907    /// also a check that this is the function it says it is.
17908    #[test]
17909    fn the_checksum_answers_what_it_has_always_answered() {
17910        let bytes: Vec<u8> =
17911            (0..1000_u32).map(|at| (at.wrapping_mul(31).wrapping_add(7) % 251) as u8).collect();
17912        for (length, expected) in [
17913            (0, 0xef46_db37_51d8_e999),
17914            (1, 0xa96c_7f0c_e858_bbb7),
17915            (3, 0x56e6_9576_32a4_87f9),
17916            (4, 0xc60d_15b1_e3ff_8f04),
17917            (5, 0x8088_1585_8624_dd4e),
17918            (7, 0xafbe_fc3d_6c6f_9a8e),
17919            (8, 0x3da5_c7aa_2696_83e0),
17920            (9, 0x465e_c429_b13c_3892),
17921            (15, 0xdee8_9d8a_065a_6233),
17922            (16, 0x1330_489a_7767_9c80),
17923            (31, 0x3391_303d_485e_846e),
17924            (32, 0x40b7_aff7_5d45_bbc8),
17925            (33, 0x4997_cae4_951c_17a5),
17926            (39, 0x5807_28fd_5c14_5739),
17927            (40, 0xf95c_f6f5_c08a_3d3b),
17928            (63, 0x2944_b4da_fc69_b206),
17929            (64, 0xbb76_f6ef_19bd_5a1b),
17930            (65, 0x814e_0c65_4a9f_d640),
17931            (127, 0x00de_aab1_31cf_f89b),
17932            (1000, 0x9e33_00c1_cde3_c58d),
17933        ] {
17934            assert_eq!(checksum(&bytes[..length]), expected, "the checksum of {length} bytes");
17935        }
17936        assert_eq!(checksum(b"the quick brown fox jumps over the lazy dog"), 0xed71_4233_c5a9_a792);
17937    }
17938    /// A declared order survives the file, and a table that declared none stays as it was.
17939    ///
17940    /// The second half is the one worth a test. The clustering section is written only when there
17941    /// is a declaration, so a file of two tables where one is clustered exercises both the present
17942    /// and the absent branch of the decoder in one directory, which is where a length bug would
17943    /// show up as one table reading the other's bytes.
17944    #[test]
17945    fn a_declared_order_comes_back_out_of_the_file() {
17946        let path = path("clustered");
17947        let shipped = vec![
17948            Field::new("key", LogicalType::BigInt),
17949            Field::new("line", LogicalType::Integer),
17950            Field::new("shipdate", LogicalType::Date),
17951        ];
17952        let plain = vec![Field::new("a", LogicalType::Integer)];
17953        let stage_zero = Clustering::new(vec![2, 0, 1], Width::Month, &shipped).expect("valid");
17954
17955        let mut writer = Writer::create(&path, "lineitem", shipped)
17956            .expect("new file")
17957            .declare(stage_zero.clone())
17958            .expect("the columns are the table's");
17959        let column = |ty: LogicalType, values: &[Value]| {
17960            Vector::from_values(ty, values).expect("the values match the type")
17961        };
17962        writer
17963            .append(
17964                &Chunk::new(vec![
17965                    column(
17966                        LogicalType::BigInt,
17967                        &[Value::BigInt(0), Value::BigInt(1), Value::BigInt(2), Value::BigInt(3)],
17968                    ),
17969                    column(
17970                        LogicalType::Integer,
17971                        &[
17972                            Value::Integer(1),
17973                            Value::Integer(1),
17974                            Value::Integer(1),
17975                            Value::Integer(1),
17976                        ],
17977                    ),
17978                    column(
17979                        LogicalType::Date,
17980                        &[Value::Date(0), Value::Date(1), Value::Date(2), Value::Date(3)],
17981                    ),
17982                ])
17983                .expect("three columns"),
17984            )
17985            .expect("four rows");
17986        let mut writer = writer.next("nation", plain).expect("a second table");
17987        writer
17988            .append(
17989                &Chunk::new(vec![column(LogicalType::Integer, &[Value::Integer(7)])])
17990                    .expect("one column"),
17991            )
17992            .expect("one row");
17993        writer.finish().expect("commit");
17994
17995        let catalog = Catalog::open(&path).expect("reopen");
17996        let lineitem = catalog.table("lineitem").expect("the clustered table");
17997        assert_eq!(lineitem.table().clustering(), Some(&stage_zero));
17998        let nation = catalog.table("nation").expect("the plain table");
17999        assert_eq!(nation.table().clustering(), None, "nobody declared one here");
18000
18001        // And the rows are still the rows, because the section goes on the end of the directory
18002        // and the easy way to break that is to leave the cursor somewhere the next read trusts.
18003        assert_eq!(lineitem.table().rows(), 4);
18004        assert_eq!(nation.table().rows(), 1);
18005        fs::remove_file(&path).ok();
18006    }
18007
18008    /// A declaration naming a column the table does not have is refused where it is made.
18009    #[test]
18010    fn a_declaration_off_the_end_of_the_table_never_reaches_the_file() {
18011        let path = path("clustered-bad");
18012        let writer = Writer::create(&path, "items", vec![Field::new("a", LogicalType::Integer)])
18013            .expect("new file");
18014        let four =
18015            (0..4).map(|at| Field::new(format!("c{at}"), LogicalType::Integer)).collect::<Vec<_>>();
18016        let wrong = Clustering::new(vec![3], Width::Exact, &four).expect("valid against four");
18017        assert!(writer.declare(wrong).is_err(), "the table has one column, not four");
18018        fs::remove_file(&path).ok();
18019    }
18020
18021    /// The sorted order is the byte order, whatever the values do before they differ.
18022    ///
18023    /// A block handed out of the writer's lock to be encoded, and given back in whatever order the
18024    /// stripes happen to finish in, is the same block with the same signature as one encoded in
18025    /// place, and lands in the same position.
18026    #[test]
18027    fn blocks_handed_out_and_given_back_out_of_order_are_the_blocks_encoded_in_place() {
18028        let values = (0..PAYLOAD_SAMPLE_BLOCKS * TEXT_PAYLOAD_VALUES * 2 + 100)
18029            .map(|at| format!("http://example{}.test/page/{at:06}", at % 7))
18030            .collect::<Vec<_>>();
18031        let filled = || {
18032            let mut dictionary = GlobalDictionary::new();
18033            for value in &values {
18034                dictionary.code(value).expect("a code for every value");
18035            }
18036            dictionary.settle().expect("a shape");
18037            dictionary
18038        };
18039        let mut in_place = filled();
18040        in_place.finish_blocks().expect("every block encodes");
18041
18042        let mut handed = filled();
18043        let out = handed.hand_out(3);
18044        assert_eq!(out.len(), PAYLOAD_SAMPLE_BLOCKS * 2, "every sealed block goes out");
18045        assert!(handed.waiting.is_empty(), "and none is left to be encoded under the lock");
18046        for job in out.iter().rev() {
18047            assert_eq!(job.place().0, 3, "a block goes back to the column it came from");
18048            handed.take_back(job.place().1, job.encode().expect("encodes")).expect("taken back");
18049        }
18050        assert!(handed.early.is_empty(), "nothing is waiting on a gap");
18051        handed.finish_blocks().expect("the last block encodes");
18052
18053        assert_eq!(handed.blocks, in_place.blocks, "the same blocks in the same order");
18054        assert_eq!(handed.grams, in_place.grams, "with the same signatures");
18055    }
18056
18057    /// A block given back twice is a bug in whoever gave it, and is said rather than written twice.
18058    #[test]
18059    fn a_block_given_back_twice_is_refused() {
18060        let mut dictionary = GlobalDictionary::new();
18061        for at in 0..PAYLOAD_SAMPLE_BLOCKS * TEXT_PAYLOAD_VALUES {
18062            dictionary.code(&format!("value {at}")).expect("a code");
18063        }
18064        dictionary.settle().expect("a shape");
18065        let out = dictionary.hand_out(0);
18066        let last = out.last().expect("blocks went out");
18067        let at = last.place().1;
18068        dictionary.take_back(at, last.encode().expect("encodes")).expect("taken back once");
18069        assert!(dictionary.take_back(at, last.encode().expect("encodes")).is_err());
18070    }
18071
18072    /// The values here are the shape the sort is built for and the shape a comparison sort is worst
18073    /// at: a common scheme, a handful of hosts, and a path that only decides the pair thirty bytes
18074    /// in. They also cover what the bucketing has to get right at the edges, which is a value that
18075    /// has run out where another carries on, the empty value, and enough entries to take the range
18076    /// down through several passes and out the bottom into the comparison that finishes it.
18077    #[test]
18078    fn the_dictionary_order_is_the_byte_order_however_deep_the_values_agree() {
18079        let mut values = vec![String::new(), "http://".to_owned()];
18080        for host in 0..7 {
18081            for path in 0..30 {
18082                values.push(format!("http://example{host}.test/page/{path:04}/index.html"));
18083                values.push(format!("http://example{host}.test/page/{path:04}"));
18084            }
18085        }
18086        values.push("http://example0.test/page/0000/index.htmlx".to_owned());
18087
18088        let mut dictionary = GlobalDictionary::new();
18089        for value in &values {
18090            dictionary.code(value).expect("a code for every value");
18091        }
18092        dictionary.finish_blocks().expect("the last block encodes");
18093        let ranked = dictionary.ranked(None).expect("a sorted order");
18094        assert_eq!(ranked.len(), values.len(), "one entry a distinct value");
18095
18096        let spellings = dictionary_values(&dictionary);
18097        let seen = ranked
18098            .iter()
18099            .map(|&(_, code)| {
18100                String::from_utf8(spellings[code as usize].clone()).expect("text in, text out")
18101            })
18102            .collect::<Vec<_>>();
18103        let mut wanted = values.clone();
18104        wanted.sort_unstable();
18105        assert_eq!(seen, wanted, "the order is the order the bytes give");
18106
18107        for &(carried, code) in &ranked {
18108            let value = &spellings[code as usize];
18109            assert_eq!(carried, head(value), "the head belongs to the value it is filed with");
18110        }
18111    }
18112
18113    /// Picking the commonest entries leaves exactly what sorting all of them and cutting left.
18114    ///
18115    /// The counts here are deliberately full of ties, including a tie that straddles the cut, which
18116    /// is where a partition and a sort can disagree if the comparison they are given is not total.
18117    #[test]
18118    fn the_commonest_entries_are_the_ones_a_full_sort_would_have_kept() {
18119        let entry =
18120            |value: u32, count: u64| FrequencyEntry { value: FrequencyValue::Code(value), count };
18121        let mut all = (0..FREQUENCY_ENTRIES as u32 * 3)
18122            .map(|code| entry(code, u64::from(code % 7) + 1))
18123            .collect::<Vec<_>>();
18124        all.push(FrequencyEntry { value: FrequencyValue::Null, count: 4 });
18125
18126        let mut sorted = all.clone();
18127        sorted.sort_unstable_by(|left, right| {
18128            right.count.cmp(&left.count).then_with(|| frequency_order(left.value, right.value))
18129        });
18130        let wanted_omitted = sorted[FREQUENCY_ENTRIES].count;
18131        sorted.truncate(FREQUENCY_ENTRIES);
18132
18133        let mut picked = all.clone();
18134        let omitted = keep_most_frequent(&mut picked);
18135        assert_eq!(omitted, wanted_omitted, "the largest count that did not make the cut");
18136        assert_eq!(picked.len(), FREQUENCY_ENTRIES, "the cut is where it says it is");
18137        assert!(
18138            picked
18139                .iter()
18140                .zip(&sorted)
18141                .all(|(one, two)| one.value == two.value && one.count == two.count),
18142            "the same entries in the same order"
18143        );
18144
18145        let mut short = all[..FREQUENCY_ENTRIES - 1].to_vec();
18146        let omitted = keep_most_frequent(&mut short);
18147        assert_eq!(omitted, 0, "nothing is omitted when everything fits");
18148        assert!(short.windows(2).all(|pair| pair[0].count >= pair[1].count), "still in order");
18149    }
18150
18151    /// A dictionary too small to bucket, and one with nothing in it, come back in order too.
18152    #[test]
18153    fn a_short_dictionary_sorts_without_a_bucketing_pass() {
18154        let empty = GlobalDictionary::new();
18155        assert!(empty.ranked(None).expect("an empty order").is_empty(), "nothing in, nothing out");
18156
18157        let mut dictionary = GlobalDictionary::new();
18158        for value in ["pear", "apple", "", "apples", "app"] {
18159            dictionary.code(value).expect("a code for every value");
18160        }
18161        dictionary.finish_blocks().expect("the one block encodes");
18162        let spellings = dictionary_values(&dictionary);
18163        let seen = dictionary
18164            .ranked(None)
18165            .expect("a sorted order")
18166            .iter()
18167            .map(|&(_, code)| spellings[code as usize].clone())
18168            .collect::<Vec<_>>();
18169        let wanted: Vec<Vec<u8>> =
18170            [&b""[..], b"app", b"apple", b"apples", b"pear"].iter().map(|v| v.to_vec()).collect();
18171        assert_eq!(seen, wanted, "shorter first where one runs out inside another");
18172    }
18173
18174    /// A demoted dictionary gives back what it kept for looking values up, the load profile is told,
18175    /// and it refuses any value after that.
18176    #[test]
18177    fn a_demoted_dictionary_holds_less_and_takes_no_more_values() {
18178        let profile = LoadProfile::begin("demoted");
18179        let mut dictionary = GlobalDictionary::new();
18180        for value in 0..50_000 {
18181            dictionary.code(&format!("https://example.com/page/{value}")).expect("a code");
18182        }
18183        let (_, grown) = dictionary.recharge(Some(&profile));
18184        assert_eq!(profile.held(), grown, "the profile holds what the dictionary does");
18185
18186        dictionary.demote();
18187        let (before, after) = dictionary.recharge(Some(&profile));
18188        assert_eq!(before, grown);
18189        // What stays is the ends, the counts and the blocks not yet written, which a load writes
18190        // as it goes, so here the drop is the hash tables and the check hashes.
18191        assert!(after < grown - grown / 4, "the lookup is let go of: {after} of {grown}");
18192        assert_eq!(profile.held(), after, "the profile was told about the drop");
18193        assert!(dictionary.code("one more").is_err(), "a demoted dictionary takes no values");
18194
18195        dictionary.demote();
18196        assert_eq!(
18197            dictionary.recharge(Some(&profile)),
18198            (after, after),
18199            "demoting twice is a no-op"
18200        );
18201        assert_eq!(dictionary.values(), 50_000, "the values coded before stay");
18202    }
18203}