Skip to main content

rudb_native/
lib.rs

1//! Rudb's single-file columnar snapshot format.
2//!
3//! A committed directory names independently readable column pages. It has two levels: a catalog
4//! directory naming every table in the file, which is what a footer slot points at and what opening
5//! a database reads, and one directory per table under it holding that table's stripes, pages and
6//! statistics. One slot write publishes all of them, so a commit is atomic across tables.
7//!
8//! This version handles scalar columns; the file header has two generation slots so an unfinished
9//! replacement directory cannot hide the last complete one. See
10//! `spec/storage-v3/12-many-tables-in-one-file.md`.
11//!
12//! # Parts and stripes
13//!
14//! A part is one appended chunk, which is a thousand rows, and it is the unit a scan decodes and
15//! hands to the pipeline. A stripe is sixty four parts, and it is the unit the directory describes
16//! and the unit the file is laid out in: one page per column per stripe, holding that column's
17//! sixty four part payloads end to end.
18//!
19//! The two are separate because they are sized by different pressures. A part wants to be small
20//! because it is a vector and vectors live in cache. A stripe wants to be large because everything
21//! the directory holds is per stripe and the directory is one buffer that has to be read and
22//! decoded before a single row can be answered. A hundred million rows of the hundred and five
23//! column ClickBench table is ninety seven thousand parts, and a directory with a page entry and a
24//! pair of bounds per part per column is several hundred megabytes, which is what made that load
25//! fail before this split existed. Sixty four parts to a stripe divides that by sixty four.
26//!
27//! Where the parts of a page start is not in the directory either, for the same reason. Each
28//! stripe writes one index page holding a length and a checksum per part per column, and a reader
29//! preads the sixty four entries belonging to the column it wants. A scan reads the whole column
30//! page once and slices it; a sparse row fetch reads the index entries and then only the part it
31//! needs.
32
33#![forbid(unsafe_code)]
34
35use std::borrow::Cow;
36use std::cmp::{Ordering, Reverse};
37use std::collections::{BTreeMap, HashMap, VecDeque};
38use std::fs::File;
39use std::mem::{size_of, size_of_val};
40use std::path::Path;
41use std::slice;
42use std::sync::atomic::{AtomicBool, AtomicU32, AtomicU64, AtomicUsize, Ordering as Atomic};
43use std::sync::{Arc, Condvar, Mutex, OnceLock, PoisonError, Weak};
44
45use rudb_common::bounds::{self, Bound, Op, scaled_as};
46use rudb_common::{Clustering, Error, Field, LogicalType, PhysicalType, Result, Value, Width};
47use rudb_encoding::sequence::Sequence;
48use rudb_encoding::{bitpack, chooser, integer, string};
49use rudb_io::{Filesystem, Mapped, OpenMode, RealFilesystem};
50use rudb_metrics::{LoadProfile, Stage};
51use rudb_storage::sieve::Sieve;
52use rudb_storage::{Probe, Range, Zone};
53use rudb_vector::string::StringColumn;
54use rudb_vector::validity::Validity;
55use rudb_vector::{Buffer, Chunk, Data, Packed, TextSource, Vector, search_below};
56
57mod anchor;
58mod distinct;
59pub mod grams;
60pub mod graph;
61pub mod host;
62mod prepare;
63mod projection;
64mod run_projection;
65use prepare::Lent;
66pub mod section;
67pub mod stats;
68mod zones;
69
70pub use anchor::{LaneStart, LogAnchor};
71pub use prepare::{Building, DICTIONARY_CAP_BYTES, Merged, Merger, Paged, Prepared, Preparer};
72pub use projection::build_sorted_projection;
73pub use run_projection::{RunProjectionPart, RunProjectionScan, build_run_projection};
74pub use section::Section;
75pub use zones::{Common, Stripes, ascending, distincts, facts, widths};
76
77const MAGIC: &[u8; 8] = b"RUDBNV10";
78const DIRECTORY: &[u8; 8] = b"RUDBDI10";
79const CATALOG: &[u8; 8] = b"RUDBCA10";
80const NONZERO_COUNTS: &[u8; 8] = b"RUDBNZ10";
81const AGGREGATE_SUMS: &[u8; 8] = b"RUDBAG10";
82const DISTINCT_COUNTS: &[u8; 8] = b"RUDBDC10";
83const INTEGER_EXTREMES: &[u8; 8] = b"RUDBEX10";
84const COMPLETE_FREQUENCIES: &[u8; 8] = b"RUDBFQ10";
85const DEVICE_CARD: &[u8; 8] = b"RUDBDV10";
86const MAX_CATALOG_FREQUENCIES: usize = 64;
87const FORMAT: u32 = 30;
88
89/// Formats this build can open.
90///
91/// More than one, for the first time, and the reason is spec/graph/10-milestones.md's G1 exit
92/// criterion: a build with the section table in it has to open a file written before the section
93/// table existed, unchanged and without a rewrite. Formats 22 and 23 are those files, and both read
94/// as a table with an empty section table, which is exactly what section 3.1 says a table with no
95/// graph sections is.
96///
97/// All three of the older ones are readable for the same reason. What took the format from 22 to 23
98/// was tags for fourteen more column types, and a file written before that has none of them in it,
99/// so nothing in an older file is a tag this build cannot read. What took it from 23 to 24 is the
100/// section table, which a file written before it simply does not have. What takes it from 24 to 25
101/// is the view section on the end of the catalog, which an older file does not have either, and a
102/// catalog that ends where the tables end reads as a catalog with no views in it. What takes it
103/// from 25 to 26 is that a global dictionary's payload blocks now say where they are, and a file
104/// written before that has them behind one another, which [`open_global_dictionary`] reads by
105/// turning the ends it finds into the same places the newer files name outright. What takes it
106/// from 26 to 27 is that those blocks are written into the file as the load goes, between the
107/// stripes, rather than behind the dictionary's index at the end, so the dictionary's page is the
108/// index and the sorted order and nothing else. A format 26 file has its blocks inside the page,
109/// and the reader tells the two apart by whether the page has room left over for them.
110///
111/// Format 28 adds per-payload-block substring signatures to global string dictionaries. Older
112/// files have no signatures and use the ordinary exact string filter. Format 29 makes each
113/// signature four times as wide, which a dictionary says with [`DICTIONARY_WIDE_GRAMS`], and a
114/// format 28 file is read with the narrow ones it has.
115///
116/// Format 30 lets the catalog end with the device card of the device the file is on, which a
117/// format 29 catalog has no room for and a format 29 reader would call trailing bytes. A catalog
118/// that ends before it is a file with no card, which is every older file.
119///
120/// This is not a general compatibility promise. Seven formats are readable because there was a
121/// specific reason for each, and the list shrinks again the moment the older ones stop being worth
122/// carrying.
123const READABLE: &[u32] = &[22, 23, 24, 25, 26, 27, 28, 29, FORMAT];
124
125const HEADER: u64 = 80;
126const SLOT_BYTES: usize = 28;
127const MAX_PAGE: usize = 256 * 1024 * 1024;
128const MAX_DIRECTORY: usize = 128 * 1024 * 1024;
129const FREQUENCIES_V2: &[u8; 8] = b"RUDBFQ2\0";
130const FREQUENCIES: &[u8; 8] = b"RUDBFQ3\0";
131const FREQUENCIES_SPANS: &[u8; 8] = b"RUDBFQ4\0";
132/// Inline spellings for string entries in the bounded frequency synopsis.
133///
134/// A planner usually asks about one literal such as the empty string. Without this block it opens
135/// a multi-million-value global dictionary and visits the payload blocks of every retained entry
136/// merely to compare that literal with at most 512 heavy hitters. The spellings are already in
137/// memory while the writer sorts the dictionary, so storing this bounded copy makes planning a
138/// directory read and leaves the dictionary unopened.
139const FREQUENCY_TEXTS: &[u8; 8] = b"RUDBFT1\0";
140/// Certified host aggregate state for the version-one anchored replacement expression.
141const HOST_GROUPS: &[u8; 8] = b"RUDBHG1\0";
142/// Exact leading counts for a bounded pair of dictionary-backed grouping keys.
143///
144/// This is a separate optional directory block rather than another frequency format. Readers that
145/// predate it still understand every earlier directory, and a table without a pair worth keeping
146/// writes no block at all.
147const PAIR_FREQUENCIES: &[u8; 8] = b"RUDBPF1\0";
148/// The columns whose synopsis keeps the rows of its leading entries only, with a bound on the rest.
149///
150/// A column whose listed values hold too many rows between them to keep every one of those rows
151/// keeps the rows of the longest leading run of entries that fits instead, which is what answers a
152/// grouping of it with a second column when the few commonest values are far above the others. A
153/// reader that took those rows for the rows of every listed value would bound what it left out by
154/// the synopsis bound, which is too small, so the tighter claim lives in a block of its own and a
155/// reader that predates it refuses the directory rather than trusting it.
156const ORDINAL_BOUNDS: &[u8; 8] = b"RUDBFO1\0";
157/// The clustering declaration, written after the frequencies and only when there is one.
158///
159/// No format bump for this, which is the convention the frequency section set in #728: a new
160/// optional trailing section with its own magic leaves every file that does not use it byte for
161/// byte what it was, and the version is bumped for a change to a layout that already exists, as
162/// #1029 did. A file with no declaration is the same bytes this build wrote yesterday.
163///
164/// The width byte in this block gained a fifth value for #1285, for a declaration that leaves the
165/// bucket to the row count, and that did not bump the format either. It is the one case where the
166/// reasoning needs saying out loud, because it is a new value in a layout that already exists
167/// rather than a new section. A build without it reading one of these says `clustering width
168/// tag differs` and refuses the table, which is what that message was written for. Bumping the
169/// format instead would have made every file this build writes unreadable to an older one, whether
170/// it has a declaration in it or not, to warn about a case that only arises when it does.
171const CLUSTERING: &[u8; 8] = b"RUDBCL1\0";
172/// The string columns whose global dictionary stopped taking values partway through the load.
173///
174/// Section 5.5 of the encoding spec: a column whose stripes are nearly all new values, or the
175/// fastest growing one once the dictionaries together pass their cap, stops adding to its
176/// dictionary, and every stripe after that is written plainly. The stripes before keep their codes,
177/// so the dictionary is still written and still decodes them, but it no longer holds every value of
178/// the column, and nothing that reads it as if it did can be trusted: not the distinct count, not
179/// the frequencies, not the sorted order's first and last value, and not the codes as a group key
180/// or a membership index. A reader that finds a column named here decodes its coded pages to plain
181/// strings and answers everything else the way it answers a column with no dictionary.
182///
183/// Same convention as [`CLUSTERING`], written only when a column was demoted, so a file with none
184/// is the bytes it always was. A build that predates it refuses a file that has one with
185/// `directory extension magic differs`, which is the right answer, because that build would trust
186/// the dictionary.
187///
188/// A stripe written after the demotion has no membership index for the column. Its slot in the
189/// stripe is written as a page of no bytes, which no real membership index is, since the smallest
190/// one holds its code count.
191const DEMOTED: &[u8; 8] = b"RUDBDM1\0";
192/// The graph section table, written after the clustering declaration and written even when empty.
193///
194/// Same convention and the same reason as the block above it, with one difference: this one is
195/// always there, so a file written by this build says which sections it has rather than leaving a
196/// reader to infer it from where the bytes ran out. Section 3.1 of the graph spec is what makes
197/// that safe to add without a format bump, because a table with no sections answers every query
198/// the way it did before, only without the graph path.
199const SECTIONS: &[u8; 8] = b"RUDBSE1\0";
200/// The table's primary, unique and foreign keys, written only when it has any.
201///
202/// Same convention as [`CLUSTERING`]: a table with no constraint writes no block, so every file that
203/// has none is the bytes it always was, and a build that predates the block refuses a file with one
204/// with `directory extension magic differs`. That is the right answer, because a build that dropped
205/// the keys would take a row that repeats one.
206const KEYS: &[u8; 8] = b"RUDBKY1\0";
207/// How many bytes of each column's global dictionary live outside its page, written only when any do.
208///
209/// From format 27 a dictionary's payload blocks are written into the file while the load runs, so
210/// they sit between the stripes and the dictionary's page covers only its index and sorted order.
211/// Nothing needs the total to read the file, because the index names every block. It is here for
212/// what a file costs a column, which [`Reader::layout`] and the statistics budget both report, and
213/// which would otherwise lose most of the bytes of every large string column.
214const DICTIONARY_PAYLOADS: &[u8; 8] = b"RUDBDP1\0";
215
216/// The most sections one table's directory may name.
217///
218/// A relationship contributes at most three sections, so this bounds a table at a few thousand
219/// relationships, which is far past anything a schema has. The bound is here so that a torn
220/// directory naming four billion of them is refused at decode rather than turned into an
221/// allocation, the same reason the extent count has one.
222const MAX_SECTIONS: usize = 4096;
223const FREQUENCY_CANDIDATES: usize = 32_768;
224const FREQUENCY_ENTRIES: usize = 512;
225const FREQUENCY_BUILD_RANK: usize = 10;
226const FREQUENCY_ORDINALS: usize = 131_072;
227const MAX_PAIR_FREQUENCIES: usize = 1024;
228/// The most exact heavy-hitter text one column may copy into the directory.
229///
230/// A column with unusually large leading values keeps the old code-only synopsis instead. The
231/// optimization must never turn a valid load into a directory-size failure.
232const FREQUENCY_TEXT_BUDGET: usize = 1024 * 1024;
233/// The most threads the two per column passes at the end of a commit are spread over.
234///
235/// A table like `hits` has ninety numeric columns, so on a machine with more cores than this the
236/// cap is what decides how long the frequencies take rather than the columns are. It is here at all
237/// because each worker holds a candidate table and a decoded part, and a hundred of those at once
238/// on a narrow machine would be worse than waiting.
239const MAX_FREQUENCY_WORKERS: usize = 32;
240
241/// How many threads the passes at the end of a commit are spread over on this machine.
242fn close_workers() -> usize {
243    std::thread::available_parallelism().map_or(1, usize::from).min(MAX_FREQUENCY_WORKERS)
244}
245
246/// How many bytes the columns closing at the same time may hold between them.
247///
248/// Closing a global dictionary decodes every value it holds, sorts them and drops them, and #1356
249/// took the columns one at a time so that five of them decoded at once were not the peak of a load.
250/// A numeric column's frequencies hold a candidate table and, past it, an exact set of its distinct
251/// values that reaches 512 MiB. The two used to run side by side with only the dictionaries under a
252/// bound, and on the ClickBench `hits` 10M load the close took a load that had held 3.1 GB to 4.8
253/// GB. A column is taken while the ones already closing leave room for it under this, and always
254/// when nothing else is closing, so every dictionary of `hits` at 10M rows closes at once and `URL`
255/// at 100M, which is past this alone, still closes on its own.
256const CLOSE_BYTES: usize = 1 << 30;
257
258/// What a numeric column's frequencies hold before its exact distinct set, which is the candidate
259/// table, its recount and the page being read, with room to spare.
260const NUMERIC_CLOSE_BYTES: usize = 4 << 20;
261
262/// The most threads one stripe's encode is spread over.
263///
264/// Higher than the frequency cap because this is the load itself rather than a pass at the end of
265/// it, and the work is one column of sixty four parts, which is large enough that a thread that
266/// takes one is not a thread that was started for nothing. A machine with more cores than this has
267/// the rest of them on the Parquet read, which is still one thread and is the other half of #808.
268const MAX_ENCODE_WORKERS: usize = 32;
269
270/// How much a writer appends before it asks the kernel to start writing it to the device.
271///
272/// Without it every byte of a load waits in the page cache for the sync at the commit, and that
273/// sync was 1.3 to 1.7 s of a ClickBench `hits` 10M load of 8 to 9 s on the 32 core box. With it
274/// the device writes while the load is still encoding. Thirty two megabytes is a few stripes of
275/// `hits`, big enough that the call costs nothing next to the write, and small enough that what
276/// is left for the commit is one stretch.
277const WRITEBACK_STRETCH: u64 = 32 << 20;
278
279/// The most bytes one column of one part may spend on a membership sieve.
280///
281/// A part is a thousand rows, so a filter sized for every one of them being distinct is about
282/// thirteen hundred bytes and this never binds in practice. It is here so that a part that somehow
283/// arrives much wider than a vector cannot put an unbounded index in the file. What does bind is the
284/// rule in `Writer::encode_pages` that a sieve may not be as large as the part it indexes, which is a cap
285/// per column rather than one number for the whole file.
286const SIEVE_BUDGET: usize = 8 * 1024;
287
288/// The most bytes one end of a per part range may spend on a string.
289///
290/// A bound is allowed to be wider than the truth and never narrower, so a long string is cut down to
291/// this many bytes for the low end and cut down and then stepped up for the high end. The reason for
292/// a cap at all is that there are nine hundred and seventy four parts of a hundred and five columns
293/// in a million rows of ClickBench and `URL` runs to hundreds of bytes, so keeping every end whole
294/// would put more in the directory than the skipping is worth. Twenty four bytes is past the point
295/// where two URLs of the same site still look alike.
296const PART_BOUND_BYTES: usize = 24;
297
298fn io(error: std::io::Error) -> Error {
299    Error::io(error.to_string())
300}
301
302fn invalid(message: &str) -> Error {
303    Error::invalid_input(format!("invalid rudb native file: {message}"))
304}
305
306/// Adds a sequence of byte counts without an overflow the caller has to think about.
307fn sum(counts: impl Iterator<Item = u64>) -> u64 {
308    counts.fold(0, u64::saturating_add)
309}
310
311/// One column's span out of a per column list, or zero when the list is shorter than the column.
312fn span_bytes(spans: &[Span], at: usize) -> u64 {
313    spans.get(at).map_or(0, |span| u64::from(span.length))
314}
315
316/// One column's page out of a per column list, or zero when that column has no page at all.
317fn page_bytes(pages: &[Option<Page>], at: usize) -> u64 {
318    pages.get(at).and_then(Option::as_ref).map_or(0, Page::bytes)
319}
320
321/// Everything one column's global dictionary costs the file, its page and the blocks outside it.
322fn dictionary_bytes(table: &Table, at: usize) -> u64 {
323    page_bytes(&table.dictionaries, at)
324        .saturating_add(table.dictionary_payloads.get(at).copied().unwrap_or(0))
325}
326
327/// The xxHash64 of `bytes`, which is what every span this format stores is checked against.
328///
329/// It walks the input as chunks rather than as offsets into it, and that is the only thing about it
330/// worth a comment. The offset form reads `bytes[at..at + 8]`, and neither the slicing nor the
331/// `try_into` behind it can be proved in range by a compiler that does not know where `at` stopped,
332/// so each of the four lanes paid for a bounds check and a length check on every thirty two bytes.
333/// A chunk carries its own length, so both fold away and the loop is the multiplies and rotates it
334/// was meant to be. That loop runs over every byte of every span a query reads, which on ClickBench
335/// 8 is about five percent of the query.
336fn checksum(bytes: &[u8]) -> u64 {
337    seeded_checksum(bytes, 0)
338}
339
340/// A hundred and twenty eight bit name for `bytes`, as two xxHash64 walks under different seeds,
341/// with the format this build writes folded in so that a name made by one format is never taken
342/// for the name of a file in another.
343///
344/// For a caller outside this crate that has to name a file by what went into it, which is what a
345/// Parquet mirror's key is. See the global dictionary's use of the same pair for the arithmetic.
346#[must_use]
347pub fn content_name(bytes: &[u8]) -> u128 {
348    let seed = u64::from(FORMAT);
349    u128::from(seeded_checksum(bytes, seed)) << 64 | u128::from(seeded_checksum(bytes, !seed))
350}
351
352/// [`content_name`] of bytes that arrive in pieces, which gives the same name as the pieces joined.
353///
354/// A Parquet mirror is named for the file's footer, and that is 930 KB on the ten million row
355/// ClickBench file. Read whole to be hashed it is a freed megabyte in every process that opens the
356/// mirror, which the allocator keeps. Read a window at a time it is a window.
357#[derive(Debug, Clone)]
358pub struct ContentNamer {
359    seeds: [u64; 2],
360    lanes: [[u64; 4]; 2],
361    held: [u8; 32],
362    filled: usize,
363    length: u64,
364}
365
366impl Default for ContentNamer {
367    fn default() -> Self {
368        let seed = u64::from(FORMAT);
369        let seeds = [seed, !seed];
370        let lanes = seeds.map(|seed| {
371            [
372                seed.wrapping_add(XXH_P1).wrapping_add(XXH_P2),
373                seed.wrapping_add(XXH_P2),
374                seed,
375                seed.wrapping_sub(XXH_P1),
376            ]
377        });
378        Self { seeds, lanes, held: [0; 32], filled: 0, length: 0 }
379    }
380}
381
382impl ContentNamer {
383    /// Takes the next piece.
384    pub fn update(&mut self, mut bytes: &[u8]) {
385        self.length += bytes.len() as u64;
386        if self.filled > 0 {
387            let take = (32 - self.filled).min(bytes.len());
388            self.held[self.filled..self.filled + take].copy_from_slice(&bytes[..take]);
389            self.filled += take;
390            bytes = &bytes[take..];
391            if self.filled < 32 {
392                return;
393            }
394            let block = self.held;
395            self.lanes.iter_mut().for_each(|lanes| checksum_block(lanes, &block));
396            self.filled = 0;
397        }
398        let mut blocks = bytes.chunks_exact(32);
399        for block in blocks.by_ref() {
400            self.lanes.iter_mut().for_each(|lanes| checksum_block(lanes, block));
401        }
402        let rest = blocks.remainder();
403        self.held[..rest.len()].copy_from_slice(rest);
404        self.filled = rest.len();
405    }
406
407    /// The name of everything taken so far.
408    #[must_use]
409    pub fn finish(&self) -> u128 {
410        let rest = &self.held[..self.filled];
411        let [first, second] = [0, 1].map(|at| {
412            if self.length < 32 {
413                checksum_tail(self.seeds[at].wrapping_add(XXH_P5).wrapping_add(self.length), rest)
414            } else {
415                finish_checksum(self.lanes[at], rest, self.length)
416            }
417        });
418        u128::from(first) << 64 | u128::from(second)
419    }
420}
421
422/// The xxHash64 of `bytes` started from `seed`, which is the same walk with a different beginning.
423///
424/// A seed is here for one caller: a global dictionary decides whether two values are the same by
425/// their hashes rather than by their bytes, and one sixty four bit hash is not enough to do that
426/// with. Twenty million distinct values collide on sixty four bits about once in a hundred thousand
427/// loads, which for a wrong answer is far too often. Two hashes of the same value under different
428/// seeds are independent, so the pair is a hundred and twenty eight bits and the same arithmetic
429/// puts that at around one in 1e24.
430fn seeded_checksum(bytes: &[u8], seed: u64) -> u64 {
431    // Asked for before the loop rather than after it, because a `ChunksExact` settles what it
432    // cannot divide when it is built and hands back the same tail whether it has been walked or not.
433    let mut blocks = bytes.chunks_exact(32);
434    let rest = blocks.remainder();
435    if bytes.len() < 32 {
436        return checksum_tail(seed.wrapping_add(XXH_P5).wrapping_add(bytes.len() as u64), rest);
437    }
438    let mut lanes = [
439        seed.wrapping_add(XXH_P1).wrapping_add(XXH_P2),
440        seed.wrapping_add(XXH_P2),
441        seed,
442        seed.wrapping_sub(XXH_P1),
443    ];
444    for block in blocks.by_ref() {
445        checksum_block(&mut lanes, block);
446    }
447    finish_checksum(lanes, rest, bytes.len() as u64)
448}
449
450const XXH_P1: u64 = 11_400_714_785_074_694_791;
451const XXH_P2: u64 = 14_029_467_366_897_019_727;
452const XXH_P3: u64 = 1_609_587_929_392_839_161;
453const XXH_P4: u64 = 9_650_029_242_287_828_579;
454const XXH_P5: u64 = 2_870_177_450_012_600_261;
455
456fn checksum_round(state: u64, word: u64) -> u64 {
457    state.wrapping_add(word.wrapping_mul(XXH_P2)).rotate_left(31).wrapping_mul(XXH_P1)
458}
459
460fn checksum_word(chunk: &[u8]) -> u64 {
461    u64::from_le_bytes(chunk.try_into().expect("eight checksum bytes"))
462}
463
464/// One thirty two byte block into the four lanes.
465fn checksum_block(lanes: &mut [u64; 4], block: &[u8]) {
466    for (lane, chunk) in lanes.iter_mut().zip(block.chunks_exact(8)) {
467        *lane = checksum_round(*lane, checksum_word(chunk));
468    }
469}
470
471/// The lanes after every whole block, folded together with what was left over and the length.
472fn finish_checksum(lanes: [u64; 4], rest: &[u8], length: u64) -> u64 {
473    let merge = |state: u64, lane: u64| {
474        (state ^ checksum_round(0, lane)).wrapping_mul(XXH_P1).wrapping_add(XXH_P4)
475    };
476    let [one, two, three, four] = lanes;
477    let combined = one
478        .rotate_left(1)
479        .wrapping_add(two.rotate_left(7))
480        .wrapping_add(three.rotate_left(12))
481        .wrapping_add(four.rotate_left(18));
482    let hash = merge(merge(merge(merge(combined, one), two), three), four);
483    checksum_tail(hash.wrapping_add(length), rest)
484}
485
486/// The fewer than thirty two bytes after the last whole block, and the final mix.
487fn checksum_tail(mut hash: u64, mut rest: &[u8]) -> u64 {
488    let mut words = rest.chunks_exact(8);
489    for chunk in words.by_ref() {
490        hash ^= checksum_round(0, checksum_word(chunk));
491        hash = hash.rotate_left(27).wrapping_mul(XXH_P1).wrapping_add(XXH_P4);
492    }
493    rest = words.remainder();
494    if rest.len() >= 4 {
495        let (head, tail) = rest.split_at(4);
496        let quarter = u32::from_le_bytes(head.try_into().expect("four checksum bytes"));
497        hash ^= u64::from(quarter).wrapping_mul(XXH_P1);
498        hash = hash.rotate_left(23).wrapping_mul(XXH_P2).wrapping_add(XXH_P3);
499        rest = tail;
500    }
501    for &byte in rest {
502        hash ^= u64::from(byte).wrapping_mul(XXH_P5);
503        hash = hash.rotate_left(11).wrapping_mul(XXH_P1);
504    }
505    hash ^= hash >> 33;
506    hash = hash.wrapping_mul(XXH_P2);
507    hash ^= hash >> 29;
508    hash = hash.wrapping_mul(XXH_P3);
509    hash ^ (hash >> 32)
510}
511
512/// The checksum of `length` bytes of `file` from `offset`, read [`DIRECTORY_WINDOW`] at a time.
513///
514/// The same xxHash64 as [`checksum`], carried across reads rather than over one buffer, so that a
515/// directory can be checked without all of it being in memory at once. The four lanes take whole
516/// thirty two byte blocks, and a read that ends partway through one keeps the tail for the next.
517fn file_checksum(file: &File, offset: u64, length: usize) -> Result<u64> {
518    walk_checksummed(file, offset, length, DIRECTORY_WINDOW, |_| Ok(()))
519}
520
521/// Reads `length` bytes at `offset` a window at a time, hands each window to `each`, and answers
522/// the checksum of all of them.
523///
524/// `window` is a multiple of thirty two, so every window but the last is whole blocks of the hash
525/// and nothing has to be carried from one read to the next.
526fn walk_checksummed(
527    file: &File,
528    offset: u64,
529    length: usize,
530    window: usize,
531    mut each: impl FnMut(&[u8]) -> Result<()>,
532) -> Result<u64> {
533    debug_assert!(window.is_multiple_of(32) && window > 0, "a window is whole blocks of the hash");
534    if length < 32 {
535        let mut bytes = vec![0; length];
536        read_at(file, offset, &mut bytes)?;
537        each(&bytes)?;
538        return Ok(checksum(&bytes));
539    }
540    let mut lanes = [XXH_P1.wrapping_add(XXH_P2), XXH_P2, 0, 0_u64.wrapping_sub(XXH_P1)];
541    let mut buffer = vec![0; window.min(length)];
542    let mut read = 0;
543    let (mut whole, mut filled) = (0, 0);
544    while read < length {
545        filled = buffer.len().min(length - read);
546        read_at(file, offset + read as u64, &mut buffer[..filled])?;
547        read += filled;
548        each(&buffer[..filled])?;
549        whole = filled / 32 * 32;
550        for block in buffer[..whole].chunks_exact(32) {
551            checksum_block(&mut lanes, block);
552        }
553    }
554    Ok(finish_checksum(lanes, &buffer[whole..filled], length as u64))
555}
556
557#[derive(Debug, Clone, Copy)]
558struct Slot {
559    offset: u64,
560    length: u32,
561    generation: u64,
562    hash: u64,
563}
564
565impl Slot {
566    fn bytes(self) -> [u8; SLOT_BYTES] {
567        let mut result = [0; SLOT_BYTES];
568        result[..8].copy_from_slice(&self.offset.to_le_bytes());
569        result[8..12].copy_from_slice(&self.length.to_le_bytes());
570        result[12..20].copy_from_slice(&self.generation.to_le_bytes());
571        result[20..28].copy_from_slice(&self.hash.to_le_bytes());
572        result
573    }
574
575    fn read(bytes: &[u8]) -> Self {
576        Self {
577            offset: u64::from_le_bytes(bytes[..8].try_into().expect("eight bytes")),
578            length: u32::from_le_bytes(bytes[8..12].try_into().expect("four bytes")),
579            generation: u64::from_le_bytes(bytes[12..20].try_into().expect("eight bytes")),
580            hash: u64::from_le_bytes(bytes[20..28].try_into().expect("eight bytes")),
581        }
582    }
583}
584
585#[derive(Debug, Clone, Copy)]
586struct Page {
587    offset: u64,
588    length: u32,
589    hash: u64,
590}
591
592impl Page {
593    /// How much of the file this page takes, for [`Reader::layout`].
594    fn bytes(&self) -> u64 {
595        u64::from(self.length)
596    }
597}
598
599#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
600enum FrequencyValue {
601    Null,
602    Integer(i128),
603    Code(u32),
604}
605
606/// A table keyed by the sixty four bits of the values the numeric frequency pass counts.
607///
608/// Every integer of every numeric column goes through one of these at least once when a table
609/// closes, and with the standard hasher that was a fifth of the close on its own, all of it SipHash
610/// guarding against an attacker who would have to choose the rows of the file being written.
611type FrequencyMap<V> = HashMap<u64, V, Spread>;
612
613/// The first pass of [`Writer::numeric_frequency`]: a Misra-Gries candidate table keyed by a value's
614/// sixty four bits, with the null counted beside it.
615///
616/// The table is an open addressed one of its own rather than a `HashMap`. On a column that is near
617/// unique, which `hits` has a dozen of, nearly every row is a value the table has not seen, and a
618/// `HashMap` spent a lookup and then a second hash and probe to insert it, and a `retain` over every
619/// bucket each time the table filled. Those were 6 percent of the CPU of loading the 10m ClickBench
620/// file, and the slowest of those columns decided how long the whole frequency step took. Here a
621/// value is found or given the empty slot it stopped at in one probe, and a decrement rebuilds the
622/// table from the few candidates that outlive it.
623///
624/// What the table holds after a stream of rows is the same set of counts either way, since that is
625/// fixed by the algorithm and not by where the counts live.
626#[derive(Debug)]
627struct Candidates {
628    /// A power of two number of slots, at most half of them in use. A count of zero is an empty
629    /// slot, which no candidate ever is, because one whose count reaches zero is dropped.
630    slots: Vec<Candidate>,
631    held: usize,
632    nulls: u32,
633    decrements: u64,
634    /// The candidates that outlive a decrement, kept so that each decrement is not an allocation.
635    survivors: Vec<Candidate>,
636}
637
638/// One slot of [`Candidates`], the value's bits beside its count so a probe reads one line.
639#[derive(Debug, Default, Clone, Copy)]
640struct Candidate {
641    bits: u64,
642    count: u32,
643}
644
645/// The slots a candidate table starts with, grown by doubling as it fills.
646const FIRST_CANDIDATE_SLOTS: usize = 64;
647
648impl Default for Candidates {
649    fn default() -> Self {
650        Self {
651            slots: vec![Candidate::default(); FIRST_CANDIDATE_SLOTS],
652            held: 0,
653            nulls: 0,
654            decrements: 0,
655            survivors: Vec::new(),
656        }
657    }
658}
659
660impl Candidates {
661    /// Counts `times` rows of `bits` and ends in the state `times` rows counted one at a time would.
662    ///
663    /// A value already held, or one there is room to hold, takes the whole run at once, because
664    /// every row after the first would find it held. A value the full table turns away goes a row
665    /// at a time, because each of its rows decrements every candidate and one of those decrements
666    /// can free the place the next row takes.
667    fn add(&mut self, bits: Option<u64>, mut times: u32) {
668        while times > 0 {
669            let room = self.held + usize::from(self.nulls != 0) < FREQUENCY_CANDIDATES;
670            match bits {
671                Some(bits) => {
672                    let (at, found) = self.find(bits);
673                    if found {
674                        self.slots[at].count = self.slots[at].count.saturating_add(times);
675                        return;
676                    }
677                    if room {
678                        self.place(at, bits, times);
679                        return;
680                    }
681                }
682                None if self.nulls != 0 => {
683                    self.nulls = self.nulls.saturating_add(times);
684                    return;
685                }
686                None if room => {
687                    self.nulls = times;
688                    return;
689                }
690                None => {}
691            }
692            self.decrement();
693            times -= 1;
694        }
695    }
696
697    /// The slot holding `bits` and `true`, or the empty slot a search for it stopped at and `false`.
698    fn find(&self, bits: u64) -> (usize, bool) {
699        let mask = self.slots.len() - 1;
700        let mut at = home(bits, self.slots.len());
701        loop {
702            let slot = self.slots[at];
703            if slot.count == 0 {
704                return (at, false);
705            }
706            if slot.bits == bits {
707                return (at, true);
708            }
709            at = (at + 1) & mask;
710        }
711    }
712
713    /// Where `bits` is held, for the recount, which reads the table without changing it.
714    fn position(&self, bits: u64) -> Option<usize> {
715        match self.find(bits) {
716            (at, true) => Some(at),
717            (_, false) => None,
718        }
719    }
720
721    /// Puts a new candidate in the empty slot `at`, which a search for it just stopped at, doubling
722    /// the table first when that would fill more than half of it.
723    fn place(&mut self, at: usize, bits: u64, count: u32) {
724        let at = if (self.held + 1) * 2 > self.slots.len() {
725            let wider = self.slots.len() * 2;
726            let old = std::mem::replace(&mut self.slots, vec![Candidate::default(); wider]);
727            for slot in old.into_iter().filter(|slot| slot.count != 0) {
728                let (to, _) = self.find(slot.bits);
729                self.slots[to] = slot;
730            }
731            self.find(bits).0
732        } else {
733            at
734        };
735        self.slots[at] = Candidate { bits, count };
736        self.held += 1;
737    }
738
739    /// Takes one from every candidate and the null, dropping the ones that reach zero.
740    fn decrement(&mut self) {
741        let mut survivors = std::mem::take(&mut self.survivors);
742        survivors.clear();
743        survivors.extend(
744            self.slots
745                .iter()
746                .filter(|slot| slot.count > 1)
747                .map(|slot| Candidate { bits: slot.bits, count: slot.count - 1 }),
748        );
749        self.slots.fill(Candidate::default());
750        self.held = survivors.len();
751        for &slot in &survivors {
752            let (at, _) = self.find(slot.bits);
753            self.slots[at] = slot;
754        }
755        self.survivors = survivors;
756        self.nulls = self.nulls.saturating_sub(1);
757        self.decrements = self.decrements.saturating_add(1);
758    }
759
760    /// Every candidate's bits and count, in no particular order.
761    fn pairs(&self) -> impl Iterator<Item = (u64, u32)> + '_ {
762        self.slots.iter().filter(|slot| slot.count != 0).map(|slot| (slot.bits, slot.count))
763    }
764}
765
766/// The slot a search for `bits` starts at in a table of `slots`, a power of two.
767///
768/// The top bits of a multiply by the golden ratio, which every bit of the value reaches, so a
769/// timestamp column whose values are all multiples of a million still spreads over the table.
770fn home(bits: u64, slots: usize) -> usize {
771    (bits.wrapping_mul(0x9E37_79B9_7F4A_7C15) >> (64 - slots.trailing_zeros())) as usize
772}
773
774/// Equal rows in a row, gathered so they are counted once.
775#[derive(Debug, Default)]
776struct Run {
777    bits: Option<u64>,
778    times: u32,
779}
780
781impl Run {
782    /// Adds one row, and hands back the run it ended if it was not the same value.
783    fn push(&mut self, bits: Option<u64>) -> Option<(Option<u64>, u32)> {
784        if self.times != 0 && self.bits == bits && self.times < u32::MAX {
785            self.times += 1;
786            return None;
787        }
788        let ended = self.take();
789        self.bits = bits;
790        self.times = 1;
791        ended
792    }
793
794    /// The run being gathered, if there is one, leaving none.
795    fn take(&mut self) -> Option<(Option<u64>, u32)> {
796        let times = std::mem::take(&mut self.times);
797        (times != 0).then_some((self.bits, times))
798    }
799}
800
801/// Builds the hasher for [`FrequencyMap`].
802#[derive(Debug, Default, Clone, Copy)]
803struct Spread;
804
805impl std::hash::BuildHasher for Spread {
806    type Hasher = SpreadHasher;
807
808    fn build_hasher(&self) -> SpreadHasher {
809        SpreadHasher(0)
810    }
811}
812
813/// Folds each word in with a full width multiply whose two halves are xored together.
814///
815/// A plain multiply leaves the low bits of the hash as poor as the low bits of the key, and the
816/// table picks its bucket from the low bits, so a timestamp column, whose values are all multiples
817/// of a million microseconds, would pile into a sixty fourth of the buckets. Folding the high half
818/// of the product back in is what gives the low bits the whole word.
819#[derive(Debug)]
820struct SpreadHasher(u64);
821
822impl SpreadHasher {
823    fn mix(&mut self, word: u64) {
824        let product = u128::from(self.0 ^ word) * 0x9E37_79B9_7F4A_7C15_u128;
825        self.0 = (product as u64) ^ ((product >> 64) as u64);
826    }
827}
828
829impl std::hash::Hasher for SpreadHasher {
830    fn write(&mut self, bytes: &[u8]) {
831        for part in bytes.chunks(8) {
832            let mut word = [0; 8];
833            word[..part.len()].copy_from_slice(part);
834            self.mix(u64::from_le_bytes(word));
835        }
836    }
837
838    fn write_u32(&mut self, value: u32) {
839        self.mix(u64::from(value));
840    }
841
842    fn write_u64(&mut self, value: u64) {
843        self.mix(value);
844    }
845
846    fn write_i128(&mut self, value: i128) {
847        self.mix(value as u64);
848        self.mix((value >> 64) as u64);
849    }
850
851    fn write_isize(&mut self, value: isize) {
852        self.mix(value as u64);
853    }
854
855    fn finish(&self) -> u64 {
856        self.0
857    }
858}
859
860#[derive(Debug, Clone)]
861struct FrequencyEntry {
862    value: FrequencyValue,
863    count: u64,
864}
865
866/// Exact leading frequencies for one column.
867///
868/// Values outside `entries` occur at most `omitted_max` times. This lets a count-descending TopN
869/// use the synopsis only when its last winner is strictly above every omitted value.
870#[derive(Debug, Clone)]
871struct FrequencySummary {
872    entries: Vec<FrequencyEntry>,
873    omitted_max: u64,
874    ordinals: Vec<u64>,
875    ordinal_entries: Vec<u16>,
876    /// Zero when `ordinals` holds the rows of every entry. Otherwise it holds the rows of a leading
877    /// run of them, and this is how many rows any value outside that run holds at most.
878    ordinal_bound: u64,
879}
880
881#[derive(Debug, Clone)]
882struct PairFrequencyEntry {
883    first_entry: u16,
884    second: Option<u32>,
885    count: u64,
886}
887
888/// Exact leading counts for one numeric frequency anchor and one stable string code space.
889///
890/// `omitted_max` covers both first-key values outside the numeric synopsis and pairs below the
891/// retained prefix. A TopN may therefore use the entries only when its boundary strictly exceeds
892/// this number.
893#[derive(Debug, Clone)]
894struct PairFrequencySummary {
895    first: u16,
896    second: u16,
897    entries: Vec<PairFrequencyEntry>,
898    omitted_max: u64,
899}
900
901/// The entries of a frequency synopsis and how many rows any value it left out can hold.
902type FrequencyHead = (Vec<FrequencyEntry>, u64);
903
904/// One column's frequency synopsis, in memory or left where it is in the file.
905///
906/// A writer holds what it counted. A reader leaves every synopsis in the file and reads one back
907/// when a query asks about its column, because they are the largest thing in a directory once they
908/// are decoded, forty eight bytes an entry and nearly twenty thousand entries over `hits`, and
909/// most queries ask about none of them. New directories give each synopsis a checked span, so
910/// opening an unrelated projection need not parse its ordinals.
911#[derive(Debug, Clone)]
912enum Frequencies {
913    Held(FrequencySummary),
914    /// Where the synopsis sits, and whether it was written with the value of each ordinal, which
915    /// is what the directory's frequency magic says and the synopsis itself does not.
916    Stored {
917        span: Span,
918        values: bool,
919        entries: usize,
920    },
921}
922
923/// The values one column's frequency synopsis lists, with a bound on everything it left out.
924///
925/// What [`Reader::frequency_prefix`] answers. The counts are exact, and `omitted_max` is how many
926/// rows any value not in the list can hold, which is zero when nothing was left out at all.
927#[derive(Debug, Clone)]
928pub struct FrequencyPrefix {
929    /// Every value the synopsis lists, with the number of rows holding it, count descending.
930    pub entries: Vec<(Value, u64)>,
931    /// How many rows the most common value outside the list holds, and zero for a complete list.
932    pub omitted_max: u64,
933}
934
935/// Sparse row ordinals covered by a numeric frequency candidate set.
936#[derive(Debug, Clone, PartialEq)]
937pub struct FrequencyOccurrences {
938    /// Upper bound for the frequency of every value absent from the fetched rows.
939    pub omitted_max: u64,
940    /// Table-wide row ordinals in ascending order.
941    pub ordinals: Vec<u64>,
942    /// The retained heavy-hitter values named by `anchor_indices`.
943    pub anchors: Vec<Value>,
944    /// The index in `anchors` at each ordinal, or empty for a legacy FQ2 directory.
945    pub anchor_indices: Vec<u16>,
946}
947
948/// Exact grouped counts for a pair of values, in descending count order.
949pub type PairFrequencyCounts = Vec<(Vec<Value>, u64)>;
950
951/// Where one column's page for one stripe sits in the file.
952///
953/// A column page has no checksum of its own because every part inside it carries one, and the
954/// stripe's index page holds those. Checking a part on the way out of the page covers exactly the
955/// bytes a reader is about to decode, and covers them once whether the reader took the whole page
956/// or pulled one part out of the middle of it.
957#[derive(Debug, Clone, Copy, Default)]
958struct Span {
959    offset: u64,
960    length: u32,
961}
962
963/// One optional page for each column of a stripe, holding only the pages that are there.
964///
965/// A stripe has three of these, the membership, sieve and part range pages. As a
966/// `Vec<Option<Page>>` each was thirty two bytes a column whether the page was there or not, and
967/// over the ten million rows of `hits` that is half a megabyte at open for 7171 pages out of 16380
968/// slots. Kept sparse and packed, a page that is there is twenty four bytes and one that is not is
969/// nothing.
970#[derive(Debug, Clone, Default)]
971struct Pages {
972    columns: usize,
973    held: Box<[StripePage]>,
974}
975
976/// A page and the column it is for, packed so that the column sits where the padding was.
977#[derive(Debug, Clone, Copy)]
978struct StripePage {
979    offset: u64,
980    hash: u64,
981    length: u32,
982    column: u32,
983}
984
985impl Pages {
986    /// The pages of `columns` columns, one slot each in column order.
987    fn from_slots(slots: Vec<Option<Page>>) -> Result<Self> {
988        let mut held = Vec::with_capacity(slots.iter().flatten().count());
989        for (column, page) in slots.iter().enumerate() {
990            if let Some(page) = page {
991                let column =
992                    u32::try_from(column).map_err(|_| invalid("too many columns for a page"))?;
993                held.push(StripePage {
994                    offset: page.offset,
995                    hash: page.hash,
996                    length: page.length,
997                    column,
998                });
999            }
1000        }
1001        Ok(Self { columns: slots.len(), held: held.into_boxed_slice() })
1002    }
1003
1004    /// The page of one column, if it has one.
1005    fn get(&self, column: usize) -> Option<Page> {
1006        let at = self.held.binary_search_by_key(&column, |placed| placed.column as usize).ok()?;
1007        let placed = self.held[at];
1008        Some(Page { offset: placed.offset, length: placed.length, hash: placed.hash })
1009    }
1010
1011    /// One slot per column, in column order, the way the directory writes them.
1012    fn slots(&self) -> impl Iterator<Item = Option<Page>> + '_ {
1013        (0..self.columns).map(|column| self.get(column))
1014    }
1015
1016    /// How much of the file one column's page takes, or zero when it has none.
1017    fn bytes(&self, column: usize) -> u64 {
1018        self.get(column).map_or(0, |page| page.bytes())
1019    }
1020}
1021
1022/// One independently readable stripe of a table.
1023#[derive(Debug, Clone)]
1024pub struct Stripe {
1025    rows: usize,
1026    /// Rows in each part, in source order. Kept in the directory so that mapping a row ordinal to a
1027    /// part, which every sparse fetch does, never reads the file.
1028    parts: Vec<u32>,
1029    /// The index page: one section per column, holding a length and a checksum for every part and
1030    /// then a checksum of the section itself, so that a reader can pread one column's section and
1031    /// still know it is intact.
1032    index: Span,
1033    pages: Vec<Span>,
1034    memberships: Pages,
1035    /// One page per column holding the membership sieve of every part of the stripe, for the
1036    /// columns that have one. A column whose parts all declined a sieve has no page at all.
1037    sieves: Pages,
1038    /// One page per column holding the two ends and the null count of every part of the stripe.
1039    ///
1040    /// The stripe's own `zone` below covers sixty four times as many rows, and on a column that is
1041    /// not the one the rows are ordered by that is the difference between skipping half the file and
1042    /// skipping all but three percent of it. On ClickBench 24 the cutoff the answer settles at
1043    /// leaves eight stripes of sixteen alive and thirty parts of nine hundred and seventy four.
1044    ///
1045    /// A page per column rather than one page for the stripe, so that a query that compares one
1046    /// column reads the ends of that column and not of the hundred and four beside it. Read lazily
1047    /// for the same reason, like the sieves.
1048    part_ranges: Pages,
1049    zone: Zone,
1050}
1051
1052impl Stripe {
1053    /// Number of rows in this stripe.
1054    #[must_use]
1055    pub fn rows(&self) -> usize {
1056        self.rows
1057    }
1058
1059    /// Number of parts in this stripe.
1060    #[must_use]
1061    pub fn parts(&self) -> usize {
1062        self.parts.len()
1063    }
1064
1065    /// The two ends and the null count of every column over the whole stripe.
1066    ///
1067    /// In the directory and so in memory, which is what makes it the one a planner can ask. The
1068    /// finer ones are a page per column per stripe in the file, read by [`Reader::skips`] when a
1069    /// scan wants to know which parts to open.
1070    #[must_use]
1071    pub fn zone(&self) -> &Zone {
1072        &self.zone
1073    }
1074}
1075
1076/// The committed table directory.
1077#[derive(Debug, Clone)]
1078pub struct Table {
1079    name: String,
1080    fields: Vec<Field>,
1081    stripes: Vec<Stripe>,
1082    rows: usize,
1083    dictionaries: Vec<Option<Page>>,
1084    /// Bytes of each column's dictionary payload that are outside its page, which is all of them
1085    /// from format 27 and none of them before. See [`DICTIONARY_PAYLOADS`].
1086    ///
1087    /// Empty rather than a row of zeros on a table that has none, and read with `get` for that
1088    /// reason, so that a table built by hand in a test does not have to know about it.
1089    dictionary_payloads: Vec<u64>,
1090    /// The columns whose dictionary stopped taking values partway through the load, see
1091    /// [`DEMOTED`].
1092    ///
1093    /// Empty rather than a row of `false` on a table that has none, and read with `get`, for the
1094    /// same reason `dictionary_payloads` is.
1095    demoted: Vec<bool>,
1096    frequencies: Vec<Option<Frequencies>>,
1097    /// What [`ORDINAL_BOUNDS`] says about each column, read with `get`, and zero for a column whose
1098    /// synopsis keeps the rows of every entry it lists or keeps no rows at all.
1099    ordinal_bounds: Vec<u64>,
1100    pair_frequencies: Vec<PairFrequencySummary>,
1101    /// String spellings aligned with each column's frequency entries.
1102    ///
1103    /// Empty for files written before `RUDBFT1`. A `None` entry is the null frequency entry; every
1104    /// code entry in a column named by the block has its exact bytes here.
1105    frequency_texts: Vec<Vec<Option<Vec<u8>>>>,
1106    /// Exact candidate host aggregates and an upper bound for every omitted host.
1107    host_groups: Option<host::HostSummary>,
1108    /// How many distinct values each column holds, for the columns that know.
1109    ///
1110    /// A dictionary entry is made the first time a value is seen and nothing ever removes one, so
1111    /// the size of the dictionary is the number of distinct values in the column. That is the whole
1112    /// story for a column with no null in it, and the wrong number by one for a column with a null
1113    /// in it, because a null row is written as the code for the empty string and makes an entry the
1114    /// dictionary would not otherwise have. The writer knows which case it is, since it counts the
1115    /// non-null rows that use each code while it builds the frequency summary, and the reader cannot
1116    /// work it out from the dictionary alone. So the writer settles it here.
1117    distincts: Vec<Option<u64>>,
1118    /// The order the rows of this table are meant to be stored in, if anybody declared one.
1119    ///
1120    /// A declaration and not a measurement. Nothing here checks that the stripes actually arrived
1121    /// in this order, and the reason it is worth storing anyway is that the order is the only thing
1122    /// about a table that a rewrite destroys without anybody noticing. The fragment ranges prune on
1123    /// whatever order the rows came in, so a table loaded sorted prunes and the same table after a
1124    /// checkpoint that did not know to keep the order quietly stops pruning and nothing says why.
1125    clustering: Option<Clustering>,
1126    /// The file generation of the commit that last wrote this table's column pages.
1127    ///
1128    /// This is what spec/graph/03-the-file-format.md section 3.2 calls the table generation, and
1129    /// the definition is deliberately about the pages rather than about the directory. A graph
1130    /// section is a restatement of a column in terms of row ids, so what invalidates one is the
1131    /// rows being renumbered, and nothing else. Adding a second table to the file, or attaching a
1132    /// section to this one, commits a new file generation without touching a single row of this
1133    /// table, and a definition that moved with those would declare every section in the file stale
1134    /// for no reason.
1135    ///
1136    /// Zero on a table written before format 23, where nothing recorded it. Real generations start
1137    /// at one, so zero can never match a section's stamp, and a table from format 22 has no
1138    /// sections for it to match anyway.
1139    generation: u64,
1140    /// The graph sections this table carries, per spec/graph/03-the-file-format.md section 3.2.
1141    ///
1142    /// Empty for every table written before the section table existed, and empty is not a
1143    /// degraded state: section 3.1 says deleting every graph section from a file changes no answer,
1144    /// only the time, so a table with none here answers every query the same way and slower. That
1145    /// is what lets this field arrive without a migration.
1146    sections: Vec<Section>,
1147    /// The keys and foreign keys the table was created with, which the file keeps so that a
1148    /// reopened table refuses the rows it refused before.
1149    constraints: Constraints,
1150}
1151
1152/// A table's primary, unique and foreign keys, as the file stores them.
1153///
1154/// Columns are places in the table, and a foreign key names the table it points at by name alone,
1155/// because every table in one file is in one schema.
1156#[derive(Debug, Clone, Default, PartialEq, Eq)]
1157pub struct Constraints {
1158    /// Each key's columns, and whether it is the primary key rather than a unique one.
1159    pub keys: Vec<(Vec<u16>, bool)>,
1160    /// Each foreign key.
1161    pub foreign: Vec<StoredForeign>,
1162}
1163
1164impl Constraints {
1165    /// Whether there is nothing here, which is what writes no block.
1166    #[must_use]
1167    pub fn is_empty(&self) -> bool {
1168        self.keys.is_empty() && self.foreign.is_empty()
1169    }
1170}
1171
1172/// One `FOREIGN KEY`, as the file stores it.
1173#[derive(Debug, Clone, PartialEq, Eq)]
1174pub struct StoredForeign {
1175    /// The columns of this table.
1176    pub columns: Vec<u16>,
1177    /// The table it points at.
1178    pub table: String,
1179    /// The columns of that table, paired with `columns` one for one.
1180    pub referenced: Vec<u16>,
1181}
1182
1183impl Table {
1184    /// The SQL table name held by this snapshot.
1185    #[must_use]
1186    pub fn name(&self) -> &str {
1187        &self.name
1188    }
1189
1190    /// Columns in their SQL order.
1191    #[must_use]
1192    pub fn fields(&self) -> &[Field] {
1193        &self.fields
1194    }
1195
1196    /// Committed row count.
1197    #[must_use]
1198    pub fn rows(&self) -> usize {
1199        self.rows
1200    }
1201
1202    /// Independently readable stripes.
1203    #[must_use]
1204    pub fn stripes(&self) -> &[Stripe] {
1205        &self.stripes
1206    }
1207
1208    /// The order the rows are meant to be stored in, if this table was declared with one.
1209    #[must_use]
1210    pub fn clustering(&self) -> Option<&Clustering> {
1211        self.clustering.as_ref()
1212    }
1213
1214    /// The keys and foreign keys this table was created with.
1215    #[must_use]
1216    pub fn constraints(&self) -> &Constraints {
1217        &self.constraints
1218    }
1219
1220    /// The generation every section of this table is judged against.
1221    ///
1222    /// See the field. A caller deciding whether to read a section asks [`Section::usable`] with
1223    /// this.
1224    #[must_use]
1225    pub fn generation(&self) -> u64 {
1226        self.generation
1227    }
1228
1229    /// Every graph section this table names, including the kinds this build does not know.
1230    ///
1231    /// Including them is the point. A caller that wants only the ones it can use asks
1232    /// [`Section::usable`], and a caller rewriting the directory carries the rest through, so a
1233    /// file opened by an older build and written again does not silently lose a section that build
1234    /// had no name for.
1235    #[must_use]
1236    pub fn sections(&self) -> &[Section] {
1237        &self.sections
1238    }
1239}
1240
1241/// One table's line in the catalog directory.
1242///
1243/// The small level of the two. It holds what opening a database needs and nothing else: the name to
1244/// bind, the shape to plan against, the row count, and where the table's own directory sits. A file
1245/// of eight tables is eight of these, and reading them costs the same whether the tables hold a
1246/// thousand rows or a billion.
1247///
1248/// The name, the fields and the row count are repeated here rather than pointed at inside the table
1249/// directory, which is the entire point of having two levels. A catalog that pointed at them would
1250/// have to read every table directory at open to answer what tables there are, which is the cost
1251/// this level exists to avoid.
1252#[derive(Debug, Clone)]
1253struct Entry {
1254    name: String,
1255    fields: Vec<Field>,
1256    rows: usize,
1257    /// Where this table's own directory sits, with the checksum it was committed under.
1258    directory: Page,
1259    /// Legacy nonzero counts. New files leave these empty and derive filtered counts from
1260    /// reusable column frequencies when a query needs them.
1261    nonzero: Vec<Option<u64>>,
1262    /// Exact sum and non-null count for signed integer columns.
1263    aggregates: Vec<Option<(i128, u64)>>,
1264    /// Exact non-null distinct values when the writer finished counting the column.
1265    distincts: Vec<Option<u64>>,
1266    /// Exact integer or date bounds; the inner `None` means every row is null.
1267    extremes: Vec<StoredIntegerExtremes>,
1268    /// Complete bounded numeric frequencies, including NULL when present.
1269    frequencies: Vec<StoredNumericFrequencies>,
1270}
1271
1272type StoredIntegerExtremes = Option<Option<(i128, i128)>>;
1273type StoredNumericFrequencies = Option<NumericFrequencies>;
1274
1275/// One view's line in the catalog directory.
1276///
1277/// A view has no pages, so unlike a table it is entirely here and there is no second level under it.
1278/// What it is made of is text: the body the binder binds again at every reference, and the whole
1279/// statement written back out, which is what `duckdb_views()` reports and nothing else reads.
1280///
1281/// The columns are a cache and they are written down anyway, which is worth saying out loud because
1282/// a cache in a file looks like a mistake. It is what the pin does. Create a view on a file, open
1283/// the file again in another process, and `duckdb_views()` answers `column_count` and `is_bound`
1284/// true without anything having bound the body, so the list survived the write. Not writing it
1285/// would answer null and false there, and the only way back would be to bind every view at open,
1286/// which is the thing the cache exists to avoid.
1287#[derive(Debug, Clone, PartialEq, Eq)]
1288pub struct ViewEntry {
1289    /// The view's own name, without the schema, the way a table entry holds its name.
1290    pub name: String,
1291    /// The query the view stands for, as the text that was written.
1292    pub sql: String,
1293    /// The whole `CREATE VIEW` written back out.
1294    pub statement: String,
1295    /// The column names the statement gave, which rename a prefix of what the body produces.
1296    pub aliases: Vec<String>,
1297    /// The columns the last bind of the body produced.
1298    pub columns: Vec<Field>,
1299}
1300
1301/// Where one column's bytes went, taken from the directory rather than by reading pages.
1302#[derive(Debug, Clone)]
1303pub struct ColumnLayout {
1304    /// The column's name, so a report does not have to carry the field list beside this.
1305    pub name: String,
1306    /// The type, spelled the way the catalog spells it.
1307    pub kind: String,
1308    /// Every stripe's page of this column added up, which is the encoded data itself.
1309    pub pages: u64,
1310    /// Every stripe's exact code membership page for this column.
1311    pub memberships: u64,
1312    /// Every stripe's membership sieve page for this column.
1313    pub sieves: u64,
1314    /// Every stripe's per part range page for this column.
1315    pub part_ranges: u64,
1316    /// The table wide dictionary of this column, if it has one.
1317    pub dictionary: u64,
1318}
1319
1320impl ColumnLayout {
1321    /// Everything this column costs, which is what the file would lose if the column went.
1322    #[must_use]
1323    pub fn total(&self) -> u64 {
1324        self.pages
1325            .saturating_add(self.memberships)
1326            .saturating_add(self.sieves)
1327            .saturating_add(self.part_ranges)
1328            .saturating_add(self.dictionary)
1329    }
1330}
1331
1332/// Where a whole file's bytes went.
1333///
1334/// Every number here comes out of the committed directory, so taking it costs one directory read
1335/// however large the file is. That is the point: a 45 GB table has to be able to say where it went
1336/// without being read, or nobody will ask.
1337///
1338/// The parts that are not a column are kept apart rather than shared out over the columns. The
1339/// stripe index page holds a section per column and could be split, and the directory and the
1340/// header cannot be, so splitting one of the three and not the others would read as if the columns
1341/// accounted for everything. They do not, and the gap is the thing worth looking at.
1342#[derive(Debug, Clone)]
1343pub struct Layout {
1344    /// The size of the file on disk.
1345    pub file: u64,
1346    /// Committed rows.
1347    pub rows: usize,
1348    /// Committed stripes.
1349    pub stripes: usize,
1350    /// Committed parts, which is how many chunks a scan reads.
1351    pub parts: usize,
1352    /// One entry per column, in the table's column order.
1353    pub columns: Vec<ColumnLayout>,
1354    /// Every stripe's index page, which carries a length and a checksum for every part of every
1355    /// column and is charged per stripe rather than per column.
1356    pub indexes: u64,
1357    /// The committed directory itself, the one that was read to build this.
1358    pub directory: u64,
1359    /// The fixed header, which holds the magic, the format and the two directory slots.
1360    pub header: u64,
1361}
1362
1363impl Layout {
1364    /// Everything the columns cost together.
1365    #[must_use]
1366    pub fn columns_total(&self) -> u64 {
1367        self.columns.iter().map(ColumnLayout::total).fold(0, u64::saturating_add)
1368    }
1369
1370    /// What the file holds that this does not account for.
1371    ///
1372    /// A committed file is written once and never rewritten in place, so an earlier directory and
1373    /// the pages of an earlier snapshot are still in it. That is the honest place for them: they
1374    /// are bytes on disk that no column owns.
1375    #[must_use]
1376    pub fn unaccounted(&self) -> u64 {
1377        self.file
1378            .saturating_sub(self.columns_total())
1379            .saturating_sub(self.indexes)
1380            .saturating_sub(self.directory)
1381            .saturating_sub(self.header)
1382    }
1383}
1384
1385/// How one part of one column is stored, which is one row of `pragma_storage_info`.
1386///
1387/// Everything here is read off the file rather than worked out from the schema, because the whole
1388/// question this answers is what the encoder chose, and the encoder chooses per part. Two files
1389/// holding the same rows in a different order give different answers and that difference is the
1390/// reason to ask.
1391///
1392/// The encoding costs a read of the column's page, so this is not free the way [`Layout`] is. It is
1393/// one read per column per stripe rather than one per part, because a part is a few kilobytes out
1394/// of a page that is a quarter of a megabyte.
1395#[derive(Debug, Clone)]
1396pub struct StoredPart {
1397    /// Which stripe the part belongs to.
1398    pub stripe: usize,
1399    /// Which part of that stripe it is, counting from zero inside the stripe.
1400    pub part: usize,
1401    /// The table wide row number the part starts at.
1402    pub row: usize,
1403    /// How many rows it holds.
1404    pub rows: usize,
1405    /// What the encoder made of it, as a line of text like `DICT(PACKED, PACKED)`.
1406    pub encoding: String,
1407    /// The stored bytes of the part, which is what it costs in the file.
1408    pub bytes: u64,
1409    /// Where in the file the column page holding this part starts.
1410    pub page: u64,
1411    /// Where in that page the part starts.
1412    pub offset: u64,
1413    /// The smallest value the part holds, when the stored ranges say.
1414    pub low: Option<Value>,
1415    /// The largest, same.
1416    pub high: Option<Value>,
1417    /// How many of its rows are null, when the stored ranges say.
1418    pub nulls: Option<usize>,
1419}
1420
1421/// Seeds the second hash a global dictionary tells its values apart by.
1422///
1423/// Any value that is not zero does, since zero is the seed [`checksum`] already uses and the point
1424/// is only that the two hashes of one value are not the same number. This one is the fractional part
1425/// of the golden ratio in sixty four bits, which is the constant everything else here is built out
1426/// of and is as good a nothing-up-my-sleeve number as any.
1427const DICTIONARY_CHECK_SEED: u64 = 11_400_714_819_323_198_485;
1428
1429/// One column's table wide dictionary while the load is running.
1430///
1431/// The thing to understand about this is what it does not hold. A dictionary of `URL` at a hundred
1432/// million ClickBench rows has about eighteen million distinct values and 1.3 GB of bytes in them,
1433/// and five columns like it are twelve of the seventeen gigabytes a load of `hits` peaks at. So the
1434/// bytes are not kept. A value's bytes go into [`GlobalDictionary::filling`], and when that reaches
1435/// [`TEXT_PAYLOAD_VALUES`] values the block is sealed, handed out at the end of the merge that
1436/// sealed it to be encoded with the stripe's pages, and never seen in that form again. What is left
1437/// is the encoded block, which is two to three times smaller, and that is the same bytes the file
1438/// is going to hold anyway.
1439///
1440/// Two things needed the raw bytes and neither needs them now. Deciding whether a value has been
1441/// seen before was a hash lookup and then a comparison of the bytes, and is now a hash lookup and a
1442/// comparison of a second hash under a different seed, which is [`DICTIONARY_CHECK_SEED`] and the
1443/// argument for why that is sound. Sorting the values at the end needed all of them at once, and
1444/// now reads the blocks back through [`GlobalDictionary::decoded`] one column at a time, which is
1445/// one column's bytes rather than every column's.
1446///
1447/// The offsets going block relative comes free with it, and takes the four gigabyte wall with it.
1448/// They were `u32` into a per column payload, so a column could not hold more than four gigabytes of
1449/// values however much memory the machine had, and `URL` and `Referer` are within a small factor of
1450/// that at a hundred million rows. A `u32` into a block of 1,024 values is not a bound anything real
1451/// reaches. The stored form is unchanged, because [`encode_offsets`] was already subtracting a per
1452/// block base before writing.
1453#[derive(Debug)]
1454struct GlobalDictionary {
1455    /// Keyed by the value's hash, which is already well spread, so the maps hash it once more
1456    /// with a multiply rather than with SipHash. SipHash here was one percent of a ClickBench load,
1457    /// and every stripe's merge of a column waits on the one before it.
1458    primary: HashMap<u64, u32, Spread>,
1459    collisions: HashMap<u64, Vec<u32>, Spread>,
1460    /// Every value's hash under [`DICTIONARY_CHECK_SEED`], in code order.
1461    checks: Vec<u64>,
1462    /// Where every value ends inside the payload block it is in, in code order.
1463    ends: Vec<u32>,
1464    counts: Vec<u64>,
1465    nulls: u64,
1466    /// The values of the block being filled, back to back.
1467    filling: Vec<u8>,
1468    /// One conservative four-byte substring signature per encoded payload block, in block order.
1469    ///
1470    /// Made where the block is encoded rather than where it is sealed, because sealing is under the
1471    /// writer's lock and every byte of every value going through [`gram_bits`] was 2.9 of the 14
1472    /// seconds the 10m ClickBench load spent on the 32 core box.
1473    grams: Vec<[u8; TEXT_GRAM_BYTES]>,
1474    /// Blocks that have filled and not been handed out to be encoded yet, each with its block number.
1475    ///
1476    /// Empty except inside the merge that filled them, and while the column is still too small to
1477    /// settle a shape on.
1478    waiting: Vec<(usize, Vec<u8>)>,
1479    /// Blocks kept raw to settle a shape on, spread across the column, each with its number.
1480    ///
1481    /// At most [`PAYLOAD_SAMPLE_BLOCKS`] of them and so at most a few megabytes. Spread rather than
1482    /// taken off the front for the reason [`settle_shape`] gives, and kept rather than read back
1483    /// because reading back is a decode and this is a sample of a column that is still growing.
1484    sample: Vec<(usize, Vec<u8>)>,
1485    /// How far apart the blocks in `sample` are, which doubles every time there are too many.
1486    stride: usize,
1487    /// What the blocks encoded so far were encoded with, once the column is big enough to settle it.
1488    shape: Option<chooser::Settled>,
1489    /// How many blocks had filled when that shape was settled.
1490    settled: usize,
1491    /// The blocks that are encoded and not yet in the file, in block order, following `placed`.
1492    ///
1493    /// Empty between stripes, because [`Writer::place_blocks`] writes them the moment they come
1494    /// back. Only a dictionary that never meets a writer, which is a test's, keeps them here.
1495    blocks: Vec<Vec<u8>>,
1496    /// Blocks that came back encoded ahead of a block before them, by block number.
1497    ///
1498    /// Two stripes merged one after the other can have their pages built in the other order, and a
1499    /// block cannot go into `blocks` until every block before it is there. They wait here until the
1500    /// gap closes, which is at most until the stripe merged just before this one is written.
1501    early: BTreeMap<usize, EncodedBlock>,
1502    /// Where every block already written to the file is, in block order.
1503    placed: Vec<Placed>,
1504    /// What the dictionary held the last time it was asked, see [`Self::recharge`], which is also
1505    /// what the load profile was told when there is one.
1506    charged: u64,
1507    /// Whether the dictionary stopped taking values, see [`Self::demote`].
1508    demoted: bool,
1509}
1510
1511/// Where one payload block of a global dictionary is in the file, and its checksum.
1512#[derive(Debug, Clone, Copy)]
1513struct Placed {
1514    start: u64,
1515    length: u64,
1516    hash: u64,
1517}
1518
1519/// Sorted `(head, code)` entries and the decoded bytes and block bases they were sorted over.
1520type RankedDictionary = (Vec<(u64, u32)>, Vec<u8>, Vec<u64>);
1521
1522impl GlobalDictionary {
1523    fn new() -> Self {
1524        Self {
1525            primary: HashMap::default(),
1526            collisions: HashMap::default(),
1527            checks: Vec::new(),
1528            ends: Vec::new(),
1529            counts: Vec::new(),
1530            nulls: 0,
1531            filling: Vec::new(),
1532            grams: Vec::new(),
1533            waiting: Vec::new(),
1534            sample: Vec::new(),
1535            stride: 1,
1536            shape: None,
1537            settled: 0,
1538            blocks: Vec::new(),
1539            early: BTreeMap::new(),
1540            placed: Vec::new(),
1541            charged: 0,
1542            demoted: false,
1543        }
1544    }
1545
1546    /// How many distinct values this dictionary holds, which is one past its largest code.
1547    fn values(&self) -> usize {
1548        self.ends.len()
1549    }
1550
1551    /// About how many bytes closing this dictionary holds at once: every value decoded, a sort
1552    /// entry for each, and beside it a code or a frequency candidate.
1553    fn closing_bytes(&self) -> usize {
1554        let values = self.values();
1555        let decoded = (0..values.div_ceil(TEXT_PAYLOAD_VALUES))
1556            .map(|block| self.ends[((block + 1) * TEXT_PAYLOAD_VALUES).min(values) - 1] as usize)
1557            .sum::<usize>();
1558        // The values in byte order, which is a head and a code each, and then either the codes
1559        // they were sorted as or the count and code each frequency candidate is, whichever is
1560        // larger, since the two are not held at once.
1561        let beside = size_of::<u32>().max(size_of::<(u64, Option<u32>)>());
1562        decoded.saturating_add(values.saturating_mul(size_of::<(u64, u32)>() + beside))
1563    }
1564
1565    /// About what the dictionary holds in memory, by capacity rather than by length.
1566    ///
1567    /// A hash table is charged its buckets, which is a power of two over eight sevenths of what it
1568    /// says it can hold, and a byte of control per bucket. The blocks waiting to be encoded and the
1569    /// ones kept to settle a shape on are counted one by one, and there are only ever a few.
1570    fn held_bytes(&self) -> u64 {
1571        fn table<K, V, S>(map: &HashMap<K, V, S>) -> usize {
1572            (map.capacity() * 8 / 7).next_power_of_two() * (size_of::<(K, V)>() + 1)
1573        }
1574        fn spilled<T>(values: &Vec<T>) -> usize {
1575            values.capacity() * size_of::<T>()
1576        }
1577        let raw = |blocks: &Vec<(usize, Vec<u8>)>| {
1578            spilled(blocks) + blocks.iter().map(|(_, block)| block.capacity()).sum::<usize>()
1579        };
1580        let bytes = table(&self.primary)
1581            + table(&self.collisions)
1582            + self.collisions.values().map(spilled).sum::<usize>()
1583            + spilled(&self.checks)
1584            + spilled(&self.ends)
1585            + spilled(&self.counts)
1586            + self.filling.capacity()
1587            + spilled(&self.grams)
1588            + raw(&self.waiting)
1589            + raw(&self.sample)
1590            + self.blocks.iter().map(Vec::capacity).sum::<usize>()
1591            + spilled(&self.placed);
1592        bytes as u64
1593    }
1594
1595    /// Tells `profile` what the dictionary has grown or shrunk by since the last time, and hands
1596    /// back what it held then and what it holds now.
1597    fn recharge(&mut self, profile: Option<&LoadProfile>) -> (u64, u64) {
1598        let before = self.charged;
1599        let now = self.held_bytes();
1600        if let Some(profile) = profile {
1601            if now >= before {
1602                profile.hold(now - before);
1603            } else {
1604                profile.release(before - now);
1605            }
1606        }
1607        self.charged = now;
1608        (before, now)
1609    }
1610
1611    /// Stops the dictionary taking values, for good.
1612    ///
1613    /// The block being filled is sealed so that it goes out with the others, and what the
1614    /// dictionary keeps for looking values up is let go of, which on a column of mostly new values
1615    /// is most of what it holds. What stays is what the close needs to write the dictionary's page:
1616    /// where every value ends, how often each was seen and where its blocks went. The stripes that
1617    /// were coded against it still need that page to be read. See [`DEMOTED`].
1618    fn demote(&mut self) {
1619        if self.demoted {
1620            return;
1621        }
1622        self.seal_rest();
1623        self.release_lookup();
1624        self.demoted = true;
1625    }
1626
1627    /// Frees what the dictionary keeps for coding new values, once none are coming.
1628    ///
1629    /// The hash tables, the check hash of every value and the blocks kept to settle a shape on are
1630    /// what a merge looks values up in. The close reads the counts, the ends and the written blocks
1631    /// and none of these, which are most of what the dictionary holds per value, so they go before
1632    /// the close takes memory of its own rather than after.
1633    fn release_lookup(&mut self) {
1634        self.primary = HashMap::default();
1635        self.collisions = HashMap::default();
1636        self.checks = Vec::new();
1637        self.sample = Vec::new();
1638        self.filling = Vec::new();
1639    }
1640
1641    /// How many blocks are encoded, written or not, which is the number the next one has to have.
1642    fn encoded(&self) -> usize {
1643        self.placed.len() + self.blocks.len()
1644    }
1645
1646    #[cfg(test)]
1647    fn code(&mut self, text: &str) -> Result<u32> {
1648        let bytes = text.as_bytes();
1649        self.code_hashed(bytes, checksum(bytes), seeded_checksum(bytes, DICTIONARY_CHECK_SEED))
1650    }
1651
1652    /// The code for a value whose two hashes the caller already has.
1653    ///
1654    /// A stripe prepared outside the writer's lock hashed every value it holds while it was coding
1655    /// them, and merging it into this dictionary is one of these a distinct value rather than two
1656    /// hashes of every row. See [`prepare`].
1657    fn code_hashed(&mut self, text: &[u8], hash: u64, check: u64) -> Result<u32> {
1658        if let Some(&code) = self.primary.get(&hash) {
1659            if self.checks.get(code as usize) == Some(&check) {
1660                return Ok(code);
1661            }
1662            if let Some(codes) = self.collisions.get(&hash)
1663                && let Some(code) =
1664                    codes.iter().copied().find(|&code| self.checks[code as usize] == check)
1665            {
1666                return Ok(code);
1667            }
1668            let code = self.insert(text, check)?;
1669            self.collisions.entry(hash).or_default().push(code);
1670            return Ok(code);
1671        }
1672        let code = self.insert(text, check)?;
1673        self.primary.insert(hash, code);
1674        Ok(code)
1675    }
1676
1677    fn insert(&mut self, text: &[u8], check: u64) -> Result<u32> {
1678        if self.demoted {
1679            return Err(Error::internal("a value was coded against a demoted dictionary"));
1680        }
1681        let code = u32::try_from(self.ends.len())
1682            .map_err(|_| invalid("global dictionary has too many values"))?;
1683        self.filling.extend_from_slice(text);
1684        self.ends.push(
1685            u32::try_from(self.filling.len())
1686                .map_err(|_| invalid("a global dictionary value exceeds 4 GiB"))?,
1687        );
1688        self.checks.push(check);
1689        self.counts.push(0);
1690        if self.ends.len().is_multiple_of(TEXT_PAYLOAD_VALUES) {
1691            self.seal();
1692        }
1693        Ok(code)
1694    }
1695
1696    /// Closes the block being filled and puts it in the queue to be encoded.
1697    ///
1698    /// Also keeps a copy of it if it lands on the sample's stride, and halves the sample when that
1699    /// has left too many, which is what keeps the kept blocks spread evenly over however much of the
1700    /// column exists rather than bunched at whichever end was cheap to remember.
1701    fn seal(&mut self) {
1702        let at = self.ends.len().div_ceil(TEXT_PAYLOAD_VALUES) - 1;
1703        let bytes = std::mem::take(&mut self.filling);
1704        if at.is_multiple_of(self.stride) {
1705            self.sample.push((at, bytes.clone()));
1706            if self.sample.len() > PAYLOAD_SAMPLE_BLOCKS {
1707                self.stride *= 2;
1708                let stride = self.stride;
1709                self.sample.retain(|(at, _)| at % stride == 0);
1710            }
1711        }
1712        self.waiting.push((at, bytes));
1713    }
1714
1715    /// The values of one block, as slices into the bytes the block was filled with.
1716    fn slices<'a>(&self, at: usize, bytes: &'a [u8]) -> Vec<&'a [u8]> {
1717        block_values(self.block_ends(at), bytes)
1718    }
1719
1720    /// Where every value of one block ends, relative to the block.
1721    fn block_ends(&self, at: usize) -> &[u32] {
1722        let first = (at * TEXT_PAYLOAD_VALUES).min(self.ends.len());
1723        let last = (first + TEXT_PAYLOAD_VALUES).min(self.ends.len());
1724        &self.ends[first..last]
1725    }
1726
1727    /// Takes every waiting block out to be encoded somewhere else, if the column has a shape to
1728    /// encode them with.
1729    ///
1730    /// This is what keeps the encoding out of the writer's lock. A block needs its bytes, where its
1731    /// values end and the shape, and nothing else of the dictionary, so it goes out with a copy of
1732    /// the four kilobytes of ends it has and comes back through [`GlobalDictionary::take_back`].
1733    fn hand_out(&mut self, column: usize) -> Vec<Unencoded> {
1734        let Some(shape) = &self.shape else { return Vec::new() };
1735        let waiting = std::mem::take(&mut self.waiting);
1736        waiting
1737            .into_iter()
1738            .map(|(at, bytes)| Unencoded {
1739                column,
1740                at,
1741                ends: self.block_ends(at).to_vec(),
1742                bytes,
1743                shape: shape.clone(),
1744            })
1745            .collect()
1746    }
1747
1748    /// Takes back one block that was handed out, and moves every block that is now next in line
1749    /// into `blocks`.
1750    fn take_back(&mut self, at: usize, block: EncodedBlock) -> Result<()> {
1751        if at < self.encoded() || self.early.insert(at, block).is_some() {
1752            return Err(Error::internal("a dictionary block came back twice"));
1753        }
1754        while let Some(block) = self.early.remove(&self.encoded()) {
1755            self.push_block(block);
1756        }
1757        Ok(())
1758    }
1759
1760    /// Appends the next encoded block and its signature.
1761    fn push_block(&mut self, (bytes, grams): EncodedBlock) {
1762        self.blocks.push(bytes);
1763        self.grams.push(*grams);
1764    }
1765
1766    /// Settles the shape the waiting blocks are about to be encoded with, if there is enough column
1767    /// to settle one on.
1768    ///
1769    /// Settled again once the column has grown fourfold, because the sample it was settled on then
1770    /// covered a quarter of what exists now and a dictionary in first seen order does not look the
1771    /// same at both ends. Blocks already encoded keep the shape they were encoded with. They can,
1772    /// because a block says what it is: nothing reading one asks the column what shape to expect.
1773    fn settle(&mut self) -> Result<()> {
1774        if self.sample.len() < PAYLOAD_SAMPLE_BLOCKS {
1775            return Ok(());
1776        }
1777        self.settle_on_sample()
1778    }
1779
1780    /// Settles a shape on whatever sample there is, for a column the load ended before it had
1781    /// enough of to settle one the usual way.
1782    ///
1783    /// Such a column has fewer than [`PAYLOAD_SAMPLE_BLOCKS`] blocks, so the sample is every block
1784    /// it has. Trying every candidate on each of them instead runs at two to six megabytes a second,
1785    /// and once `hits` stored its string columns with a dictionary, the forty or so small ones were
1786    /// more than half the CPU of a million row load, all of it in the close.
1787    fn settle_rest(&mut self) -> Result<()> {
1788        if self.shape.is_some() || self.sample.is_empty() {
1789            return Ok(());
1790        }
1791        self.settle_on_sample()
1792    }
1793
1794    fn settle_on_sample(&mut self) -> Result<()> {
1795        let complete = self.ends.len() / TEXT_PAYLOAD_VALUES;
1796        if self.shape.is_some() && complete < self.settled.saturating_mul(4) {
1797            return Ok(());
1798        }
1799        let sample =
1800            self.sample.iter().map(|(at, bytes)| self.slices(*at, bytes)).collect::<Vec<_>>();
1801        self.shape = Some(string::with_symbols(settle_shape(&sample)?, &sample));
1802        self.settled = complete;
1803        Ok(())
1804    }
1805
1806    /// Seals the part block at the end of the load, if there is one.
1807    fn seal_rest(&mut self) {
1808        // Asked of the values rather than of the bytes, because a block of empty strings has values
1809        // in it and no bytes, and a column of nulls is exactly that. A demoted dictionary sealed its
1810        // part block when it was demoted and has taken nothing since.
1811        if !self.demoted && !self.ends.len().is_multiple_of(TEXT_PAYLOAD_VALUES) {
1812            self.seal();
1813        }
1814    }
1815
1816    /// Encodes the waiting block at `at`, with the settled shape when there is one and by trying
1817    /// everything when the column was too small to settle one.
1818    fn encode_waiting(&self, at: usize) -> Result<EncodedBlock> {
1819        let (block, bytes) = &self.waiting[at];
1820        let values = self.slices(*block, bytes);
1821        let encoded = match &self.shape {
1822            Some(shape) => string::encode_with(&values, shape)?,
1823            None => string::encode(&values)?,
1824        };
1825        Ok((encoded, block_grams(&values)))
1826    }
1827
1828    /// [`finish_dictionaries`] for one dictionary on this thread, for the tests that hold one.
1829    #[cfg(test)]
1830    fn finish_blocks(&mut self) -> Result<()> {
1831        self.seal_rest();
1832        let made = (0..self.waiting.len())
1833            .map(|at| self.encode_waiting(at))
1834            .collect::<Result<Vec<_>>>()?;
1835        for ((at, _), block) in std::mem::take(&mut self.waiting).into_iter().zip(made) {
1836            if self.encoded() != at {
1837                return Err(Error::internal("a dictionary block was encoded out of order"));
1838            }
1839            self.push_block(block);
1840        }
1841        Ok(())
1842    }
1843
1844    /// Every value of this dictionary read back out of its encoded blocks, as the bytes back to back
1845    /// and where each block starts in them.
1846    ///
1847    /// This is the one place the whole column is in memory at once and the reason [`Writer::close`]
1848    /// takes the columns one at a time rather than across threads. One column's values is 1.3 GB on
1849    /// the worst ClickBench column, and five columns of that at once is the peak this was all meant
1850    /// to remove.
1851    ///
1852    /// The blocks are spread over threads instead. Each block's decoded length is already known from
1853    /// the ends of its values, so the answer is laid out before anything is decoded and every thread
1854    /// decodes its own run of blocks straight into its own part of it. On the 10m ClickBench sample
1855    /// this was a second of the close for `URL` alone, on one core of thirty two, and the close is
1856    /// what a load waits on once its stripes are written.
1857    ///
1858    /// The blocks already written are read back out of `file`, so what a thread holds beyond the
1859    /// answer is one encoded block. They were written moments or minutes ago and are almost always
1860    /// still in the page cache, so this is a copy rather than a read of the disk.
1861    fn decoded(&self, file: Option<&dyn rudb_io::File>) -> Result<(Vec<u8>, Vec<u64>)> {
1862        let count = self.placed.len() + self.blocks.len();
1863        if count != self.values().div_ceil(TEXT_PAYLOAD_VALUES) {
1864            return Err(invalid("global dictionary blocks do not cover its values"));
1865        }
1866        let mut bases = Vec::with_capacity(count);
1867        let mut total = 0_usize;
1868        for block in 0..count {
1869            bases.push(total as u64);
1870            let last = ((block + 1) * TEXT_PAYLOAD_VALUES).min(self.values()) - 1;
1871            total = total
1872                .checked_add(self.ends[last] as usize)
1873                .ok_or_else(|| invalid("global dictionary does not fit in memory"))?;
1874        }
1875        let mut flat = vec![0_u8; total];
1876        let mut outs = Vec::with_capacity(count);
1877        let mut rest = flat.as_mut_slice();
1878        for block in 0..count {
1879            let end = bases.get(block + 1).map_or(total, |&base| base as usize);
1880            let (out, after) = rest.split_at_mut(end - bases[block] as usize);
1881            outs.push((block, out));
1882            rest = after;
1883        }
1884        let one = |run: &mut [(usize, &mut [u8])]| -> Result<()> {
1885            let mut stored = Vec::new();
1886            for (block, out) in run {
1887                let encoded = match self.placed.get(*block) {
1888                    Some(place) => {
1889                        let file = file.ok_or_else(|| {
1890                            Error::internal("a written dictionary block has no file")
1891                        })?;
1892                        let length = usize::try_from(place.length).map_err(|_| {
1893                            invalid("global dictionary block does not fit in memory")
1894                        })?;
1895                        stored.resize(length, 0);
1896                        read_at(file, place.start, &mut stored)?;
1897                        if checksum(&stored) != place.hash {
1898                            return Err(invalid(
1899                                "a global dictionary block did not read back as written",
1900                            ));
1901                        }
1902                        stored.as_slice()
1903                    }
1904                    None => &self.blocks[*block - self.placed.len()],
1905                };
1906                let decoded = string::decode_flat(encoded)?;
1907                if decoded.bytes().len() != out.len() {
1908                    return Err(invalid(
1909                        "a global dictionary block is not the length its ends say",
1910                    ));
1911                }
1912                out.copy_from_slice(decoded.bytes());
1913            }
1914            Ok(())
1915        };
1916        // Sixteen blocks a thread at the least, because a thread costs about what decoding a few
1917        // blocks does and most columns have one or two.
1918        let workers = close_workers().min(count / 16).max(1);
1919        if workers <= 1 {
1920            one(&mut outs)?;
1921        } else {
1922            let per = count.div_ceil(workers);
1923            std::thread::scope(|scope| {
1924                outs.chunks_mut(per)
1925                    .map(|run| scope.spawn(|| one(run)))
1926                    .collect::<Vec<_>>()
1927                    .into_iter()
1928                    .try_for_each(|handle| {
1929                        handle.join().map_err(|_| {
1930                            Error::internal("a global dictionary decode worker panicked")
1931                        })?
1932                    })
1933            })?;
1934        }
1935        drop(outs);
1936        Ok((flat, bases))
1937    }
1938
1939    /// Where the value at `code` sits in the bytes [`GlobalDictionary::decoded`] handed back.
1940    ///
1941    /// A block's first value starts at the block, and every other value starts where the one before
1942    /// it ended, which is what makes 1,024 values 1,024 numbers rather than 1,025.
1943    fn value_span(ends: &[u32], bases: &[u64], code: usize) -> (usize, usize) {
1944        let Some(&base) = bases.get(code / TEXT_PAYLOAD_VALUES) else { return (0, 0) };
1945        let Some(&end) = ends.get(code) else { return (0, 0) };
1946        let base = base as usize;
1947        let from =
1948            if code.is_multiple_of(TEXT_PAYLOAD_VALUES) { 0 } else { ends[code - 1] as usize };
1949        (base + from, base + end as usize)
1950    }
1951
1952    /// This dictionary's values in sorted order, each as the first eight bytes of the value and the
1953    /// code that holds it, so entry `rank` describes the value that sits at `rank` when the values
1954    /// are sorted by their bytes.
1955    ///
1956    /// Codes themselves stay in first appearance order, which is what lets the writer hand one out
1957    /// the moment it sees a value rather than waiting for the last stripe, and which also keeps a
1958    /// stripe's codes close together because the data is clustered. This is what puts the values
1959    /// back in order for anything that needs it, and it is separate from the codes so that getting
1960    /// it costs a sort of the distinct values at the end rather than a rewrite of every code page.
1961    ///
1962    /// The order is the byte order of the values and nothing else. The heads are attached after the
1963    /// sort rather than sorted on, because padding with zero on the right is order preserving for
1964    /// byte strings and so sorting by head and then by bytes lands in the same place as sorting by
1965    /// bytes: a shorter value differs from a longer one that starts the same way at a position
1966    /// where the shorter one has run out, and zero is below every byte that could be there.
1967    ///
1968    /// The heads are kept because a reader searching this order wants a comparison it can make out
1969    /// of the index alone. What they buy there depends entirely on the column and is much less than
1970    /// it looks on the columns that cost the most, which [`sort_by_value`] measures.
1971    fn ranked_with_values(&self, file: Option<&dyn rudb_io::File>) -> Result<RankedDictionary> {
1972        let (flat, bases) = self.decoded(file)?;
1973        let value = |code: u32| {
1974            let (from, to) = Self::value_span(&self.ends, &bases, code as usize);
1975            flat.get(from..to).unwrap_or_default()
1976        };
1977        let mut codes = (0..self.values() as u32).collect::<Vec<_>>();
1978        sort_by_value_across(&mut codes, value, close_workers());
1979        let order = codes.into_iter().map(|code| (head(value(code)), code)).collect();
1980        Ok((order, flat, bases))
1981    }
1982
1983    #[cfg(test)]
1984    fn ranked(&self, file: Option<&dyn rudb_io::File>) -> Result<Vec<(u64, u32)>> {
1985        self.ranked_with_values(file).map(|(order, _, _)| order)
1986    }
1987}
1988
1989/// Appends pages and commits a new directory.
1990///
1991/// One writer covers a whole file rather than one table. [`Writer::next`] closes the table it is on
1992/// and opens another over the same file, and [`Writer::finish`] commits every table it has closed in
1993/// one generation. That is what makes a checkpoint atomic across tables: there is one slot write at
1994/// the end of it and a reader sees every table at the generation before it or every table at the
1995/// generation after it.
1996#[derive(Debug)]
1997pub struct Writer {
1998    /// The file, through `rudb-io` rather than `std::fs`, so that a test can hand the writer a
1999    /// simulated filesystem and crash a load at every call it makes.
2000    file: Box<dyn rudb_io::File>,
2001    /// Where the next write goes, counted here rather than asked of the file.
2002    ///
2003    /// The file's own cursor is not ours. Building the numeric frequencies reads pages back through
2004    /// [`read_at`], and a positional read is only positional about where it reads from: `pread`
2005    /// leaves the cursor alone, and the call Windows has for it moves the cursor to the end of what
2006    /// it read. A writer that asked the file where it was would then write the directory over a
2007    /// page it had already written, which is what it did.
2008    at: u64,
2009    /// How far into the file the kernel has been asked to start writing, see [`WRITEBACK_STRETCH`].
2010    written_back: u64,
2011    table: Table,
2012    generation: u64,
2013    /// The first and the last source position in every stripe, in the order the stripes were
2014    /// written.
2015    order: Vec<((u64, u64), (u64, u64))>,
2016    next_order: u64,
2017    dictionaries: Vec<Option<GlobalDictionary>>,
2018    /// Which columns still have a global dictionary, shared with every [`Preparer`] this writer
2019    /// hands out so that a stripe prepared after a column lost its dictionary is not coded for it.
2020    coded: Arc<prepare::Coding>,
2021    /// One per column, folding the rows into a summary and a sketch as they go past.
2022    ///
2023    /// `None` for a column with no hash rule, which is the interval and the nested types. See
2024    /// [`stats::Gather`] for why the statistics are built here rather than by reading the file back
2025    /// once it is committed.
2026    gathers: Vec<Option<stats::Gather>>,
2027    /// The dictionaries and the statistics while a [`Merger`] has them, which is from
2028    /// [`Writer::merger`] until the table is closed. `dictionaries` and `gathers` are empty then.
2029    lent: Option<Arc<Lent>>,
2030    pending: Vec<PendingChunk>,
2031    /// The tables already closed in this generation, in the order they were written.
2032    closed: Vec<Entry>,
2033    /// The views the next commit writes down, which [`Writer::with_views`] sets.
2034    ///
2035    /// Carried forward from the committed generation by [`Writer::open`], so a writer that was only
2036    /// opened to append a table does not have to know about views to avoid dropping them.
2037    views: Vec<ViewEntry>,
2038    /// The device card the next commit writes down, which is [`card_for`] the file.
2039    card: Option<KeptCard>,
2040    /// The log anchor the next commit writes down, carried forward by [`Writer::open`] and set by
2041    /// [`Writer::with_log_anchor`].
2042    anchor: Option<LogAnchor>,
2043    /// Where the stages this writer runs are charged, which [`Writer::with_profile`] sets.
2044    ///
2045    /// The writer runs the page builder, the dictionary blocks, the writes and the publish, and it
2046    /// charges them once per stripe and once per worker, never per chunk. See
2047    /// `rudb_metrics::LoadProfile` for why that is the grain.
2048    profile: Option<Arc<LoadProfile>>,
2049}
2050
2051/// A chunk that has arrived and is waiting for the rest of its stripe.
2052///
2053/// The rows are kept rather than the pages they encode to, which is the whole of #808's first half.
2054/// Encoding on arrival put every column of every part on the thread that called `append_at`, and
2055/// that thread is the only one the load has. Encoding at the flush instead means a stripe's worth
2056/// of work is on the table at once, and a stripe splits by column into a hundred and five pieces
2057/// that share nothing.
2058#[derive(Debug)]
2059struct PendingChunk {
2060    order: (u64, u64),
2061    chunk: Chunk,
2062}
2063
2064/// What the writer still needs of a part once its columns are encoded: where in the source it came
2065/// from, how many rows it has and how large those rows were.
2066///
2067/// A stripe waiting for the writer's lock carries these rather than its chunks, so its rows are
2068/// freed as soon as they are encoded and not after the stripe is written. See [`prepare`].
2069#[derive(Debug, Clone, Copy)]
2070struct Part {
2071    order: (u64, u64),
2072    rows: usize,
2073    footprint: usize,
2074}
2075
2076impl Part {
2077    fn of(pending: &PendingChunk) -> Self {
2078        Self {
2079            order: pending.order,
2080            rows: pending.chunk.len(),
2081            footprint: pending.chunk.footprint(),
2082        }
2083    }
2084}
2085
2086/// One column's share of a stripe, which is what one encode worker produces.
2087///
2088/// Indexed by part, so a stripe is a column of these and the write loop reads down one of them.
2089/// That is also the order the loop wanted: `flush_pending` walks a column at a time and lays its
2090/// parts next to each other, and it used to reach across a row of parts to do it.
2091#[derive(Debug, Default)]
2092struct ColumnStripe {
2093    pages: Vec<Vec<u8>>,
2094    /// Each page's checksum, taken where the page is built so that the writer, which holds its
2095    /// lock while it writes, does not walk every byte of the stripe a second time.
2096    sums: Vec<u64>,
2097    codes: Vec<Option<Vec<u32>>>,
2098    sieves: Vec<Option<Sieve>>,
2099    ranges: Vec<Range>,
2100}
2101
2102/// Whether a column of this type is coded against a global dictionary.
2103///
2104/// A dictionary, its codes and the membership index beside them are about bytes and not about
2105/// text, so a blob gets one the same as a varchar does. ClickBench's `hits.parquet` stores every
2106/// string column as a plain byte array, which reads back as a blob, and those columns were being
2107/// written as a length and the bytes for every row: 533 MB for the first million rows where DuckDB
2108/// writes 142.
2109fn coded_type(ty: &LogicalType) -> bool {
2110    matches!(ty, LogicalType::Varchar | LogicalType::Blob)
2111}
2112
2113/// The tag a directory gives a column's global dictionary.
2114///
2115/// A varchar's is 1, as it always was. A blob's is 2, so that a reader from before blobs had
2116/// dictionaries meets a tag it does not know and refuses the file, rather than laying the rest of
2117/// the directory out as if the blob columns had no dictionary and reading everything after the
2118/// first one from the wrong place.
2119fn dictionary_tag(ty: &LogicalType) -> u8 {
2120    if ty == &LogicalType::Blob { 2 } else { 1 }
2121}
2122
2123/// Roughly what encoding a column of this type costs, for ordering the encode queue.
2124///
2125/// Only the order matters and only roughly. A string column hashes and copies every value into a
2126/// dictionary and is in a different class from everything else, and among the fixed widths the wide
2127/// ones carry more bytes through the cascade than the narrow ones. Anything finer than that would
2128/// be a cost model, and the queue already absorbs a wrong guess: it only has to avoid finishing on
2129/// a column nobody else can help with.
2130fn weight(ty: &LogicalType) -> usize {
2131    match ty {
2132        LogicalType::Varchar | LogicalType::Blob | LogicalType::Bit => 64,
2133        LogicalType::HugeInt
2134        | LogicalType::UHugeInt
2135        | LogicalType::Uuid
2136        | LogicalType::Interval => 16,
2137        LogicalType::BigInt
2138        | LogicalType::UBigInt
2139        | LogicalType::Timestamp
2140        | LogicalType::Time
2141        | LogicalType::TimeTz
2142        | LogicalType::TimestampTz
2143        | LogicalType::TimestampS
2144        | LogicalType::TimestampMs
2145        | LogicalType::TimestampNs
2146        | LogicalType::Double
2147        | LogicalType::Decimal { .. } => 8,
2148        LogicalType::Integer | LogicalType::UInteger | LogicalType::Date | LogicalType::Float => 4,
2149        LogicalType::SmallInt | LogicalType::USmallInt => 2,
2150        _ => 1,
2151    }
2152}
2153
2154/// Parts in one stripe.
2155///
2156/// Sixty four thousand rows is the smallest stripe that keeps the ClickBench directory in single
2157/// digit megabytes at a hundred million rows, and it puts a four byte column's page at a quarter of
2158/// a megabyte, which is the size a sequential read wants. Larger stripes buy a smaller directory
2159/// and cost a sparse fetch, which has to read a page index before it can reach one part.
2160pub const STRIPE_PARTS: usize = 64;
2161
2162/// How many rows the writer wants to see before it decides whether a varchar column gets to keep
2163/// its global dictionary.
2164///
2165/// See [`prepare::drops_dictionary`]. A stripe is up to [`STRIPE_PARTS`] parts, so most tables give it
2166/// far more than this and it binds only on a table that is smaller than one stripe. A handful of
2167/// rows says nothing about whether a column repeats itself, and the answer that costs nothing when
2168/// the sample is that small is the one the writer has always given, which is to keep the dictionary.
2169const DICTIONARY_DECIDE_ROWS: usize = 4_096;
2170
2171/// Out of ten. A varchar column loses its dictionary when more than this many rows in ten of the
2172/// first stripe held a value that stripe had not seen before.
2173///
2174/// See [`prepare::drops_dictionary`]. Nine and not five, because the properties a dictionary buys are
2175/// worth keeping everywhere they are real. On ClickBench the widest string column is `Referer` at
2176/// 0.131 of its first stripe and every other one is below that, so nothing there is near this and
2177/// every one of them keeps its dictionary, which is what a group by on codes wants. On TPC-H
2178/// `o_comment` and `c_comment` are at 0.97 and are what this catches.
2179///
2180/// `l_comment` sits at 0.883 and so keeps its dictionary. Eight was built and measured rather than
2181/// argued about, and it is not a clear win: it takes `select l_comment from lineitem` from 4.335 G
2182/// instructions to 3.473 G and the file from 280.2 MB to 260.4 MB, and it takes a `like` over the
2183/// same column from 3.29 G to 4.27 G, because a dictionary runs the predicate once a distinct value
2184/// and there are 3.6 M of those to 6.0 M rows. The 22 query suite came out 6.91 s against 7.07 s in
2185/// favour of nine. So nine stays until there is a reason to prefer one of those shapes. See #1137.
2186const DICTIONARY_DISTINCT_IN_TEN: usize = 9;
2187
2188/// Bytes one part takes in a stripe's index page: four for the length, eight for the checksum.
2189const INDEX_ENTRY: usize = size_of::<u32>() + size_of::<u64>();
2190
2191/// Bytes one column's section of a stripe's index page takes, including its own trailing checksum.
2192fn index_section(parts: usize) -> Result<usize> {
2193    parts
2194        .checked_mul(INDEX_ENTRY)
2195        .and_then(|bytes| bytes.checked_add(size_of::<u64>()))
2196        .ok_or_else(|| invalid("index page length overflow"))
2197}
2198
2199impl Writer {
2200    /// Opens a committed file and starts a table in the generation after the one it holds.
2201    ///
2202    /// The tables already in the file are carried forward by name and by directory pointer, and
2203    /// their pages are not read. Nothing in the file is overwritten: the new table's pages and the
2204    /// new catalog go on the end, past the catalog the committed generation points at, and the one
2205    /// write that is not an append is the slot in the header that [`Writer::finish`] does last.
2206    ///
2207    /// That slot is the other one. A file committed at generation 1 is named by the slot at 16 and
2208    /// generation 2 writes the one at 44, so until the last four bytes of the commit land the file
2209    /// still reads as the generation before it, and a slot torn across a write fails its checksum
2210    /// and the reader falls back to the one beside it. This is what the second slot has always been
2211    /// for.
2212    ///
2213    /// # Errors
2214    ///
2215    /// If the file has no valid committed directory, is not this build's format, repeats the name
2216    /// of a table already in it that holds rows, has a field with no scalar encoding, or cannot be
2217    /// written.
2218    pub fn open(
2219        path: impl AsRef<Path>,
2220        name: impl Into<String>,
2221        fields: Vec<Field>,
2222    ) -> Result<Self> {
2223        Self::open_in(&RealFilesystem::new(), path, name, fields)
2224    }
2225
2226    /// [`Writer::open`] on a file in `fs`, which is how a crash test runs an append against the
2227    /// simulated filesystem.
2228    ///
2229    /// # Errors
2230    ///
2231    /// The same as [`Writer::open`].
2232    pub fn open_in(
2233        fs: &dyn Filesystem,
2234        path: impl AsRef<Path>,
2235        name: impl Into<String>,
2236        fields: Vec<Field>,
2237    ) -> Result<Self> {
2238        for field in &fields {
2239            type_tag(&field.ty)?;
2240        }
2241        let name = name.into();
2242        let file = fs.open(path.as_ref(), OpenMode::ReadWrite)?;
2243        let size = file.len()?;
2244        let (slot, bytes, _) = committed_slot(&*file, size)?;
2245        let (mut closed, views, card, anchor) = decode_catalog(&bytes, size)?;
2246        let card = card_for(path.as_ref(), card);
2247        // A table already in the file under this name is only in the way if it holds rows. One that
2248        // holds none has no pages for this generation to carry and no reader that could lose
2249        // anything, so the table being started here takes its place in the catalog rather than
2250        // colliding with it, and `finish` writes the new entry where the old one was.
2251        //
2252        // That is not a corner. It is the shape every loading script writes: the schema goes in one
2253        // statement and the rows go in the next, and a checkpoint between them commits the empty
2254        // table. Before this, the second statement had to build the whole table in memory because
2255        // the first had already put the name in the file, which is how a load of a table larger
2256        // than memory became a load that needed memory the size of the table.
2257        if let Some(at) = closed.iter().position(|held| held.name == name) {
2258            if closed[at].rows > 0 {
2259                return Err(invalid("two tables in one native file have the same name"));
2260            }
2261            closed.remove(at);
2262        }
2263        // The generation of the slot whose bytes checksummed, and not the highest number in the
2264        // header. A slot torn across a write can hold any number at all, and taking that one would
2265        // be choosing which slot to overwrite from a value nothing has vouched for, which is how a
2266        // half written commit gets to destroy the one good copy beside it.
2267        let generation = slot
2268            .generation
2269            .checked_add(1)
2270            .ok_or_else(|| invalid("native file generation overflow"))?;
2271        Ok(Self {
2272            file,
2273            // The end of the file, so that the committed generation's catalog stays where its slot
2274            // says it is and keeps naming a file a reader can still open.
2275            at: size,
2276            written_back: size,
2277            dictionaries: fields
2278                .iter()
2279                .map(|field| coded_type(&field.ty).then(GlobalDictionary::new))
2280                .collect(),
2281            coded: Arc::new(prepare::Coding::new(fields.iter().map(|field| coded_type(&field.ty)))),
2282            gathers: fields.iter().map(|field| stats::Gather::new(&field.ty, generation)).collect(),
2283            lent: None,
2284            table: Table {
2285                name,
2286                dictionaries: vec![None; fields.len()],
2287                dictionary_payloads: Vec::new(),
2288                demoted: Vec::new(),
2289                distincts: vec![None; fields.len()],
2290                fields,
2291                stripes: Vec::new(),
2292                rows: 0,
2293                frequencies: Vec::new(),
2294                ordinal_bounds: Vec::new(),
2295                pair_frequencies: Vec::new(),
2296                frequency_texts: Vec::new(),
2297                host_groups: None,
2298                clustering: None,
2299                constraints: Constraints::default(),
2300                generation,
2301                sections: Vec::new(),
2302            },
2303            generation,
2304            order: Vec::new(),
2305            next_order: 0,
2306            pending: Vec::with_capacity(STRIPE_PARTS),
2307            closed,
2308            views,
2309            card,
2310            anchor,
2311            profile: None,
2312        })
2313    }
2314
2315    /// Creates a new v10 file and its first table.
2316    ///
2317    /// # Errors
2318    ///
2319    /// If the file exists, a field has no scalar encoding, or the path cannot be written.
2320    pub fn create(
2321        path: impl AsRef<Path>,
2322        name: impl Into<String>,
2323        fields: Vec<Field>,
2324    ) -> Result<Self> {
2325        Self::create_in(&RealFilesystem::new(), path, name, fields)
2326    }
2327
2328    /// [`Writer::create`] with the file made in `fs` rather than on the real filesystem.
2329    ///
2330    /// Every call the writer makes on the file from here to [`Writer::finish`] goes to that
2331    /// filesystem, which is what lets a test built on `rudb_io::SimFilesystem` stop a load at any
2332    /// one of them and look at what a crash there would leave on the disk.
2333    ///
2334    /// # Errors
2335    ///
2336    /// The same as [`Writer::create`].
2337    pub fn create_in(
2338        fs: &dyn Filesystem,
2339        path: impl AsRef<Path>,
2340        name: impl Into<String>,
2341        fields: Vec<Field>,
2342    ) -> Result<Self> {
2343        for field in &fields {
2344            type_tag(&field.ty)?;
2345        }
2346        let file = fs.open(path.as_ref(), OpenMode::CreateNew)?;
2347        let mut header = [0; HEADER as usize];
2348        header[..8].copy_from_slice(MAGIC);
2349        header[8..12].copy_from_slice(&FORMAT.to_le_bytes());
2350        file.write_at(0, &header)?;
2351        Ok(Self {
2352            file,
2353            at: HEADER,
2354            written_back: HEADER,
2355            dictionaries: fields
2356                .iter()
2357                .map(|field| coded_type(&field.ty).then(GlobalDictionary::new))
2358                .collect(),
2359            coded: Arc::new(prepare::Coding::new(fields.iter().map(|field| coded_type(&field.ty)))),
2360            gathers: fields.iter().map(|field| stats::Gather::new(&field.ty, 1)).collect(),
2361            lent: None,
2362            table: Table {
2363                name: name.into(),
2364                dictionaries: vec![None; fields.len()],
2365                dictionary_payloads: Vec::new(),
2366                demoted: Vec::new(),
2367                distincts: vec![None; fields.len()],
2368                fields,
2369                stripes: Vec::new(),
2370                rows: 0,
2371                frequencies: Vec::new(),
2372                ordinal_bounds: Vec::new(),
2373                pair_frequencies: Vec::new(),
2374                frequency_texts: Vec::new(),
2375                host_groups: None,
2376                clustering: None,
2377                constraints: Constraints::default(),
2378                generation: 1,
2379                sections: Vec::new(),
2380            },
2381            generation: 1,
2382            order: Vec::new(),
2383            next_order: 0,
2384            pending: Vec::with_capacity(STRIPE_PARTS),
2385            closed: Vec::new(),
2386            views: Vec::new(),
2387            card: card_for(path.as_ref(), None),
2388            anchor: None,
2389            profile: None,
2390        })
2391    }
2392
2393    /// Creates a new file that holds no table at all, committed and ready to open.
2394    ///
2395    /// A database somebody dropped the last table out of is still a database, and until this there
2396    /// was no way to write one down. Every other way into this file goes through a table, because
2397    /// [`Writer::create`] takes the first one and [`Writer::finish`] commits the one it is on, so a
2398    /// catalog with nothing in it could be read and not written. The format already allowed it: the
2399    /// catalog is a count and that many entries, and a count of nought encodes and decodes the same
2400    /// way every other count does, which is why nothing here is a version change.
2401    ///
2402    /// It hands back nothing rather than a writer, because a writer with no table is a writer with
2403    /// nothing to append to. A file that is going to hold a table is [`Writer::create`], and one
2404    /// that is going to have a table added to it later is [`Writer::open`], which reads what this
2405    /// wrote the same way it reads any other generation.
2406    ///
2407    /// It takes the views anyway, because a database with no table can still have views in it. A
2408    /// view over `range` or over another view names no table, so dropping the last table out of a
2409    /// database does not have to leave the catalog with nothing worth writing down. The log anchor
2410    /// is the same: the log a database with no table wrote is still a log the file has to account
2411    /// for.
2412    ///
2413    /// # Errors
2414    ///
2415    /// If the file exists or the path cannot be written.
2416    pub fn empty(
2417        path: impl AsRef<Path>,
2418        views: &[ViewEntry],
2419        anchor: Option<&LogAnchor>,
2420    ) -> Result<()> {
2421        let file = RealFilesystem::new().open(path.as_ref(), OpenMode::CreateNew)?;
2422        let mut header = [0; HEADER as usize];
2423        header[..8].copy_from_slice(MAGIC);
2424        header[8..12].copy_from_slice(&FORMAT.to_le_bytes());
2425        file.write_at(0, &header)?;
2426        let catalog = encode_catalog(&[], views, card_for(path.as_ref(), None).as_ref(), anchor)?;
2427        file.write_at(HEADER, &catalog)?;
2428        // The same two syncs in the same order as [`Writer::finish`], and for the same reason. The
2429        // catalog is on the disk before the slot names it, so a file this is interrupted in the
2430        // middle of is a header with no valid slot rather than a slot pointing at nothing.
2431        file.sync()?;
2432        let slot = Slot {
2433            offset: HEADER,
2434            length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
2435            generation: 1,
2436            hash: checksum(&catalog),
2437        };
2438        file.write_at(slot_offset(1), &slot.bytes())?;
2439        file.sync()?;
2440        Ok(())
2441    }
2442
2443    /// Closes the table this writer is on and starts another one in the same file.
2444    ///
2445    /// Nothing is published here. The closed table's directory is written so that the bytes are on
2446    /// disk and its span is known, and the catalog that names it is only written by
2447    /// [`Writer::finish`], so a crash between two tables leaves the previous generation intact.
2448    ///
2449    /// # Errors
2450    ///
2451    /// If the name repeats a table already closed, a field has no scalar encoding, or the table
2452    /// being closed cannot be written.
2453    pub fn next(mut self, name: impl Into<String>, fields: Vec<Field>) -> Result<Self> {
2454        for field in &fields {
2455            type_tag(&field.ty)?;
2456        }
2457        let name = name.into();
2458        let entry = self.close()?;
2459        if entry.name == name {
2460            return Err(invalid("two tables in one native file have the same name"));
2461        }
2462        // An empty table the committed generation holds under this name steps aside for this one,
2463        // the same as it does for the first table in [`Writer::open`], and for the same reason: it
2464        // has no pages to carry and the load writing it now is the one that fills it.
2465        if let Some(at) = self.closed.iter().position(|held| held.name == name) {
2466            if self.closed[at].rows > 0 {
2467                return Err(invalid("two tables in one native file have the same name"));
2468            }
2469            self.closed.remove(at);
2470        }
2471        let Self { file, at, generation, mut closed, views, card, anchor, .. } = self;
2472        closed.push(entry);
2473        Ok(Self {
2474            file,
2475            written_back: at,
2476            at,
2477            generation,
2478            closed,
2479            views,
2480            card,
2481            anchor,
2482            profile: None,
2483            dictionaries: fields
2484                .iter()
2485                .map(|field| coded_type(&field.ty).then(GlobalDictionary::new))
2486                .collect(),
2487            coded: Arc::new(prepare::Coding::new(fields.iter().map(|field| coded_type(&field.ty)))),
2488            gathers: fields.iter().map(|field| stats::Gather::new(&field.ty, generation)).collect(),
2489            lent: None,
2490            table: Table {
2491                name,
2492                dictionaries: vec![None; fields.len()],
2493                dictionary_payloads: Vec::new(),
2494                demoted: Vec::new(),
2495                distincts: vec![None; fields.len()],
2496                fields,
2497                stripes: Vec::new(),
2498                rows: 0,
2499                frequencies: Vec::new(),
2500                ordinal_bounds: Vec::new(),
2501                pair_frequencies: Vec::new(),
2502                frequency_texts: Vec::new(),
2503                host_groups: None,
2504                clustering: None,
2505                constraints: Constraints::default(),
2506                generation,
2507                sections: Vec::new(),
2508            },
2509            order: Vec::new(),
2510            next_order: 0,
2511            pending: Vec::with_capacity(STRIPE_PARTS),
2512        })
2513    }
2514
2515    /// Sets the views the next commit writes down, replacing whatever was carried forward.
2516    ///
2517    /// It replaces rather than adds because the caller has the whole catalog in front of it and the
2518    /// writer does not. A view that was dropped is a view that is not in the list any more, and
2519    /// there is no other way for the writer to hear about that, since nothing else it is told about
2520    /// mentions views at all.
2521    ///
2522    /// A writer that is never told anything writes back the views it read at [`Writer::open`], so a
2523    /// checkpoint that only had a table to append does not quietly drop them.
2524    #[must_use]
2525    pub fn with_views(mut self, views: Vec<ViewEntry>) -> Self {
2526        self.views = views;
2527        self
2528    }
2529
2530    /// Sets the log anchor the next commit writes down, which says how much of the log the file
2531    /// holds once it is published.
2532    #[must_use]
2533    pub fn with_log_anchor(mut self, anchor: LogAnchor) -> Self {
2534        self.anchor = Some(anchor);
2535        self
2536    }
2537
2538    /// Charges the stages this writer runs to `profile`.
2539    ///
2540    /// For the table being written now. [`Writer::next`] starts the next table without one,
2541    /// because a second table's stripes charged to the first table's load would be a profile of
2542    /// neither.
2543    #[must_use]
2544    pub fn with_profile(mut self, profile: Arc<LoadProfile>) -> Self {
2545        self.profile = Some(profile);
2546        self
2547    }
2548
2549    /// Sets what the table's global dictionaries may hold between them before the one growing
2550    /// fastest stops taking values, which is [`DICTIONARY_CAP_BYTES`] unless this says
2551    /// otherwise. It applies to every [`Preparer`] and [`Merger`] this writer has handed out too.
2552    #[must_use]
2553    pub fn with_dictionary_cap(self, bytes: u64) -> Self {
2554        self.coded.cap(bytes);
2555        self
2556    }
2557
2558    /// Records the order this table's rows are meant to be stored in.
2559    ///
2560    /// The declaration goes in the table directory and comes back out of
2561    /// [`Table::clustering`]. Nothing here sorts anything, and nothing here checks that the rows
2562    /// handed to [`Writer::append`] arrive in the order this claims. That is deliberate for now:
2563    /// the thing that was missing was a place to write the order down, and a loader that honours
2564    /// the declaration is the next piece rather than this one.
2565    ///
2566    /// The declaration applies to the table the writer is currently on, so it is set after
2567    /// [`Writer::next`] rather than once for the file.
2568    ///
2569    /// # Errors
2570    ///
2571    /// If the declaration names a column this table does not have.
2572    pub fn declare(mut self, clustering: Clustering) -> Result<Self> {
2573        // Rebuilt against this table's own column count rather than trusted, because the caller
2574        // built it against a catalog entry and the two could have drifted.
2575        self.table.clustering = Some(Clustering::new(
2576            clustering.columns().to_vec(),
2577            clustering.width(),
2578            &self.table.fields,
2579        )?);
2580        Ok(self)
2581    }
2582
2583    /// Records the keys and foreign keys of the table the writer is on, which come back out of
2584    /// [`Table::constraints`]. Nothing here checks the rows against them, since the catalog already
2585    /// did before it let the rows in.
2586    ///
2587    /// # Errors
2588    ///
2589    /// If a key or a foreign key names a column this table does not have, or has no columns.
2590    pub fn constrain(mut self, constraints: Constraints) -> Result<Self> {
2591        let width = self.table.fields.len();
2592        let fits = |columns: &[u16]| {
2593            !columns.is_empty() && columns.iter().all(|&column| usize::from(column) < width)
2594        };
2595        if !constraints.keys.iter().all(|(columns, _)| fits(columns))
2596            || !constraints.foreign.iter().all(|foreign| {
2597                fits(&foreign.columns) && foreign.referenced.len() == foreign.columns.len()
2598            })
2599        {
2600            return Err(invalid("a constraint names a column the table does not have"));
2601        }
2602        self.table.constraints = constraints;
2603        Ok(self)
2604    }
2605
2606    /// Appends bytes at the end of the file and moves the writer's own offset past them.
2607    ///
2608    /// Every write in here goes through this, so that [`Writer::at`] is the only answer to where
2609    /// anything is and the file's cursor is never consulted for it.
2610    fn put(&mut self, bytes: &[u8]) -> Result<()> {
2611        self.file.write_at(self.at, bytes)?;
2612        self.at = self
2613            .at
2614            .checked_add(bytes.len() as u64)
2615            .ok_or_else(|| invalid("native file length overflow"))?;
2616        if self.at - self.written_back >= WRITEBACK_STRETCH {
2617            self.file.start_writeback(self.written_back, self.at - self.written_back);
2618            self.written_back = self.at;
2619        }
2620        Ok(())
2621    }
2622
2623    /// Writes one chunk as independently readable column pages.
2624    ///
2625    /// # Errors
2626    ///
2627    /// If its width or types differ from the declared table, or a page exceeds its bound.
2628    pub fn append(&mut self, chunk: &Chunk) -> Result<()> {
2629        let order = (self.next_order, 0);
2630        self.next_order = self.next_order.saturating_add(1);
2631        self.append_at(order, chunk)
2632    }
2633
2634    /// Writes one chunk and records its source position for directory ordering.
2635    ///
2636    /// Pages may be encoded by parallel pipeline instances and reach the file in completion order.
2637    /// The stripe they land in is sorted by this key at commit, and [`Self::finish`] rejects a
2638    /// sequence whose parts do not come out in source order once the stripes are sorted, because a
2639    /// stripe groups whatever arrived together and cannot put a late part back where it belongs.
2640    ///
2641    /// # Errors
2642    ///
2643    /// The same as [`Self::append`].
2644    pub fn append_at(&mut self, order: (u64, u64), chunk: &Chunk) -> Result<()> {
2645        if chunk.is_empty() {
2646            return Ok(());
2647        }
2648        self.admit(chunk)?;
2649        if self.pending.last().is_some_and(|last| last.order > order) {
2650            self.flush_pending()?;
2651        }
2652        // Cloned rather than encoded, and a clone of a chunk that owns its buffers is a copy of
2653        // them. Sixty four parts of a hundred and five columns is tens of megabytes held for the
2654        // length of a stripe and a few seconds of memory traffic over a whole ClickBench load,
2655        // against the hundreds of seconds of encode this is what lets off one thread.
2656        self.pending.push(PendingChunk { order, chunk: chunk.clone() });
2657        if self.pending.len() == STRIPE_PARTS {
2658            self.flush_pending()?;
2659        }
2660        Ok(())
2661    }
2662
2663    /// Writes a run of chunks as one stripe of its own.
2664    ///
2665    /// [`Self::append_at`] decides where a stripe ends by watching the orders go past, which works
2666    /// when one caller hands over every chunk in source order and does not when several do. A
2667    /// writer being fed by more than one pipeline instance sees the orders interleave, and a stripe
2668    /// that ends every time two of them cross is a stripe of one or two parts.
2669    ///
2670    /// So the grouping moves to the caller. Whoever is buffering hands over a run it already knows
2671    /// is contiguous and in order, and gets a stripe holding exactly that run. The orders still
2672    /// have to come out in source order once the stripes are sorted, which [`Self::finish`] checks,
2673    /// so the runs from different callers may interleave with each other but may not overlap.
2674    ///
2675    /// # Errors
2676    ///
2677    /// The same as [`Self::append`], and if the run is longer than [`STRIPE_PARTS`].
2678    pub fn append_stripe(&mut self, parts: Vec<((u64, u64), Chunk)>) -> Result<()> {
2679        if parts.len() > STRIPE_PARTS {
2680            return Err(invalid("a stripe was handed more parts than it holds"));
2681        }
2682        // Whatever an earlier caller left behind is its own stripe rather than the front of this
2683        // one, because the two runs are from different places in the source and a stripe is a run.
2684        self.flush_pending()?;
2685        for (order, chunk) in parts {
2686            if chunk.is_empty() {
2687                continue;
2688            }
2689            self.admit(&chunk)?;
2690            self.pending.push(PendingChunk { order, chunk });
2691        }
2692        self.flush_pending()
2693    }
2694
2695    /// Checks a chunk against the declared table and counts its rows in.
2696    fn admit(&mut self, chunk: &Chunk) -> Result<()> {
2697        if chunk.width() != self.table.fields.len() {
2698            return Err(invalid("chunk width differs from table schema"));
2699        }
2700        for (index, field) in self.table.fields.iter().enumerate() {
2701            if chunk.column(index)?.logical_type() != &field.ty {
2702                return Err(invalid("chunk type differs from table schema"));
2703            }
2704        }
2705        self.table.rows = self
2706            .table
2707            .rows
2708            .checked_add(chunk.len())
2709            .ok_or_else(|| invalid("row count overflow"))?;
2710        Ok(())
2711    }
2712
2713    /// One column's parts of a stripe as pages, for a column with no global dictionary.
2714    fn encode_pages(columns: &[&Vector]) -> Result<ColumnStripe> {
2715        let mut stripe = ColumnStripe {
2716            pages: Vec::with_capacity(columns.len()),
2717            sums: Vec::with_capacity(columns.len()),
2718            codes: Vec::with_capacity(columns.len()),
2719            sieves: Vec::with_capacity(columns.len()),
2720            ranges: Vec::with_capacity(columns.len()),
2721        };
2722        let mut settling = Settling::default();
2723        for &column in columns {
2724            Self::encode_page(&mut stripe, &mut settling, column)?;
2725        }
2726        Ok(stripe)
2727    }
2728
2729    /// One more part of a column with no global dictionary as a page, after the ones already in
2730    /// `stripe`. The parts have to come in order, since `settling` carries from one to the next.
2731    fn encode_page(
2732        stripe: &mut ColumnStripe,
2733        settling: &mut Settling,
2734        column: &Vector,
2735    ) -> Result<()> {
2736        let bytes = encode(column, settling)?;
2737        if bytes.len() > MAX_PAGE {
2738            return Err(invalid("column page exceeds the configured bound"));
2739        }
2740        // The range is built first because the sieve reads it rather than walking the column a
2741        // second time to find out how wide it is.
2742        let range = Range::of(column);
2743        // A sieve at least as large as the part it indexes is not written. A reader reads the
2744        // sieve to decide whether to read the part, so when the sieve is the larger of the two
2745        // it has already spent more than the read it is trying to avoid, and that holds even if
2746        // it rejects every time. It is a necessary condition rather than the whole rule, which
2747        // is that a sieve pays when its bytes are under the rejection rate times the part's,
2748        // but the rejection rate depends on what a query probes for and the writer does not
2749        // know that. The necessary half needs two numbers that are both in hand here.
2750        //
2751        // A column with a global dictionary gets none, because it already has an exact
2752        // membership index per stripe. Those do not come through here. See [`prepare`].
2753        let sieve =
2754            Sieve::of(column, &range, SIEVE_BUDGET).filter(|sieve| sieve.len() < bytes.len());
2755        stripe.sums.push(checksum(&bytes));
2756        stripe.pages.push(bytes);
2757        stripe.codes.push(None);
2758        stripe.sieves.push(sieve);
2759        stripe.ranges.push(range);
2760        Ok(())
2761    }
2762
2763    /// Writes every encoded dictionary block that is not in the file yet and forgets its bytes.
2764    ///
2765    /// This is what keeps a load from holding its dictionaries' payload. The blocks land between
2766    /// stripes wherever the writer is, which is fine because the index says where each one is.
2767    fn place_blocks(&mut self) -> Result<()> {
2768        if let Some(lent) = self.lent.clone() {
2769            return self.place_lent_blocks(&lent);
2770        }
2771        let mut dictionaries = std::mem::take(&mut self.dictionaries);
2772        let placed = dictionaries.iter_mut().flatten().try_for_each(|dictionary| {
2773            for block in std::mem::take(&mut dictionary.blocks) {
2774                let start = self.at;
2775                self.put(&block)?;
2776                dictionary.placed.push(Placed {
2777                    start,
2778                    length: block.len() as u64,
2779                    hash: checksum(&block),
2780                });
2781            }
2782            Ok(())
2783        });
2784        self.dictionaries = dictionaries;
2785        placed
2786    }
2787
2788    /// [`Writer::place_blocks`] while a [`Merger`] has the dictionaries.
2789    ///
2790    /// A column whose merge is running is passed over rather than waited for, because the writer's
2791    /// lock is held here and a merge of `URL` can take tens of milliseconds. Its blocks go out with
2792    /// a later stripe, or at the close.
2793    fn place_lent_blocks(&mut self, lent: &Lent) -> Result<()> {
2794        for column in lent.columns() {
2795            let Ok(mut held) = column.try_lock() else { continue };
2796            let Some(dictionary) = held.dictionary.as_mut() else { continue };
2797            for block in std::mem::take(&mut dictionary.blocks) {
2798                let start = self.at;
2799                self.put(&block)?;
2800                dictionary.placed.push(Placed {
2801                    start,
2802                    length: block.len() as u64,
2803                    hash: checksum(&block),
2804                });
2805            }
2806        }
2807        Ok(())
2808    }
2809
2810    /// Takes the dictionaries and the statistics back from the [`Merger`] that has them.
2811    ///
2812    /// A merge that starts after this is refused, since whatever it merged would be lost.
2813    fn reclaim(&mut self) -> Result<()> {
2814        let Some(lent) = self.lent.take() else { return Ok(()) };
2815        let (dictionaries, gathers) = lent.reclaim()?;
2816        self.dictionaries = dictionaries;
2817        self.gathers = gathers;
2818        Ok(())
2819    }
2820
2821    /// Writes the buffered parts as one stripe, each column's parts contiguous on disk.
2822    ///
2823    /// The same four steps a caller holding this writer behind a lock takes, with nobody else
2824    /// waiting between them. See [`prepare`].
2825    fn flush_pending(&mut self) -> Result<()> {
2826        if self.pending.is_empty() {
2827            return Ok(());
2828        }
2829        let held = std::mem::take(&mut self.pending);
2830        let prepared = self.preparer().prepare_held(held)?;
2831        let merged = self.merge_held(prepared)?;
2832        let paged = merged.pages()?;
2833        self.write_paged(paged)
2834    }
2835
2836    /// Writes one stripe whose pages are built, each column's parts contiguous on disk.
2837    fn write_stripe(&mut self, held: &[Part], encoded: Vec<ColumnStripe>) -> Result<()> {
2838        let width = self.table.fields.len();
2839        let parts = held.len();
2840        if encoded.len() != width {
2841            return Err(Error::internal("a stripe came to the writer with the wrong columns"));
2842        }
2843        let profile = self.profile.clone();
2844        if let Some(profile) = &profile {
2845            let rows = held.iter().map(|part| part.rows as u64).sum();
2846            let raw = held.iter().map(|part| part.footprint as u64).sum();
2847            let pages =
2848                encoded.iter().flat_map(|stripe| &stripe.pages).map(|page| page.len() as u64).sum();
2849            profile.moved(Stage::Pages, raw, pages, rows);
2850        }
2851        // Before a byte of the stripe is written, so that the blocks the stripe's pages were built
2852        // with, and any that were waiting on them, are let go of now rather than a stripe later.
2853        let timing = profile.as_deref().map(|profile| profile.span(Stage::Dictionary));
2854        let before = self.at;
2855        self.place_blocks()?;
2856        drop(timing);
2857        if let Some(profile) = &profile {
2858            profile.moved(Stage::Dictionary, 0, self.at - before, 0);
2859        }
2860        let timing = profile.as_deref().map(|profile| profile.span(Stage::Write));
2861        let before = self.at;
2862        let mut pages = Vec::with_capacity(width);
2863        let mut memberships = vec![None; width];
2864        let mut ranges = Vec::with_capacity(width);
2865        let mut index = Vec::with_capacity(width.saturating_mul(index_section(parts)?));
2866        // Every page of the stripe goes to the file in one call after the loop, since they sit
2867        // back to back from where the stripe starts and a page is often a few kilobytes.
2868        let start = self.at;
2869        let mut out = Vec::with_capacity(width.saturating_mul(parts));
2870        for stripe in &encoded {
2871            let offset = self.at;
2872            let section = index.len();
2873            let mut length = 0_usize;
2874            if stripe.sums.len() != stripe.pages.len() {
2875                return Err(Error::internal("a stripe's pages came without their checksums"));
2876            }
2877            for (bytes, &sum) in stripe.pages.iter().zip(&stripe.sums) {
2878                put_u32(
2879                    &mut index,
2880                    u32::try_from(bytes.len()).map_err(|_| invalid("part length overflow"))?,
2881                );
2882                put_u64(&mut index, sum);
2883                out.push(bytes.as_slice());
2884                length = length
2885                    .checked_add(bytes.len())
2886                    .ok_or_else(|| invalid("column page length overflow"))?;
2887            }
2888            let hash = checksum(&index[section..]);
2889            put_u64(&mut index, hash);
2890            if length > MAX_PAGE {
2891                return Err(invalid("column page exceeds the configured bound"));
2892            }
2893            self.at = self
2894                .at
2895                .checked_add(length as u64)
2896                .ok_or_else(|| invalid("native file length overflow"))?;
2897            pages.push(Span {
2898                offset,
2899                length: u32::try_from(length).map_err(|_| invalid("page length overflow"))?,
2900            });
2901            ranges.push(merged_range(stripe.ranges.iter().cloned()));
2902        }
2903        self.file.write_parts_at(start, &out)?;
2904        drop(out);
2905        for (membership, stripe) in memberships.iter_mut().zip(&encoded) {
2906            if stripe.codes.iter().all(Option::is_none) {
2907                continue;
2908            }
2909            let lists = stripe
2910                .codes
2911                .iter()
2912                .map(|codes| codes.clone().unwrap_or_default())
2913                .collect::<Vec<_>>();
2914            let bytes = encode_membership(&merged_codes(lists));
2915            let offset = self.at;
2916            self.put(&bytes)?;
2917            *membership = Some(Page {
2918                offset,
2919                length: u32::try_from(bytes.len())
2920                    .map_err(|_| invalid("membership page length overflow"))?,
2921                hash: checksum(&bytes),
2922            });
2923        }
2924        let mut sieves = vec![None; width];
2925        for (page, stripe) in sieves.iter_mut().zip(&encoded) {
2926            if stripe.sieves.iter().all(Option::is_none) {
2927                continue;
2928            }
2929            let bytes = encode_sieves(stripe.sieves.iter())?;
2930            let offset = self.at;
2931            self.put(&bytes)?;
2932            *page = Some(Page {
2933                offset,
2934                length: u32::try_from(bytes.len())
2935                    .map_err(|_| invalid("sieve page length overflow"))?,
2936                hash: checksum(&bytes),
2937            });
2938        }
2939        // A stripe of one part has the same rows in it as that part, so its own bounds are already
2940        // the part's and a page here would say what the directory says. Everywhere else the page is
2941        // written unless it comes to more than the column it indexes, which is the rule the sieves
2942        // go by and for the same reason: a reader reads this to decide whether to read the column,
2943        // so a page larger than the column has spent more than the read it is avoiding.
2944        let mut part_ranges = vec![None; width];
2945        if parts > 1 {
2946            for ((page, stripe), span) in part_ranges.iter_mut().zip(&encoded).zip(&pages) {
2947                let bytes = encode_part_ranges(&stripe.ranges)?;
2948                if bytes.len() >= span.length as usize {
2949                    continue;
2950                }
2951                let offset = self.at;
2952                self.put(&bytes)?;
2953                *page = Some(Page {
2954                    offset,
2955                    length: u32::try_from(bytes.len())
2956                        .map_err(|_| invalid("part range page length overflow"))?,
2957                    hash: checksum(&bytes),
2958                });
2959            }
2960        }
2961        let offset = self.at;
2962        self.put(&index)?;
2963        let index = Span {
2964            offset,
2965            length: u32::try_from(index.len())
2966                .map_err(|_| invalid("index page length overflow"))?,
2967        };
2968        let mut rows = 0_usize;
2969        let mut lengths = Vec::with_capacity(parts);
2970        let mut span = None;
2971        for part in held {
2972            rows = rows.checked_add(part.rows).ok_or_else(|| invalid("row count overflow"))?;
2973            lengths.push(u32::try_from(part.rows).map_err(|_| invalid("part row count overflow"))?);
2974            span = Some(span.map_or((part.order, part.order), |(first, _)| (first, part.order)));
2975        }
2976        self.order.push(span.ok_or_else(|| invalid("a stripe was flushed with no parts"))?);
2977        self.table.stripes.push(Stripe {
2978            rows,
2979            parts: lengths,
2980            index,
2981            pages,
2982            memberships: Pages::from_slots(memberships)?,
2983            sieves: Pages::from_slots(sieves)?,
2984            part_ranges: Pages::from_slots(part_ranges)?,
2985            zone: Zone::from_ranges(ranges),
2986        });
2987        drop(timing);
2988        if let Some(profile) = &profile {
2989            profile.moved(Stage::Write, 0, self.at - before, rows as u64);
2990        }
2991        Ok(())
2992    }
2993
2994    /// Finds exact heavy hitters without keeping a hash table for every numeric column while the
2995    /// load is live. The pages are already in the target file, so one column at a time uses a
2996    /// bounded Misra-Gries candidate table and then recounts only those candidates.
2997    ///
2998    /// The first of those passes also counts the column's distinct values exactly, up to the cap in
2999    /// [`distinct`], which is the number a string column gets from its dictionary. It comes back
3000    /// beside the summary because a column whose heavy hitters cannot be proved can still have been
3001    /// counted.
3002    ///
3003    /// The tables are keyed by a value's sixty four bits rather than by [`FrequencyValue`], and a
3004    /// null is counted beside them. Every integer type the format stores fits in those bits, so
3005    /// within one column two values share bits only if they are the same value, and a sixteen byte
3006    /// entry keeps the whole candidate table in the second level cache where the forty eight byte
3007    /// one did not. The null takes part in the candidate table exactly as a key would: it holds a
3008    /// place while its count is above zero, and it is decremented with the rest.
3009    ///
3010    /// `counted` is false for a column whose sketch says its distinct values are far past what the
3011    /// exact set holds. It still gets its frequencies, and a count only if it turns out to have
3012    /// fewer values than the candidate table, which is the count that costs nothing.
3013    fn numeric_frequency(
3014        &self,
3015        column: usize,
3016        counted: bool,
3017        dense: Option<(u64, usize)>,
3018    ) -> Result<(Option<FrequencySummary>, Option<u64>)> {
3019        let signed = match self.table.fields[column].ty {
3020            LogicalType::TinyInt
3021            | LogicalType::SmallInt
3022            | LogicalType::Integer
3023            | LogicalType::BigInt
3024            | LogicalType::Date
3025            | LogicalType::Timestamp => true,
3026            LogicalType::UTinyInt
3027            | LogicalType::USmallInt
3028            | LogicalType::UInteger
3029            | LogicalType::UBigInt => false,
3030            _ => return Ok((None, None)),
3031        };
3032        let value_of = |bits: Option<u64>| match bits {
3033            None => FrequencyValue::Null,
3034            Some(bits) => integer_value(bits, signed),
3035        };
3036        // A column the writer's tally held whole has its exact counts already, gathered as the rows
3037        // went past, so the pages are not read back to count them again. On `hits` that is most of
3038        // the flag and enum columns. The tally only speaks for the whole column when it saw every
3039        // row, which is the same check the statistics make before they are written.
3040        let tallied = self
3041            .gathers
3042            .get(column)
3043            .and_then(Option::as_ref)
3044            .filter(|gather| gather.rows() == self.table.rows as u64)
3045            .and_then(stats::Gather::frequencies)
3046            .and_then(|(values, nulls)| {
3047                let entries = values
3048                    .iter()
3049                    .map(|(value, count)| {
3050                        let value = value_of(Some(frequency_bits(value)?));
3051                        Some(FrequencyEntry { value, count: *count })
3052                    })
3053                    .chain((nulls != 0).then_some(Some(FrequencyEntry {
3054                        value: FrequencyValue::Null,
3055                        count: nulls,
3056                    })))
3057                    .collect::<Option<Vec<_>>>()?;
3058                Some((entries, values.len() as u64))
3059            });
3060        // A column the sketch expects to fit the exact set is counted there, every value with the
3061        // rows holding it, which is its distinct count and its frequencies from one read of its
3062        // pages. Only a column past the set's cap goes through the candidate table.
3063        let exact = match (&tallied, counted) {
3064            (None, true) => self.exact_frequency(column, signed, dense)?,
3065            _ => None,
3066        };
3067        let (mut entries, decrements, distinct_count) = match (tallied, exact) {
3068            (Some((entries, distinct)), _) => (entries, 0, Some(distinct)),
3069            (None, Some((Some(entries), distinct))) => (entries, 0, Some(distinct)),
3070            (None, Some((None, distinct))) => return Ok((None, Some(distinct))),
3071            (None, None) => {
3072                // Rows arrive a run of equal values at a time, because a sorted column is runs and
3073                // a flag column is mostly one value, so a run is counted and inserted once rather
3074                // than per row.
3075                let mut first = Candidates::default();
3076                let mut run = Run::default();
3077                self.visit_numeric(column, signed, |_, bits| {
3078                    if let Some((ended, times)) = run.push(bits) {
3079                        first.add(ended, times);
3080                    }
3081                })?;
3082                if let Some((bits, times)) = run.take() {
3083                    first.add(bits, times);
3084                }
3085                // Until a candidate is turned away the table holds every value the column has, so
3086                // its size is the count.
3087                let (nulls, decrements) = (first.nulls, first.decrements);
3088                let distinct_count = (decrements == 0).then_some(first.held as u64);
3089                let (exact, null_count) = if decrements == 0 {
3090                    let exact = first
3091                        .pairs()
3092                        .map(|(bits, count)| (bits, u64::from(count)))
3093                        .collect::<FrequencyMap<_>>();
3094                    (exact, (nulls != 0).then_some(u64::from(nulls)))
3095                } else {
3096                    let mut lower = first.pairs().map(|(_, count)| count).collect::<Vec<_>>();
3097                    if nulls != 0 {
3098                        lower.push(nulls);
3099                    }
3100                    lower.sort_unstable_by(|left, right| right.cmp(left));
3101                    if lower.len() < FREQUENCY_BUILD_RANK
3102                        || u64::from(lower[FREQUENCY_BUILD_RANK - 1]) <= decrements
3103                    {
3104                        return Ok((None, distinct_count));
3105                    }
3106                    // Counted beside the slot each candidate sits in, since the table is not
3107                    // changed again and a lookup in it is the one probe the first pass made.
3108                    let mut recounts = vec![0_u64; first.slots.len()];
3109                    let mut null_count = (nulls != 0).then_some(0_u64);
3110                    let mut recount = |bits: Option<u64>, times: u32| {
3111                        let held = match bits {
3112                            Some(bits) => first.position(bits).map(|at| &mut recounts[at]),
3113                            None => null_count.as_mut(),
3114                        };
3115                        if let Some(count) = held {
3116                            *count = count.saturating_add(u64::from(times));
3117                        }
3118                    };
3119                    let mut run = Run::default();
3120                    self.visit_numeric(column, signed, |_, bits| {
3121                        if let Some((bits, times)) = run.push(bits) {
3122                            recount(bits, times);
3123                        }
3124                    })?;
3125                    if let Some((bits, times)) = run.take() {
3126                        recount(bits, times);
3127                    }
3128                    let exact = first
3129                        .slots
3130                        .iter()
3131                        .zip(&recounts)
3132                        .filter(|(slot, _)| slot.count != 0)
3133                        .map(|(slot, &count)| (slot.bits, count))
3134                        .collect::<FrequencyMap<_>>();
3135                    (exact, null_count)
3136                };
3137                let entries = exact
3138                    .into_iter()
3139                    .map(|(bits, count)| FrequencyEntry { value: value_of(Some(bits)), count })
3140                    .chain(
3141                        null_count
3142                            .map(|count| FrequencyEntry { value: FrequencyValue::Null, count }),
3143                    )
3144                    .collect::<Vec<_>>();
3145                (entries, decrements, distinct_count)
3146            }
3147        };
3148        let mut omitted_max = keep_most_frequent(&mut entries).max(decrements);
3149        // A complete value-to-count table is also the result of grouping this column.
3150        // Keep up to two leading frequencies for selectivity and equality predicates,
3151        // but leave multi-value grouped counts to the encoded rows at query time.
3152        if omitted_max == 0 && entries.len() > 1 {
3153            let retained = entries.len().saturating_sub(1).min(2);
3154            omitted_max = entries[retained].count;
3155            entries.truncate(retained);
3156        }
3157        // The rows of every listed value when they fit, and otherwise the rows of the longest leading
3158        // run that fits, but only when the tenth listed value is held by more rows than the first
3159        // one left out, because a bound no smaller than the leading counts vouches for nothing.
3160        let mut covered = 0;
3161        let mut kept_rows = 0_u64;
3162        for entry in &entries {
3163            match kept_rows.checked_add(entry.count) {
3164                Some(total) if total <= FREQUENCY_ORDINALS as u64 => kept_rows = total,
3165                _ => break,
3166            }
3167            covered += 1;
3168        }
3169        let ordinal_bound = entries.get(covered).map_or(0, |entry| entry.count);
3170        let worth_keeping = covered == entries.len()
3171            || (covered >= FREQUENCY_BUILD_RANK
3172                && entries[FREQUENCY_BUILD_RANK - 1].count > ordinal_bound.max(omitted_max));
3173        let mut ordinals = Vec::new();
3174        let mut ordinal_entries = Vec::new();
3175        if worth_keeping {
3176            let mut kept = FrequencyMap::default();
3177            let mut null_kept = None;
3178            for (at, entry) in entries.iter().enumerate().take(covered) {
3179                let at = u16::try_from(at)
3180                    .map_err(|_| invalid("too many retained frequency entries"))?;
3181                match entry.value {
3182                    FrequencyValue::Integer(value) => {
3183                        kept.insert(value as u64, at);
3184                    }
3185                    FrequencyValue::Null => null_kept = Some(at),
3186                    FrequencyValue::Code(_) => {}
3187                }
3188            }
3189            ordinals.reserve(usize::try_from(kept_rows).unwrap_or(FREQUENCY_ORDINALS));
3190            ordinal_entries.reserve(usize::try_from(kept_rows).unwrap_or(FREQUENCY_ORDINALS));
3191            self.visit_numeric(column, signed, |ordinal, bits| {
3192                let held = match bits {
3193                    Some(bits) => kept.get(&bits).copied(),
3194                    None => null_kept,
3195                };
3196                if let Some(entry) = held {
3197                    ordinals.push(ordinal);
3198                    ordinal_entries.push(entry);
3199                }
3200            })?;
3201        }
3202        Ok((
3203            Some(FrequencySummary {
3204                entries,
3205                omitted_max,
3206                ordinals,
3207                ordinal_entries,
3208                ordinal_bound: if worth_keeping { ordinal_bound } else { 0 },
3209            }),
3210            distinct_count,
3211        ))
3212    }
3213
3214    /// The bits of a column's lowest value and how many values its range holds, when counting it in
3215    /// a [`distinct::DenseCounts`] would take no more memory than the set it would otherwise be
3216    /// charged, or a mebibyte, whichever is more.
3217    ///
3218    /// Only for a column the statistics saw every row of, since otherwise its ends may not be its
3219    /// ends, and a table of fewer than `u32::MAX` rows, so that a count fits in its slot.
3220    fn dense_range(&self, gather: &stats::Gather, set: usize) -> Option<(u64, usize)> {
3221        let rows = self.table.rows;
3222        if gather.rows() != rows as u64 || u32::try_from(rows).is_err() {
3223            return None;
3224        }
3225        let (low, high) = gather.span()?;
3226        let len = usize::try_from(high.checked_sub(low)?.checked_add(1)?).ok()?;
3227        #[allow(clippy::cast_sign_loss, clippy::cast_possible_truncation)]
3228        let bits = low as u64;
3229        (len.checked_mul(size_of::<u32>())? <= set.max(1 << 20)).then_some((bits, len))
3230    }
3231
3232    /// Counts every value of an integer column and the rows holding it, and hands back the
3233    /// frequency entries worth keeping beside the distinct count, or nothing for a column with more
3234    /// values than [`distinct::ExactCounts`] keeps.
3235    ///
3236    /// The entries are `None` for a column with no value common enough to be worth a synopsis. The
3237    /// rule is the one the candidate table applied. A column with more values than that table holds
3238    /// keeps its frequencies only if its tenth commonest value is held by more rows than a
3239    /// Misra-Gries table of [`FREQUENCY_CANDIDATES`] could have decremented it by, which is its rows
3240    /// over one more than the candidates. The counts kept are exact either way, so the largest one
3241    /// left out is exact too and not the table's bound on it.
3242    ///
3243    /// Only the commonest entries and the ones tied with the first left out are built, since on a
3244    /// column of a million values the rest are thrown away the moment they are ranked.
3245    fn exact_frequency(
3246        &self,
3247        column: usize,
3248        signed: bool,
3249        dense: Option<(u64, usize)>,
3250    ) -> Result<Option<(Option<Vec<FrequencyEntry>>, u64)>> {
3251        // A column whose ends are close together is counted in a flat array. A value outside the
3252        // ends it was given, which would be a bug in the statistics, sends it to the set instead.
3253        if let Some((low, len)) = dense {
3254            let mut counts = distinct::DenseCounts::new(low, len);
3255            let nulls =
3256                self.count_numeric(column, signed, |bits, times| counts.insert(bits, times))?;
3257            if let Some(distinct) = counts.count() {
3258                let Some(distinct) = distinct else { return Ok(None) };
3259                return Ok(Some(self.frequent_entries(signed, distinct, nulls, |visit| {
3260                    counts.visit(visit);
3261                })));
3262            }
3263        }
3264        let mut set = distinct::ExactCounts::new();
3265        let nulls = self.count_numeric(column, signed, |bits, times| set.insert(bits, times))?;
3266        let Some(distinct) = set.count() else {
3267            return Ok(None);
3268        };
3269        Ok(Some(self.frequent_entries(signed, distinct, nulls, |visit| {
3270            set.visit(visit);
3271        })))
3272    }
3273
3274    /// Hands every run of equal non-null values in an integer column to `add` as its bits and its
3275    /// length, and answers how many rows were null.
3276    fn count_numeric(
3277        &self,
3278        column: usize,
3279        signed: bool,
3280        mut add: impl FnMut(u64, u32),
3281    ) -> Result<u64> {
3282        let mut nulls = 0_u64;
3283        let mut run = Run::default();
3284        let mut take = |bits: Option<u64>, times: u32| match bits {
3285            Some(bits) => add(bits, times),
3286            None => nulls += u64::from(times),
3287        };
3288        self.visit_numeric(column, signed, |_, bits| {
3289            if let Some((bits, times)) = run.push(bits) {
3290                take(bits, times);
3291            }
3292        })?;
3293        if let Some((bits, times)) = run.take() {
3294            take(bits, times);
3295        }
3296        Ok(nulls)
3297    }
3298
3299    /// The frequency entries worth keeping out of a column's exact counts, which `visit` hands over
3300    /// as bits and rows once for each call it gets. See [`Self::exact_frequency`] for the rule.
3301    fn frequent_entries(
3302        &self,
3303        signed: bool,
3304        distinct: u64,
3305        nulls: u64,
3306        mut visit: impl FnMut(&mut dyn FnMut(u64, u64)),
3307    ) -> (Option<Vec<FrequencyEntry>>, u64) {
3308        // The commonest counts, one more than the entries kept so that the first left out is here.
3309        let mut top = std::collections::BinaryHeap::with_capacity(FREQUENCY_ENTRIES + 2);
3310        let mut rank = |count: u64| {
3311            if top.len() <= FREQUENCY_ENTRIES {
3312                top.push(Reverse(count));
3313            } else if top.peek().is_some_and(|&Reverse(least)| count > least) {
3314                top.pop();
3315                top.push(Reverse(count));
3316            }
3317        };
3318        visit(&mut |_, count| rank(count));
3319        if nulls != 0 {
3320            rank(nulls);
3321        }
3322        let top = top.into_sorted_vec();
3323        let values = distinct + u64::from(nulls != 0);
3324        if values > FREQUENCY_CANDIDATES as u64 {
3325            let bound = self.table.rows as u64 / (FREQUENCY_CANDIDATES as u64 + 1);
3326            if top.get(FREQUENCY_BUILD_RANK - 1).is_none_or(|&Reverse(count)| count <= bound) {
3327                return (None, distinct);
3328            }
3329        }
3330        let least = top.get(FREQUENCY_ENTRIES).map_or(0, |&Reverse(count)| count);
3331        let mut entries = Vec::with_capacity(FREQUENCY_ENTRIES + 1);
3332        visit(&mut |bits, count| {
3333            if count >= least {
3334                entries.push(FrequencyEntry { value: integer_value(bits, signed), count });
3335            }
3336        });
3337        if nulls != 0 && nulls >= least {
3338            entries.push(FrequencyEntry { value: FrequencyValue::Null, count: nulls });
3339        }
3340        (Some(entries), distinct)
3341    }
3342
3343    /// Hands every row of an integer column to `visit` as its ordinal and its sixty four bits, or
3344    /// `None` for a null.
3345    ///
3346    /// `signed` says which of the two readings the column has. A packed unsigned column would come
3347    /// back from `signed_block` as a base plus a code in `i64`, which wraps for a value past the top
3348    /// of `BIGINT`, so only a signed column takes the block path.
3349    fn visit_numeric(
3350        &self,
3351        column: usize,
3352        signed: bool,
3353        mut visit: impl FnMut(u64, Option<u64>),
3354    ) -> Result<()> {
3355        let ty = &self.table.fields[column].ty;
3356        let mut start = 0_u64;
3357        let mut block = Vec::new();
3358        for stripe in &self.table.stripes {
3359            let spans = read_index(&self.file, stripe, column)?;
3360            let page = stripe.pages[column];
3361            let mut bytes = vec![0; page.length as usize];
3362            read_at(&self.file, page.offset, &mut bytes)?;
3363            for (span, &rows) in spans.iter().zip(&stripe.parts) {
3364                let part = part_bytes(&bytes, *span)?;
3365                if checksum(part) != span.hash {
3366                    return Err(invalid("column page checksum differs while building frequencies"));
3367                }
3368                let rows = rows as usize;
3369                let vector = decode(ty, rows, part, None)?;
3370                // Every signed layout a numeric column decodes to, which is every column of `hits`,
3371                // comes out as one run of `i64` and is walked as a slice. The row path below is for
3372                // the unsigned types and anything else that cannot be handed over that way.
3373                if signed && vector.signed_block(&mut block) && block.len() == rows {
3374                    if vector.none_null() {
3375                        for (row, &value) in block.iter().enumerate() {
3376                            visit(start.saturating_add(row as u64), Some(value as u64));
3377                        }
3378                    } else {
3379                        for (row, &value) in block.iter().enumerate() {
3380                            let bits = (!vector.is_null_at(row)).then_some(value as u64);
3381                            visit(start.saturating_add(row as u64), bits);
3382                        }
3383                    }
3384                    start = start.saturating_add(rows as u64);
3385                    continue;
3386                }
3387                // row at a time: frequency construction visits decoded values to update bounded candidates.
3388                for row in 0..rows {
3389                    let bits = if vector.is_null_at(row) {
3390                        None
3391                    } else {
3392                        // An unsigned column has no signed reading, and the documented fallback is
3393                        // the value itself. Every width the format stores fits in sixty four bits,
3394                        // so nothing is lost on the way through.
3395                        let widened = match vector.signed_at(row) {
3396                            Some(value) => Some(value as u64),
3397                            None => match vector.value_at(row) {
3398                                Value::UTinyInt(value) => Some(u64::from(value)),
3399                                Value::USmallInt(value) => Some(u64::from(value)),
3400                                Value::UInteger(value) => Some(u64::from(value)),
3401                                Value::UBigInt(value) => Some(value),
3402                                _ => None,
3403                            },
3404                        };
3405                        Some(widened.ok_or_else(|| {
3406                            invalid("numeric frequency page did not contain an integer value")
3407                        })?)
3408                    };
3409                    visit(start.saturating_add(row as u64), bits);
3410                }
3411                start = start.saturating_add(rows as u64);
3412            }
3413        }
3414        Ok(())
3415    }
3416
3417    /// The columns that get numeric frequencies, which are the integer, date and timestamp ones.
3418    fn numeric_columns(&self) -> Vec<usize> {
3419        self.table
3420            .fields
3421            .iter()
3422            .enumerate()
3423            .filter_map(|(column, field)| {
3424                matches!(
3425                    field.ty,
3426                    LogicalType::TinyInt
3427                        | LogicalType::SmallInt
3428                        | LogicalType::Integer
3429                        | LogicalType::BigInt
3430                        | LogicalType::UTinyInt
3431                        | LogicalType::USmallInt
3432                        | LogicalType::UInteger
3433                        | LogicalType::UBigInt
3434                        | LogicalType::Date
3435                        | LogicalType::Timestamp
3436                )
3437                .then_some(column)
3438            })
3439            .collect()
3440    }
3441
3442    /// Reads one stable dictionary code column only at sorted table-wide row ordinals.
3443    #[allow(dead_code)]
3444    fn stable_codes_at(&self, column: usize, ordinals: &[u64]) -> Result<Option<Vec<Option<u32>>>> {
3445        if self.dictionaries.get(column).and_then(Option::as_ref).is_none() {
3446            return Ok(None);
3447        }
3448        if ordinals.windows(2).any(|pair| pair[0] >= pair[1]) {
3449            return Err(invalid("frequency ordinals are not sorted and unique"));
3450        }
3451        let mut out = Vec::with_capacity(ordinals.len());
3452        let mut wanted = 0;
3453        let mut stripe_start = 0_u64;
3454        for stripe in &self.table.stripes {
3455            let stripe_end = stripe_start.saturating_add(stripe.rows as u64);
3456            if wanted == ordinals.len() || ordinals[wanted] >= stripe_end {
3457                stripe_start = stripe_end;
3458                continue;
3459            }
3460            let spans = read_index(&self.file, stripe, column)?;
3461            let page = stripe.pages[column];
3462            let mut bytes = vec![0; page.length as usize];
3463            read_at(&self.file, page.offset, &mut bytes)?;
3464            let mut part_start = stripe_start;
3465            for (span, &rows) in spans.iter().zip(&stripe.parts) {
3466                let part_end = part_start.saturating_add(u64::from(rows));
3467                if wanted < ordinals.len() && ordinals[wanted] < part_end {
3468                    let part = part_bytes(&bytes, *span)?;
3469                    if checksum(part) != span.hash {
3470                        return Err(invalid(
3471                            "column page checksum differs while building pair frequencies",
3472                        ));
3473                    }
3474                    let upto = ordinals.partition_point(|&ordinal| ordinal < part_end);
3475                    let positions = ordinals[wanted..upto]
3476                        .iter()
3477                        .map(|&ordinal| {
3478                            usize::try_from(ordinal.saturating_sub(part_start))
3479                                .map_err(|_| invalid("frequency row offset does not fit in memory"))
3480                        })
3481                        .collect::<Result<Vec<_>>>()?;
3482                    if !decode_selected_stable_codes(rows as usize, part, &positions, &mut out)? {
3483                        return Ok(None);
3484                    }
3485                    wanted = upto;
3486                }
3487                part_start = part_end;
3488            }
3489            stripe_start = stripe_end;
3490        }
3491        if wanted != ordinals.len() {
3492            return Err(invalid("frequency ordinal is outside the table"));
3493        }
3494        Ok(Some(out))
3495    }
3496
3497    /// Derives bounded two-key leaders from numeric anchor ordinals and stable string codes.
3498    #[allow(dead_code)]
3499    fn pair_frequencies(
3500        &self,
3501        frequencies: &[Option<Frequencies>],
3502    ) -> Result<Vec<PairFrequencySummary>> {
3503        let anchors = frequencies
3504            .iter()
3505            .enumerate()
3506            .filter_map(|(column, summary)| {
3507                // A writer holds every synopsis it counted, so there is nothing stored to skip.
3508                match summary {
3509                    Some(Frequencies::Held(summary)) => Some(summary),
3510                    _ => None,
3511                }
3512                .filter(|summary| {
3513                    !summary.ordinals.is_empty()
3514                        && summary.ordinal_entries.len() == summary.ordinals.len()
3515                        && summary.ordinal_bound == 0
3516                })
3517                .cloned()
3518                .map(|summary| (column, summary))
3519            })
3520            .collect::<Vec<_>>();
3521        let strings = self
3522            .dictionaries
3523            .iter()
3524            .enumerate()
3525            .filter_map(|(column, dictionary)| dictionary.as_ref().map(|_| column))
3526            .collect::<Vec<_>>();
3527        let mut summaries = Vec::new();
3528        for (first, anchors) in anchors {
3529            for &second in &strings {
3530                if summaries.len() == MAX_PAIR_FREQUENCIES {
3531                    return Ok(summaries);
3532                }
3533                let Some(codes) = self.stable_codes_at(second, &anchors.ordinals)? else {
3534                    continue;
3535                };
3536                if codes.len() != anchors.ordinal_entries.len() {
3537                    return Err(invalid("pair frequency columns have different lengths"));
3538                }
3539                let mut counts = HashMap::<(u16, Option<u32>), u64>::new();
3540                for (&anchor, code) in anchors.ordinal_entries.iter().zip(codes) {
3541                    *counts.entry((anchor, code)).or_default() += 1;
3542                }
3543                let mut entries = counts
3544                    .into_iter()
3545                    .map(|((first_entry, second), count)| PairFrequencyEntry {
3546                        first_entry,
3547                        second,
3548                        count,
3549                    })
3550                    .collect::<Vec<_>>();
3551                entries.sort_unstable_by(|left, right| {
3552                    right
3553                        .count
3554                        .cmp(&left.count)
3555                        .then_with(|| left.first_entry.cmp(&right.first_entry))
3556                        .then_with(|| left.second.cmp(&right.second))
3557                });
3558                let pair_omitted = entries.get(FREQUENCY_ENTRIES).map_or(0, |entry| entry.count);
3559                entries.truncate(FREQUENCY_ENTRIES);
3560                summaries.push(PairFrequencySummary {
3561                    first: u16::try_from(first)
3562                        .map_err(|_| invalid("pair frequency column index overflows"))?,
3563                    second: u16::try_from(second)
3564                        .map_err(|_| invalid("pair frequency column index overflows"))?,
3565                    entries,
3566                    omitted_max: anchors.omitted_max.max(pair_omitted),
3567                });
3568            }
3569        }
3570        Ok(summaries)
3571    }
3572
3573    /// Writes the directory of the table this writer is on and says where it went.
3574    ///
3575    /// Everything [`Writer::finish`] used to do except the two writes that publish. Pulling it out
3576    /// is what lets a second table follow a first: the bytes of a closed table are complete and
3577    /// addressable while nothing yet points at them, and the pointer is the last write of the
3578    /// commit.
3579    ///
3580    /// # Errors
3581    ///
3582    /// If directory encoding or writing fails.
3583    fn close(&mut self) -> Result<Entry> {
3584        self.reclaim()?;
3585        self.flush_pending()?;
3586        // The rest of a table is its statistics, its dictionaries and its directory. The dictionary
3587        // work is charged as its own stage, because ranking a global dictionary can be most of what
3588        // this costs, and the rest as publish.
3589        let profile = self.profile.clone();
3590        let timing = profile.as_deref().map(|profile| profile.span(Stage::Publish));
3591        let before = self.at;
3592        let mut stripes = std::mem::take(&mut self.order)
3593            .into_iter()
3594            .zip(std::mem::take(&mut self.table.stripes))
3595            .collect::<Vec<_>>();
3596        stripes.sort_by_key(|(order, _)| order.0);
3597        let mut previous: Option<(u64, u64)> = None;
3598        for ((first, last), _) in &stripes {
3599            if previous.is_some_and(|previous| previous >= *first) {
3600                return Err(invalid("chunks did not arrive in source order"));
3601            }
3602            previous = Some(*last);
3603        }
3604        self.table.stripes = stripes.into_iter().map(|(_, stripe)| stripe).collect();
3605        drop(timing);
3606        let timing = profile.as_deref().map(|profile| profile.span(Stage::Dictionary));
3607        let placing = self.at;
3608        finish_dictionaries(&mut self.dictionaries)?;
3609        self.place_blocks()?;
3610        for dictionary in self.dictionaries.iter_mut().flatten() {
3611            dictionary.release_lookup();
3612            dictionary.recharge(profile.as_deref());
3613        }
3614        let (numeric, closed) = self.close_columns()?;
3615        let (frequencies, distincts): (Vec<Option<FrequencySummary>>, Vec<_>) =
3616            numeric.into_iter().unzip();
3617        let frequencies =
3618            frequencies.into_iter().map(|held| held.map(Frequencies::Held)).collect::<Vec<_>>();
3619        // Pair leaders are query results, not reusable column statistics.
3620        let pairs = Vec::new();
3621        self.table.frequencies = frequencies;
3622        self.table.distincts = distincts;
3623        self.table.pair_frequencies = pairs;
3624        if let Some(profile) = &profile {
3625            profile.release(self.dictionaries.iter().flatten().map(|held| held.charged).sum());
3626        }
3627        self.table.demoted = self
3628            .dictionaries
3629            .iter()
3630            .map(|dictionary| dictionary.as_ref().is_some_and(|held| held.demoted))
3631            .collect();
3632        if !self.table.demoted.contains(&true) {
3633            self.table.demoted = Vec::new();
3634        }
3635        self.dictionaries = Vec::new();
3636        self.table.dictionary_payloads = vec![0; self.table.fields.len()];
3637        self.table.frequency_texts = vec![Vec::new(); self.table.fields.len()];
3638        self.table.host_groups = None;
3639        for (index, closed) in closed.into_iter().enumerate() {
3640            let Some(closed) = closed else { continue };
3641            let ClosedDictionary { distinct, frequencies, texts, hosts, encoded, payload } = closed;
3642            self.table.distincts[index] = distinct;
3643            self.table.frequencies[index] = frequencies.map(Frequencies::Held);
3644            self.table.frequency_texts[index] = texts;
3645            if hosts.is_some() {
3646                self.table.host_groups = hosts;
3647            }
3648            let offset = self.at;
3649            self.put(&encoded.index)?;
3650            self.put(&encoded.ranks)?;
3651            self.put(&encoded.grams)?;
3652            self.table.dictionary_payloads[index] = payload;
3653            let length = encoded
3654                .index
3655                .len()
3656                .checked_add(encoded.ranks.len())
3657                .and_then(|len| len.checked_add(encoded.grams.len()))
3658                .ok_or_else(|| invalid("dictionary page length overflow"))?;
3659            self.table.dictionaries[index] = Some(Page {
3660                offset,
3661                length: u32::try_from(length)
3662                    .map_err(|_| invalid("dictionary page length overflow"))?,
3663                hash: checksum(&encoded.index),
3664            });
3665        }
3666        drop(timing);
3667        let timing = profile.as_deref().map(|profile| profile.span(Stage::Publish));
3668        let placed = self.at - placing;
3669        self.write_stats()?;
3670        let directory = encode_directory(&self.table)?;
3671        if directory.len() > MAX_DIRECTORY {
3672            return Err(invalid("directory exceeds the configured bound"));
3673        }
3674        let offset = self.at;
3675        self.put(&directory)?;
3676        drop(timing);
3677        if let Some(profile) = &profile {
3678            profile.moved(Stage::Dictionary, 0, placed, 0);
3679            profile.moved(Stage::Publish, 0, self.at - before - placed, 0);
3680        }
3681        Ok(Entry {
3682            name: self.table.name.clone(),
3683            fields: self.table.fields.clone(),
3684            rows: self.table.rows,
3685            nonzero: vec![None; self.table.fields.len()],
3686            aggregates: table_aggregate_sums(&self.table),
3687            distincts: self.table.distincts.clone(),
3688            extremes: table_integer_extremes(&self.table),
3689            frequencies: table_complete_numeric_frequencies(&self.table),
3690            directory: Page {
3691                offset,
3692                length: u32::try_from(directory.len())
3693                    .map_err(|_| invalid("directory length overflow"))?,
3694                hash: checksum(&directory),
3695            },
3696        })
3697    }
3698
3699    /// Every numeric column's frequencies and every global dictionary's page and statistics, by
3700    /// column, as many columns at a time as [`CLOSE_BYTES`] allows.
3701    ///
3702    /// The two kinds read what is already written and write nothing, so they share one set of
3703    /// threads. Each was most of a second on `hits` with the other waiting for it, and neither keeps
3704    /// every core busy on its own. The most expensive column that fits is the one taken next, so
3705    /// the long ones start first and the short ones fill in behind them. A column that does not fit
3706    /// waits for one that is closing to finish, unless nothing is closing, in which case it goes
3707    /// alone.
3708    ///
3709    /// A numeric column is charged the exact distinct set its sketch says it will need, and one the
3710    /// sketch puts far past what that set can hold does not build it, because the set would fill,
3711    /// give up and have held 512 MiB for nothing. A column with no sketch is charged the whole set.
3712    /// Each job charges itself as its own span, publish for the numeric ones and dictionary for the
3713    /// rest, because it runs on a thread of its own and a span on this one would see the wall time
3714    /// and none of the CPU.
3715    #[allow(clippy::type_complexity)]
3716    fn close_columns(
3717        &self,
3718    ) -> Result<(Vec<(Option<FrequencySummary>, Option<u64>)>, Vec<Option<ClosedDictionary>>)> {
3719        let numeric = self.numeric_columns().into_iter().map(|column| {
3720            let gather = self.gathers.get(column).and_then(Option::as_ref);
3721            let estimate = gather.and_then(stats::Gather::distinct);
3722            let counted = !estimate.is_some_and(distinct::beyond);
3723            let set =
3724                if counted { distinct::bytes_for(estimate.unwrap_or(f64::INFINITY)) } else { 0 };
3725            let dense = gather.filter(|_| counted).and_then(|gather| self.dense_range(gather, set));
3726            let set = dense.map_or(set, |(_, len)| len * size_of::<u32>());
3727            let cost = self.table.rows.saturating_mul(weight(&self.table.fields[column].ty));
3728            (Closing::Numeric { column, counted, dense }, NUMERIC_CLOSE_BYTES + set, cost)
3729        });
3730        let dictionaries =
3731            self.dictionaries.iter().enumerate().filter_map(|(index, dictionary)| {
3732                let dictionary = dictionary.as_ref()?;
3733                let bytes = dictionary.closing_bytes();
3734                Some((Closing::Dictionary { index, dictionary }, bytes, bytes))
3735            });
3736        let mut jobs = numeric.chain(dictionaries).collect::<Vec<_>>();
3737        jobs.sort_by_key(|&(_, _, cost)| cost);
3738        let columns = self.table.fields.len();
3739        let mut frequencies = vec![(None, None); columns];
3740        let mut closed = (0..columns).map(|_| None).collect::<Vec<_>>();
3741        let profile = self.profile.as_deref();
3742        let run = |job: Closing<'_>, bytes: usize| -> Result<Closed> {
3743            let _holding = profile.map(|profile| profile.holding(bytes as u64));
3744            let closed = match job {
3745                Closing::Numeric { column, counted, dense } => {
3746                    let _timing = profile.map(|profile| profile.span(Stage::Publish));
3747                    Closed::Numeric(column, self.numeric_frequency(column, counted, dense)?)
3748                }
3749                Closing::Dictionary { index, dictionary } => {
3750                    let _timing = profile.map(|profile| profile.span(Stage::Dictionary));
3751                    Closed::Dictionary(index, self.close_dictionary(index, dictionary)?)
3752                }
3753            };
3754            // A dictionary's decoded values or a column's distinct set were just dropped, and the
3755            // next job is about to take as much again.
3756            rudb_common::heap::release();
3757            Ok(closed)
3758        };
3759        let workers = close_workers().min(jobs.len());
3760        let pieces = if workers <= 1 {
3761            jobs.into_iter().map(|(job, bytes, _)| run(job, bytes)).collect::<Result<Vec<_>>>()?
3762        } else {
3763            // The columns not taken yet, cheapest first, and the bytes the ones closing now hold.
3764            let state = Mutex::new((jobs, 0_usize));
3765            let finished = Condvar::new();
3766            std::thread::scope(|scope| {
3767                (0..workers)
3768                    .map(|_| {
3769                        scope.spawn(|| {
3770                            let mut mine = Vec::new();
3771                            loop {
3772                                let mut held = state.lock().map_err(|_| {
3773                                    Error::internal("a native close worker panicked")
3774                                })?;
3775                                let (job, bytes) = loop {
3776                                    let (jobs, busy) = &mut *held;
3777                                    if jobs.is_empty() {
3778                                        return Ok(mine);
3779                                    }
3780                                    let fits = jobs.iter().rposition(|&(_, bytes, _)| {
3781                                        *busy == 0 || busy.saturating_add(bytes) <= CLOSE_BYTES
3782                                    });
3783                                    if let Some(at) = fits {
3784                                        let (job, bytes, _) = jobs.remove(at);
3785                                        *busy += bytes;
3786                                        break (job, bytes);
3787                                    }
3788                                    held = finished.wait(held).map_err(|_| {
3789                                        Error::internal("a native close worker panicked")
3790                                    })?;
3791                                };
3792                                drop(held);
3793                                // Given back on the way out whether the close worked, failed or
3794                                // panicked, so that a worker waiting for room is never left waiting.
3795                                let _room = Room { state: &state, finished: &finished, bytes };
3796                                mine.push(run(job, bytes)?);
3797                            }
3798                        })
3799                    })
3800                    .collect::<Vec<_>>()
3801                    .into_iter()
3802                    .map(|handle| {
3803                        handle
3804                            .join()
3805                            .map_err(|_| Error::internal("a native close worker panicked"))?
3806                    })
3807                    .collect::<Result<Vec<_>>>()
3808            })?
3809            .into_iter()
3810            .flatten()
3811            .collect()
3812        };
3813        for piece in pieces {
3814            match piece {
3815                Closed::Numeric(column, summary) => frequencies[column] = summary,
3816                Closed::Dictionary(index, one) => closed[index] = Some(one),
3817            }
3818        }
3819        Ok((frequencies, closed))
3820    }
3821
3822    /// One global dictionary's page and statistics, built from what is already in the file.
3823    ///
3824    /// Nothing is written here, so that [`Self::close`] can run this beside the numeric frequencies
3825    /// and put the pages down afterwards in column order, which is where they always went. The
3826    /// column's values are decoded in here and dropped before it returns, and
3827    /// [`Self::close_columns`] decides how many columns are in here at once.
3828    fn close_dictionary(
3829        &self,
3830        _index: usize,
3831        dictionary: &GlobalDictionary,
3832    ) -> Result<ClosedDictionary> {
3833        let (order, flat, bases) = dictionary.ranked_with_values(Some(&*self.file))?;
3834        // A code nothing counted is a code no non-null row of this column holds, which is the
3835        // empty string a null was written as and nothing else, because a code is only ever made by
3836        // a row asking for one. A demoted dictionary counted the stripes before its demotion and
3837        // none after, so it has no count or frequency of the column to give.
3838        let (distinct, frequencies, texts) = if dictionary.demoted {
3839            (None, None, Vec::new())
3840        } else {
3841            let distinct = dictionary.counts.iter().filter(|count| **count != 0).count() as u64;
3842            let (frequencies, texts) = code_frequency(dictionary, &flat, &bases)?;
3843            (Some(distinct), Some(frequencies), texts)
3844        };
3845        // Deriving a fixed SQL host expression at load time materializes its answer.
3846        let hosts = None;
3847        drop(flat);
3848        drop(bases);
3849        let encoded = encode_global_dictionary(dictionary, &order, &dictionary.placed, true)?;
3850        let payload = dictionary
3851            .placed
3852            .iter()
3853            .try_fold(0_u64, |sum, place| sum.checked_add(place.length))
3854            .ok_or_else(|| invalid("global dictionary payload overflow"))?;
3855        Ok(ClosedDictionary { distinct, frequencies, texts, hosts, encoded, payload })
3856    }
3857
3858    /// Writes the statistics sections for the table being closed, as far as the budget reaches.
3859    ///
3860    /// Called from [`Self::close`] after the last stripe and after the dictionaries, which is the
3861    /// first moment the table's column bytes are final and the last moment before the directory is
3862    /// encoded. Both halves matter: the budget is a share of the column bytes, and a section that
3863    /// went in after the directory would be a section the directory does not name.
3864    ///
3865    /// Nothing here can fail the write. A column whose gather came back blind gets no sections, a
3866    /// column the budget could not reach gets none, and section 3.1 says both of those plan the way
3867    /// they planned before statistics existed. The two errors that are returned are an encode
3868    /// failure and a section count past the bound, and neither is a thing a column can cause.
3869    fn write_stats(&mut self) -> Result<()> {
3870        let gathers = std::mem::take(&mut self.gathers);
3871        let rows = self.table.rows as u64;
3872        let mut payloads = Vec::new();
3873        for (column, gather) in gathers.into_iter().enumerate() {
3874            let Some(gather) = gather else { continue };
3875            // A gather that saw a different number of rows than the table committed is a gather
3876            // that missed some, and a distinct count over some of a column is the one error an
3877            // estimator cannot see coming. This has no way of happening today, since a table is
3878            // written once and every chunk goes through `flush_pending`, and that is exactly why it
3879            // is worth a line: it stays true only while that stays true.
3880            if gather.rows() != rows {
3881                continue;
3882            }
3883            let Some(stats) = gather.finish() else { continue };
3884            let mut summary = Vec::new();
3885            stats.summary.encode(&mut summary)?;
3886            let mut sketches = Vec::new();
3887            stats.sketches.encode(&mut sketches)?;
3888            payloads.push((column, summary, sketches));
3889        }
3890        if payloads.is_empty() {
3891            return Ok(());
3892        }
3893        let summaries = payloads.iter().map(|(_, summary, _)| summary.len()).collect::<Vec<_>>();
3894        let sketches = payloads.iter().map(|(_, _, sketches)| sketches.len()).collect::<Vec<_>>();
3895        let allowance = stats::allowance(stats::column_bytes(&self.table), stats::BUDGET_SHARE);
3896        // Nothing is spent yet. A table this writer is closing is one it wrote from nothing, so the
3897        // only statistics sections it can have are the ones about to go in.
3898        let keep = stats::kept(&summaries, &sketches, allowance, 0);
3899        for ((column, summary, sketches), &(built, sketched)) in payloads.iter().zip(&keep) {
3900            if !built {
3901                continue;
3902            }
3903            let id = u64::try_from(*column).map_err(|_| invalid("column index overflow"))?;
3904            let sections = [
3905                // A summary is a header the whole way down: there is nothing behind it a reader
3906                // could decide not to read.
3907                (*section::SUMMARY, summary, summary.len() as u32),
3908                (*section::SKETCHES, sketches, rudb_stats::sketches::HEADER_BYTES),
3909            ];
3910            let wanted = 1 + usize::from(sketched);
3911            for (kind, bytes, header_bytes) in sections.into_iter().take(wanted) {
3912                let written = write_section(
3913                    &*self.file,
3914                    &mut self.at,
3915                    &section::Attachment { kind, id, flags: 0, header_bytes, bytes },
3916                    self.generation,
3917                )?;
3918                self.table.sections.push(written);
3919            }
3920        }
3921        if self.table.sections.len() > MAX_SECTIONS {
3922            return Err(invalid("the table would name more sections than the bound allows"));
3923        }
3924        Ok(())
3925    }
3926
3927    /// Commits every table this writer has written and syncs the file before publishing its header
3928    /// slot.
3929    ///
3930    /// The table handed back is the one the writer was on, which is the last of them. Callers that
3931    /// wrote several already know the others, since they named them.
3932    ///
3933    /// # Errors
3934    ///
3935    /// If directory encoding, writing, or syncing fails.
3936    pub fn finish(mut self) -> Result<Table> {
3937        let entry = self.close()?;
3938        let profile = self.profile.take();
3939        let _timing = profile.as_deref().map(|profile| profile.span(Stage::Publish));
3940        let mut tables = std::mem::take(&mut self.closed);
3941        tables.push(entry);
3942        let catalog =
3943            encode_catalog(&tables, &self.views, self.card.as_ref(), self.anchor.as_ref())?;
3944        if catalog.len() > MAX_DIRECTORY {
3945            return Err(invalid("catalog exceeds the configured bound"));
3946        }
3947        let offset = self.at;
3948        self.put(&catalog)?;
3949        if let Some(profile) = &profile {
3950            profile.moved(Stage::Publish, 0, catalog.len() as u64, 0);
3951        }
3952        // Every page and every table directory is on the disk before anything points at them. The
3953        // slot write below is what makes this generation the one a reader picks, so the order of
3954        // these two syncs is the whole of the commit.
3955        synced(&*self.file, profile.as_deref())?;
3956        let slot = Slot {
3957            offset,
3958            length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
3959            generation: self.generation,
3960            hash: checksum(&catalog),
3961        };
3962        // The one write that is not an append, and the last one. It goes back over the slot in the
3963        // header, so it names its offset rather than going through `put`, and `at` does not move.
3964        // Which of the two slots it is alternates with the generation, so the one naming the
3965        // generation before this is still intact and still valid until this write lands.
3966        self.file.write_at(slot_offset(self.generation), &slot.bytes())?;
3967        synced(&*self.file, profile.as_deref())?;
3968        Ok(self.table)
3969    }
3970
3971    /// Commits a generation that changes the views and leaves every table exactly where it is.
3972    ///
3973    /// There was no way to do this before views existed, because everything that could change the
3974    /// catalog also wrote a table, so the only way to say something new about a file was to go
3975    /// through a table. A view is the first thing that can change on its own. Without this, adding
3976    /// a view to a database with eight tables in it would rewrite all eight, since the append path
3977    /// needs a table to append and the fallback is the whole file.
3978    ///
3979    /// It is the same commit as [`Writer::finish`] with nothing appended before it. The table
3980    /// entries are carried forward by directory pointer the way an append carries them, the new
3981    /// catalog goes on the end, and the slot write at the end is what publishes it.
3982    ///
3983    /// The log anchor is `anchor` when there is one and the one the file holds when not.
3984    ///
3985    /// # Errors
3986    ///
3987    /// If the file has no valid committed directory, is not this build's format, or cannot be
3988    /// written.
3989    pub fn restate(
3990        path: impl AsRef<Path>,
3991        views: &[ViewEntry],
3992        anchor: Option<&LogAnchor>,
3993    ) -> Result<()> {
3994        let file = RealFilesystem::new().open(path.as_ref(), OpenMode::ReadWrite)?;
3995        let size = file.len()?;
3996        let (slot, bytes, _) = committed_slot(&*file, size)?;
3997        let (closed, _, card, held) = decode_catalog(&bytes, size)?;
3998        let anchor = anchor.cloned().or(held);
3999        let generation = slot
4000            .generation
4001            .checked_add(1)
4002            .ok_or_else(|| invalid("native file generation overflow"))?;
4003        let catalog = encode_catalog(
4004            &closed,
4005            views,
4006            card_for(path.as_ref(), card).as_ref(),
4007            anchor.as_ref(),
4008        )?;
4009        if catalog.len() > MAX_DIRECTORY {
4010            return Err(invalid("catalog exceeds the configured bound"));
4011        }
4012        file.write_at(size, &catalog)?;
4013        file.sync()?;
4014        let slot = Slot {
4015            offset: size,
4016            length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
4017            generation,
4018            hash: checksum(&catalog),
4019        };
4020        file.write_at(slot_offset(generation), &slot.bytes())?;
4021        file.sync()?;
4022        Ok(())
4023    }
4024
4025    /// Commits a generation that writes down the device card this process has for the device the
4026    /// file is on, and changes nothing else. It writes nothing when the file already holds that
4027    /// card or the process has none.
4028    ///
4029    /// This is what `PRAGMA device_card_refresh` calls after it measures. Any other commit writes
4030    /// the card too, but a refresh that changes nothing else has no commit to ride on.
4031    ///
4032    /// # Errors
4033    ///
4034    /// The same as [`Writer::restate`].
4035    pub fn keep_device_card(path: impl AsRef<Path>) -> Result<()> {
4036        let path = path.as_ref();
4037        let (_, size, _, bytes, _) = slot_bytes(path)?;
4038        let (_, views, held, _) = decode_catalog(&bytes, size)?;
4039        if card_for(path, held.clone()) == held {
4040            return Ok(());
4041        }
4042        Self::restate(path, &views, None)
4043    }
4044
4045    /// Adds exact count, sum, distinct, bound, and bounded frequency certificates to an older file without
4046    /// rewriting table pages. The old slot remains readable until the new catalog is synced.
4047    pub fn certify_summaries(path: impl AsRef<Path>) -> Result<()> {
4048        let path = path.as_ref();
4049        let (_, size, slot, bytes, _) = slot_bytes(path)?;
4050        let (mut entries, views, card, anchor) = decode_catalog(&bytes, size)?;
4051        let native = Catalog::open(path)?;
4052        for entry in &mut entries {
4053            let reader = native.table(&entry.name)?;
4054            entry.nonzero.fill(None);
4055            entry.aggregates = reader_aggregate_sums(&reader)?;
4056            entry.distincts = (0..entry.fields.len())
4057                .map(|column| reader.distinct_values(column))
4058                .collect::<Result<Vec<_>>>()?;
4059            entry.extremes = reader_integer_extremes(&reader)?;
4060            entry.frequencies = reader_complete_numeric_frequencies(&reader)?;
4061        }
4062        let generation = slot
4063            .generation
4064            .checked_add(1)
4065            .ok_or_else(|| invalid("native file generation overflow"))?;
4066        let catalog =
4067            encode_catalog(&entries, &views, card_for(path, card).as_ref(), anchor.as_ref())?;
4068        if catalog.len() > MAX_DIRECTORY {
4069            return Err(invalid("catalog exceeds the configured bound"));
4070        }
4071        let file = RealFilesystem::new().open(path, OpenMode::ReadWrite)?;
4072        file.write_at(size, &catalog)?;
4073        file.sync()?;
4074        let slot = Slot {
4075            offset: size,
4076            length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
4077            generation,
4078            hash: checksum(&catalog),
4079        };
4080        file.write_at(slot_offset(generation), &slot.bytes())?;
4081        file.sync()?;
4082        Ok(())
4083    }
4084
4085    /// The earlier name for [`Self::certify_summaries`].
4086    pub fn certify_counts(path: impl AsRef<Path>) -> Result<()> {
4087        Self::certify_summaries(path)
4088    }
4089}
4090
4091/// Appends one run of bytes at `at` and moves it past them, answering where they went.
4092///
4093/// The append half of [`attach`], which cannot use [`Writer::put`] because it is not writing a
4094/// table. Every byte a section costs goes through here, so the offsets in an extent table come
4095/// from one place.
4096fn append(file: &dyn rudb_io::File, at: &mut u64, bytes: &[u8]) -> Result<u64> {
4097    let offset = *at;
4098    file.write_at(offset, bytes)?;
4099    *at =
4100        at.checked_add(bytes.len() as u64).ok_or_else(|| invalid("native file length overflow"))?;
4101    Ok(offset)
4102}
4103
4104/// Writes one attachment's payload as extents and returns the entry that names it.
4105///
4106/// The split is by bytes, and the `first` of each extent is therefore a byte count. A section kind
4107/// whose extents should break on a row boundary instead will want to hand its extents over already
4108/// split; nothing needs that yet, and guessing at the shape of it now would be guessing.
4109fn write_section(
4110    file: &dyn rudb_io::File,
4111    at: &mut u64,
4112    one: &section::Attachment<'_>,
4113    generation: u64,
4114) -> Result<Section> {
4115    // A payload of nothing is the exception, and it is not a special case so much as a different
4116    // reading of the same field: an entry with no bytes has no header to be longer than them, and
4117    // `header_bytes` is what the structure would have cost. See `Section::refused`.
4118    if !one.bytes.is_empty() && one.header_bytes as usize > one.bytes.len() {
4119        return Err(invalid("a section's header is longer than its payload"));
4120    }
4121    let mut extents = Vec::new();
4122    let mut first = 0_u64;
4123    let extent_size =
4124        if one.kind == *section::RUN_PROJECTION && one.flags == run_projection::RLE_PAGES {
4125            run_projection::RLE_PAGE_BYTES
4126        } else if one.kind == *section::SORTED_PROJECTION || one.kind == *section::RUN_PROJECTION {
4127            1 << 19
4128        } else {
4129            section::MAX_EXTENT as usize
4130        };
4131    for chunk in one.bytes.chunks(extent_size) {
4132        let offset = append(file, at, chunk)?;
4133        extents.push(section::Extent {
4134            offset,
4135            length: u32::try_from(chunk.len()).map_err(|_| invalid("extent length overflow"))?,
4136            hash: checksum(chunk),
4137            first,
4138        });
4139        first += chunk.len() as u64;
4140    }
4141    let mut table = Vec::with_capacity(extents.len() * section::EXTENT_BYTES);
4142    section::encode_extents(&extents, &mut table)?;
4143    // A payload of nothing is a section of no extents and no extent table, and its `extent_page`
4144    // is zero rather than the end of the file. Section 3.7 wants that entry to exist: it is how a
4145    // relationship that did not fit the budget is recorded as not built rather than forgotten.
4146    let extent_page = if table.is_empty() { 0 } else { append(file, at, &table)? };
4147    Ok(Section {
4148        kind: one.kind,
4149        id: one.id,
4150        generation,
4151        extents: u32::try_from(extents.len()).map_err(|_| invalid("too many extents"))?,
4152        extent_page,
4153        extent_bytes: u32::try_from(table.len()).map_err(|_| invalid("extent table overflow"))?,
4154        hash: checksum(&table),
4155        flags: one.flags,
4156        header_bytes: one.header_bytes,
4157    })
4158}
4159
4160/// Attaches graph sections to a table already committed in a file, without rewriting a page.
4161///
4162/// This is the second pass spec/graph/03-the-file-format.md section 3.8 asks for. A key map has to
4163/// exist before the link that uses it can be built, and it is built by reading the key column back,
4164/// so the structures of a table cannot be written during the load that wrote the table. They are
4165/// written afterwards, by this, and the file in between the two is a correct file that answers
4166/// every query more slowly.
4167///
4168/// Nothing is overwritten. The payloads, the extent tables, the new directory for this table and
4169/// the new catalog all go on the end of the file past the committed generation, and the last write
4170/// is the header slot, exactly as [`Writer::finish`] does it. So a crash anywhere in here leaves
4171/// the generation before it intact, and leaves unreferenced trailing bytes that the next commit
4172/// writes past.
4173///
4174/// An attachment replaces any section of the same kind and id, and every other section is carried
4175/// through untouched, including one whose kind this build does not know. The table's own generation
4176/// is carried through too, because attaching a section moves no row: see [`Table::generation`].
4177///
4178/// # Errors
4179///
4180/// If the file has no valid committed directory, is an older format than this build writes, holds
4181/// no table of that name, names a section whose payload cannot be written, or would end up naming
4182/// more sections than the format allows.
4183pub fn attach(
4184    path: impl AsRef<Path>,
4185    table: &str,
4186    attachments: &[section::Attachment<'_>],
4187) -> Result<Table> {
4188    let file = RealFilesystem::new().open(path.as_ref(), OpenMode::ReadWrite)?;
4189    let file = &*file;
4190    let size = file.len()?;
4191    let (slot, bytes, _) = committed_slot(file, size)?;
4192    let (mut entries, views, card, anchor) = decode_catalog(&bytes, size)?;
4193    let card = card_for(path.as_ref(), card);
4194    let at = entries
4195        .iter()
4196        .position(|entry| entry.name == table)
4197        .ok_or_else(|| invalid(&format!("the file holds no table called {table}")))?;
4198    let mut version = [0; 4];
4199    read_at(file, 8, &mut version)?;
4200    let version = u32::from_le_bytes(version);
4201    // Readable is not the same as writable. A format 22 file has no section table, and giving its
4202    // directory one without moving the number in its header would leave a file that claims to be
4203    // format 22 and is not, which is worse than refusing. Rewriting it with this build is the
4204    // answer, and the format is at 0.3.x, so nobody has one of these that this project did not
4205    // just make.
4206    if version != FORMAT {
4207        return Err(invalid(&format!(
4208            "the file is format {version} and a graph section needs format {FORMAT}, so it has \
4209             to be written again"
4210        )));
4211    }
4212    let mut directory = vec![0; entries[at].directory.length as usize];
4213    read_at(file, entries[at].directory.offset, &mut directory)?;
4214    if checksum(&directory) != entries[at].directory.hash {
4215        return Err(invalid(&format!("the directory of table {table} does not checksum")));
4216    }
4217    let mut held = decode_directory(&directory, size)?;
4218    let mut cursor = size;
4219    for one in attachments {
4220        let written = write_section(file, &mut cursor, one, held.generation)?;
4221        held.sections.retain(|old| !(old.kind == one.kind && old.id == one.id));
4222        held.sections.push(written);
4223    }
4224    if held.sections.len() > MAX_SECTIONS {
4225        return Err(invalid("the table would name more sections than the bound allows"));
4226    }
4227    let encoded = encode_directory(&held)?;
4228    if encoded.len() > MAX_DIRECTORY {
4229        return Err(invalid("directory exceeds the configured bound"));
4230    }
4231    let offset = append(file, &mut cursor, &encoded)?;
4232    entries[at].directory = Page {
4233        offset,
4234        length: u32::try_from(encoded.len()).map_err(|_| invalid("directory length overflow"))?,
4235        hash: checksum(&encoded),
4236    };
4237    // The views the file already had, written back unchanged. Attaching a section to a table says
4238    // nothing about a view and must not drop one.
4239    let catalog = encode_catalog(&entries, &views, card.as_ref(), anchor.as_ref())?;
4240    if catalog.len() > MAX_DIRECTORY {
4241        return Err(invalid("catalog exceeds the configured bound"));
4242    }
4243    let offset = append(file, &mut cursor, &catalog)?;
4244    file.sync()?;
4245    let generation =
4246        slot.generation.checked_add(1).ok_or_else(|| invalid("native file generation overflow"))?;
4247    let committed = Slot {
4248        offset,
4249        length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
4250        generation,
4251        hash: checksum(&catalog),
4252    };
4253    file.write_at(slot_offset(generation), &committed.bytes())?;
4254    file.sync()?;
4255    Ok(held)
4256}
4257
4258/// One column's frequency synopsis as values with their row counts, shared by every clone of a
4259/// reader.
4260type Synopsis = Arc<Vec<(Value, u64)>>;
4261
4262/// Reads committed native column pages without holding the table in memory.
4263#[derive(Debug, Clone)]
4264pub struct Reader {
4265    file: Arc<File>,
4266    /// The file mapped, shared with the catalog. See [`Catalog`].
4267    map: Option<Arc<Mapped>>,
4268    table: Arc<Table>,
4269    dictionaries: Arc<Vec<OnceLock<Arc<Vector>>>>,
4270    /// Held while a global dictionary is being opened, one per column.
4271    ///
4272    /// The [`OnceLock`] above says whether one has been opened, which is the question a reader that
4273    /// already has it needs answered and is free. It does not say whether one is being opened, and
4274    /// the difference matters because every worker of a scan wants the same dictionary at the same
4275    /// moment. Without this they all miss, all read the page, all verify it and all decode it, and
4276    /// all but one throw the answer away. ClickBench 38 reads the URL dictionary, which is 515,958
4277    /// entries, and was paying for it twice.
4278    loading: Arc<Vec<Mutex<()>>>,
4279    /// Each column's frequency synopsis as values, the first time anything asks for it. See
4280    /// [`Reader::decode_frequencies`].
4281    frequency_values: Arc<Vec<OnceLock<Synopsis>>>,
4282    /// Stored frequency sections are decoded once per open table. A small directory can hold the
4283    /// summary inline, but a larger one otherwise rereads and decodes the same section on every
4284    /// plan and every summary-backed aggregate.
4285    frequency_summaries: Arc<Vec<OnceLock<Arc<FrequencySummary>>>>,
4286    /// The entries of each stored synopsis and the bound on what they leave out, read without the
4287    /// row ordinals behind them. A one column count reads only these, and the ordinals of a column
4288    /// like `UserID` are most of a megabyte.
4289    frequency_heads: Arc<Vec<OnceLock<Arc<FrequencyHead>>>>,
4290    /// Each column's summary, the first time anything asks for it. See `stats::held_summary`.
4291    summaries: Arc<Vec<OnceLock<Option<Arc<rudb_stats::Summary>>>>>,
4292    /// The distinct counts, orders and widths the planner reads off the table, gathered the first
4293    /// time a plan asks. See [`facts`].
4294    facts: Arc<OnceLock<Arc<rudb_common::ColumnFacts>>>,
4295    /// How many global dictionaries have been opened. A scan of a dictionary column should open its
4296    /// dictionary once however many workers it has, and the test that says so is the only thing
4297    /// keeping it that way.
4298    opened: Arc<AtomicUsize>,
4299    /// The membership sieves of one stripe of one column, by column and then by stripe, read the
4300    /// first time a probe asks about them. A query filters on one or two columns and never looks at
4301    /// the rest, so reading these at open would be the whole index for the sake of a fraction of it.
4302    /// A column's slots are made the first time it is asked about, for the same reason.
4303    sieves: Arc<Vec<OnceLock<Box<[SieveSlot]>>>>,
4304    /// The per part ranges of one stripe of one column, by column and then by stripe, read the
4305    /// first time something compares that column and kept after that.
4306    part_ranges: Arc<Vec<OnceLock<Box<[RangeSlot]>>>>,
4307    /// Which stripe and which part of it every part of the table is, by table wide part number.
4308    places: Arc<Vec<Place>>,
4309    cache: Arc<Shelf>,
4310    /// Where the pages above are counted against the database's budget. See [`PagePool`].
4311    pool: PagePool,
4312    /// How many whole stripe pages have been read, which is what the sharing above is judged on. A
4313    /// scan of a column should read each of its stripes once however many workers it has.
4314    pages: Arc<AtomicUsize>,
4315    /// How many index sections have been read. A scan of a column should read each of its stripes
4316    /// once here too, and the test that says so is the only thing keeping it that way.
4317    indexes: Arc<AtomicUsize>,
4318    /// Which parts of which columns have matched their checksums, a bit per part of the table for
4319    /// each column in turn.
4320    ///
4321    /// A part is written once and a later generation writes its parts somewhere else, so bytes
4322    /// that matched once match for as long as this reader is open. The page cache keeps the same
4323    /// promise for as long as it holds a page, and this one outlives the page. A scan the graph
4324    /// layer reduces reads a part at the rows it keeps and not the stripe's page, and each of those
4325    /// reads hashed the whole part again: on TPC-H q21, which reads `lineitem` three times, that was
4326    /// 4 percent of the query.
4327    verified: Arc<Vec<AtomicU64>>,
4328    /// How many parts of each stripe's page of each column are still to be read out of the mapped
4329    /// file before the page is let go, stripe by stripe and each column in turn.
4330    ///
4331    /// A page and not a part at a time, because letting go of mapped pages is a call that stops
4332    /// every thread of the process to flush what it had mapped, and a part at a time that was twice
4333    /// the kernel instructions on ClickBench 10. A stripe's page is let go by whichever worker
4334    /// reads its last part. See [`Mapped::release`].
4335    unreleased: Arc<Vec<AtomicU32>>,
4336    /// Each text column's [`grams`] sketch in row id order, read the first time a `LIKE` asks
4337    /// about the column, and `None` when the table carries none for it.
4338    text_grams: Arc<Vec<OnceLock<Option<Vec<u64>>>>>,
4339    /// The row id of every part's first row, by table wide part number.
4340    firsts: Arc<Vec<usize>>,
4341    /// The key maps, links and adjacencies of this table, each decoded the first time a plan asks.
4342    /// See [`graph::Decoded`].
4343    graph: Arc<graph::Decoded>,
4344    /// The file's size when it was opened, for [`Reader::layout`].
4345    size: u64,
4346    /// The committed directory's size, for [`Reader::layout`].
4347    directory: u64,
4348    /// What opening the file cost, which is a number rather than a claim.
4349    opening: Opening,
4350}
4351
4352/// What [`Reader::open`] read before it returned.
4353///
4354/// `spec/stats/04-in-memory.md` section 4.2 says opening a table reads the header and the directory
4355/// and nothing else, and once that document's statistics are in the file the tempting change is to
4356/// load a column summary or two on the way past, because they are small and the next query will
4357/// want them. A hundred milliseconds of that is a hundred milliseconds nobody asked for, and an
4358/// embedded database is opened by processes that are about to run one trivial query.
4359///
4360/// So the claim gets a number. Both of these are fixed by the schema and the stripe count and are
4361/// independent of how many rows the file holds, and the test that says so is what stops the
4362/// tempting change from landing quietly.
4363#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
4364pub struct Opening {
4365    /// How many times the file was read. The header, then each directory slot that looked valid
4366    /// enough to check, so three at the most.
4367    pub reads: u32,
4368    /// How many bytes those reads asked for.
4369    pub bytes: u64,
4370}
4371
4372/// What a reader has read, while it was being opened and since.
4373#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
4374pub struct Reads {
4375    /// What opening cost, before any query had been planned.
4376    pub opening: Opening,
4377    /// Whole stripe pages read since.
4378    pub pages: usize,
4379    /// Index sections read since.
4380    pub indexes: usize,
4381    /// Global dictionaries opened since. One per dictionary column that a query touched, however
4382    /// many workers touched it, which is a claim only a test can keep true.
4383    pub dictionaries: usize,
4384}
4385
4386/// Where one table wide part number lands.
4387#[derive(Debug, Clone, Copy)]
4388struct Place {
4389    stripe: u32,
4390    part: u32,
4391    rows: u32,
4392}
4393
4394/// One part's bytes inside one column page.
4395#[derive(Debug, Clone, Copy)]
4396struct PartSpan {
4397    start: usize,
4398    length: usize,
4399    hash: u64,
4400}
4401
4402/// What a reader holds for one stripe of one column.
4403///
4404/// The index is small and is loaded whether the caller wants the whole page or one part of it. The
4405/// page is loaded only by a scan, because a sparse fetch that wants a thousand rows out of sixty
4406/// four thousand would be reading sixty four times what it uses.
4407#[derive(Debug, Clone)]
4408struct CachedColumn {
4409    stripe: usize,
4410    index: Arc<Vec<PartSpan>>,
4411    page: Option<Arc<HeldPage>>,
4412}
4413
4414/// One stripe's page of one column, with which of its parts have already matched their checksums.
4415///
4416/// The bytes never change once they are read, so a part that matched once matches for as long as
4417/// the page is held. Hashing it again on every read was 3.5% of a `GROUP BY CounterID` over the
4418/// held pages of the ClickBench sample, run seventy times in one process. A part read without its
4419/// page is still checked every time, since those bytes come fresh off the file.
4420#[derive(Debug)]
4421struct HeldPage {
4422    bytes: PageBytes,
4423    checked: Vec<AtomicBool>,
4424}
4425
4426/// Where a held page's bytes are: read into memory of its own, or a range of the mapped file.
4427#[derive(Debug)]
4428enum PageBytes {
4429    Read(Vec<u8>),
4430    Mapped { map: Arc<Mapped>, offset: u64, length: usize },
4431}
4432
4433impl PageBytes {
4434    fn bytes(&self) -> &[u8] {
4435        match self {
4436            Self::Read(bytes) => bytes,
4437            // The range was checked against the mapping when the page was taken.
4438            Self::Mapped { map, offset, length } => map.get(*offset, *length).unwrap_or_default(),
4439        }
4440    }
4441}
4442
4443impl HeldPage {
4444    fn bytes(&self) -> &[u8] {
4445        self.bytes.bytes()
4446    }
4447
4448    /// The bytes of part `part`, checked against `span` the first time anyone asks for them.
4449    fn part(&self, part: usize, span: PartSpan) -> Result<&[u8]> {
4450        let bytes = part_bytes(self.bytes(), span)?;
4451        let checked = self.checked.get(part).ok_or_else(|| invalid("part index out of range"))?;
4452        if !checked.load(Atomic::Relaxed) {
4453            verify_part(bytes, span)?;
4454            checked.store(true, Atomic::Relaxed);
4455        }
4456        Ok(bytes)
4457    }
4458}
4459
4460/// Checks one part's bytes against the hash its index carries for them.
4461fn verify_part(bytes: &[u8], span: PartSpan) -> Result<()> {
4462    let got = checksum(bytes);
4463    if got != span.hash {
4464        return Err(invalid(&format!(
4465            "column page checksum differs, part at {}+{} bytes, wanted {:016x} and got {got:016x}",
4466            span.start, span.length, span.hash,
4467        )));
4468    }
4469    Ok(())
4470}
4471
4472/// One column's stripes a reader holds, and which of them somebody is reading right now.
4473///
4474/// The pages are one slot per stripe of the table rather than a list of the ones being kept, so
4475/// finding a page is an index and not a walk. That matters because the walk happened under the
4476/// lock, once per part per column, and a scan that gives a whole stripe to each of thirty two
4477/// workers keeps enough pages that walking them was the longest thing the lock was held for. The
4478/// slots cost a pointer per stripe per column, which on the ClickBench file is eight kilobytes
4479/// against the forty megabytes of pages they point at. `order` is which of them are filled, oldest
4480/// first, because that is the one thing the slots cannot say by themselves.
4481///
4482/// `loading` is what keeps a scan from reading the same page once per worker. It is a list and not
4483/// a set because it holds at most one stripe per worker on the column and is walked far less often
4484/// than a hash of it would be built.
4485///
4486/// `index` is every index this reader has ever read for the column, one slot per stripe, and it is
4487/// never evicted. An index is a few hundred bytes and a page is a quarter of a megabyte, so the two
4488/// do not belong under the same budget. Riding in the page cache meant a worker that came back to a
4489/// stripe after its page had been evicted read the index again with it, which on the full
4490/// ClickBench file was about thirteen hundred reads out of a hundred and fourteen thousand.
4491///
4492/// `touched` is which parts of each stripe have been read, a bit a part. A part is read on its own
4493/// the first time and the stripe's page is read whole only when one of its parts is asked for again.
4494/// That is the rule [`NativeText`] follows for its decoded blocks: a page earns its memory by being
4495/// wanted a second time. A process that runs one statement, which is how a script or a benchmark
4496/// uses the engine, wants each part once, and holding whole pages for it is what its peak was made
4497/// of. On ClickBench 32 the pages of `WatchID` and `ClientIP`, each two megabytes a stripe, were half
4498/// of the 108 MB the query peaked at, and ClickBench 41 held a stripe of every filter column to use
4499/// a handful of parts out of each. Read a part at a time a scan costs more calls to read the same
4500/// bytes, which on the whole suite was lost in the noise.
4501///
4502/// A page read whole goes into the pool when every part of its stripe had been read before, which
4503/// is a second scan. When only some had, it is one scan asking for a part twice, the way a `LIKE`
4504/// asks a compressed text part whether it can answer and then reads it, and `passing` holds those
4505/// pages, oldest first, down to the column's floor. That is what keeps ClickBench 21 from pooling
4506/// every page of `URL` for a second scan that never comes.
4507///
4508/// The slots by stripe are empty until the column is first read, because a table as wide as the
4509/// ClickBench one has a hundred columns a query never reads, and a slot for every stripe of each of
4510/// them was most of a megabyte a process paid at open.
4511#[derive(Debug, Default)]
4512struct Cached {
4513    pages: Vec<Option<Resident>>,
4514    loading: Vec<usize>,
4515    index: Vec<Option<Arc<Vec<PartSpan>>>>,
4516    touched: Vec<Vec<u64>>,
4517    passing: VecDeque<usize>,
4518}
4519
4520/// One page a reader holds, and whether anyone has read it since the pool last looked.
4521#[derive(Debug, Clone)]
4522struct Resident {
4523    page: Arc<HeldPage>,
4524    used: Arc<AtomicBool>,
4525}
4526
4527/// Every column's pages of one reader, with how many each column holds and the floor under that.
4528#[derive(Debug)]
4529struct Shelf {
4530    columns: Vec<Mutex<Cached>>,
4531    /// How many pages each column holds right now. Counted outside the column locks so that the
4532    /// pool can tell whether a column is at its floor without taking a lock it might be under.
4533    held: Vec<AtomicUsize>,
4534    /// How many stripes of one column are kept whatever the budget says. See
4535    /// [`CACHED_STRIPES_PER_COLUMN`] for what sets it and [`Reader::keep_stripes`] for who raises it.
4536    kept: AtomicUsize,
4537}
4538
4539/// The pages every reader of one database keeps, under one budget in bytes.
4540///
4541/// A reader lives as long as the database does, so the pages it holds are what the next query finds
4542/// already in memory. They used to be four stripes a column, oldest out first, which on TPC-H SF1
4543/// meant every query read every page of lineitem off the file again and paid the system call for
4544/// it. Keeping every page there costs 38 MB and took a third of the system time off the suite.
4545///
4546/// So the question is no longer how many stripes a column keeps but how many bytes the database
4547/// does, and one budget answers it for every reader at once. A table nobody queries gives its pages
4548/// up to one that is being queried, which a count per column cannot do.
4549///
4550/// Pages leave by the clock. Each has a bit a read sets, and when the pool is over budget it walks
4551/// from the oldest: a page with the bit set loses the bit and goes round again, and a page without
4552/// it goes. That keeps what is read over and over and lets a page one scan read once go first.
4553///
4554/// The old count is still a floor. A column never gives up a page while it holds four or fewer,
4555/// because a scan whose workers evict each other's pages reads a quarter of a megabyte for every
4556/// part it takes, and a budget of zero is the cache as it was before the pool existed.
4557#[derive(Debug, Clone, Default)]
4558pub struct PagePool {
4559    ring: Arc<Mutex<Ring>>,
4560    budget: Arc<AtomicUsize>,
4561}
4562
4563#[derive(Debug, Default)]
4564struct Ring {
4565    held: VecDeque<Held>,
4566    bytes: usize,
4567}
4568
4569/// One page in the pool, pointing back at the reader that holds it.
4570///
4571/// Weak, because a reader that has gone, which every reader does at a checkpoint, should take its
4572/// pages with it and not have them kept alive by the pool.
4573#[derive(Debug)]
4574struct Held {
4575    shelf: Weak<Shelf>,
4576    column: usize,
4577    stripe: usize,
4578    bytes: usize,
4579    used: Arc<AtomicBool>,
4580}
4581
4582impl PagePool {
4583    /// A pool that keeps up to `budget` bytes of pages beyond each column's floor.
4584    #[must_use]
4585    pub fn new(budget: usize) -> Self {
4586        let pool = Self::default();
4587        pool.budget.store(budget, Atomic::Relaxed);
4588        pool
4589    }
4590
4591    /// The bytes of pages the pool is counting now.
4592    ///
4593    /// # Panics
4594    ///
4595    /// If the pool's lock is poisoned, which takes a panic while it was held.
4596    #[must_use]
4597    pub fn bytes(&self) -> usize {
4598        self.ring.lock().map_or(0, |ring| ring.bytes)
4599    }
4600
4601    /// Counts a page a reader has just taken in, and lets pages go until the pool is back under its
4602    /// budget or it has looked at every page once.
4603    ///
4604    /// Called with no column lock held. The pages that go are chosen under the pool's lock and
4605    /// dropped under their column's lock afterwards, so no thread ever holds both.
4606    fn admit(&self, held: Held) {
4607        let budget = self.budget.load(Atomic::Relaxed);
4608        let mut gone = Vec::new();
4609        {
4610            let Ok(mut ring) = self.ring.lock() else { return };
4611            ring.bytes += held.bytes;
4612            ring.held.push_back(held);
4613            // One lap and no more. A page read since the last pass loses its bit on this one and
4614            // can only go on a later one, which is the second chance the clock is named for.
4615            let mut looked = 0;
4616            let limit = ring.held.len();
4617            while ring.bytes > budget && looked < limit {
4618                looked += 1;
4619                let Some(entry) = ring.held.pop_front() else { break };
4620                let Some(shelf) = entry.shelf.upgrade() else {
4621                    ring.bytes -= entry.bytes;
4622                    continue;
4623                };
4624                if entry.used.swap(false, Atomic::Relaxed) {
4625                    ring.held.push_back(entry);
4626                    continue;
4627                }
4628                let count = &shelf.held[entry.column];
4629                if count.load(Atomic::Relaxed) <= shelf.kept.load(Atomic::Relaxed).max(1) {
4630                    ring.held.push_back(entry);
4631                    continue;
4632                }
4633                count.fetch_sub(1, Atomic::Relaxed);
4634                ring.bytes -= entry.bytes;
4635                gone.push((shelf, entry));
4636            }
4637            // A reader that has gone leaves its entries behind, and with a budget nobody reaches
4638            // they would pile up one checkpoint after another. The front is where the oldest are.
4639            while ring.held.front().is_some_and(|entry| entry.shelf.strong_count() == 0) {
4640                if let Some(entry) = ring.held.pop_front() {
4641                    ring.bytes -= entry.bytes;
4642                }
4643            }
4644        }
4645        for (shelf, entry) in gone {
4646            let Ok(mut cached) = shelf.columns[entry.column].lock() else { continue };
4647            if let Some(slot) = cached.pages.get_mut(entry.stripe)
4648                && slot.as_ref().is_some_and(|slot| Arc::ptr_eq(&slot.used, &entry.used))
4649            {
4650                *slot = None;
4651            }
4652        }
4653    }
4654}
4655
4656/// Stripes of one column a reader keeps the bytes of, when nobody has asked for more.
4657///
4658/// This has to hold at least as many stripes as a column has workers in it at once, or the workers
4659/// evict each other's pages and read them again. Four is what a scan that hands parts out in order
4660/// needs, because then every worker is within a few parts of every other and at most a couple of
4661/// stripes are open at a time. A scan that hands a whole stripe to each worker has one stripe open
4662/// per worker for the length of that stripe, and it says so with [`Reader::keep_stripes`] rather
4663/// than paying for sixteen slots on every table that is read one part at a time.
4664///
4665/// It multiplies by the page size, which is a quarter of a megabyte for a four byte column, and by
4666/// the number of columns a query touches.
4667const CACHED_STRIPES_PER_COLUMN: usize = 4;
4668
4669/// The sieves of one stripe of one column, once somebody has asked for them.
4670type SieveSlot = OnceLock<Arc<Vec<Option<Sieve>>>>;
4671
4672type RangeSlot = OnceLock<Arc<Vec<Range>>>;
4673
4674#[derive(Debug)]
4675struct NativeText {
4676    file: Arc<File>,
4677    /// How many values the dictionary holds.
4678    values: usize,
4679    /// Where each value ends inside its payload block, packed at `offset_bits` in runs of
4680    /// [`TEXT_OFFSET_RUN`].
4681    ///
4682    /// Ends rather than starts, because then a block of 1,024 values is 1,024 numbers rather than
4683    /// 1,025: the start of a value is the end of the one before it, and the first value of a block
4684    /// starts at zero by construction. Relative to the block rather than to the payload, because a
4685    /// reader decodes a whole block and slices it, so an offset into the payload is a number it
4686    /// would have to subtract a base from anyway.
4687    ///
4688    /// The vector is the index as it was read, so the offsets start after the header, and
4689    /// [`Self::packed`] is where they are read from.
4690    offsets: Vec<u8>,
4691    /// Bits one offset is packed at, which is what the largest block of this column spans and is the
4692    /// same for every block of it.
4693    offset_bits: usize,
4694    /// The same ends unpacked, built once enough readers have asked for one at a time.
4695    ///
4696    /// Reading one offset out of the packed form costs about fifty instructions: a division to find
4697    /// the run, a bounds check to slice it, a shift to reach the bit the value starts at and a
4698    /// narrowing on the way out. That is the right price for a reader that wants a handful. It is
4699    /// the wrong price for `STRLEN` over a column, which asks for one per row and nothing else, and
4700    /// where a million of them was a third of ClickBench 28.
4701    ///
4702    /// With the ends unpacked every read is a load, and a vector of lengths is one loop over them.
4703    /// The table is built only once the reads say it will be used, which is what
4704    /// [`Self::ends_worth_unpacking`] decides and [`Self::ends_asked`] counts towards, because a
4705    /// table built for a reader that wanted three values is four bytes a value spent on nothing.
4706    value_ends: OnceLock<Option<Vec<u32>>>,
4707    /// The length of every value, worked out of [`Self::value_ends`] the first time a vector of
4708    /// lengths is asked for.
4709    ///
4710    /// A length out of the ends is two loads, a test for whether the value opens its block and a
4711    /// check that it does not end before it starts, which came to thirteen instructions a row on
4712    /// ClickBench 28. Out of this it is one load. The order is checked once for the whole table
4713    /// while it is built, and a column that fails it gets no table and goes on reading the ends,
4714    /// which is where the error is reported. Two bytes a value where every value is short enough,
4715    /// four otherwise, and only for a column something has asked the length of a vector at a time.
4716    value_lens: OnceLock<Option<Lengths>>,
4717    /// How many single offset reads have come in while the table is not built.
4718    ///
4719    /// Relaxed, and read only against a threshold, so two threads racing here means the table is
4720    /// built one read early or one read late. Counting stops the moment the table exists, because
4721    /// [`OnceLock::get`] settles it before this is touched.
4722    ends_asked: AtomicUsize,
4723    /// How many entries the sorted order has, which is the value count.
4724    ranks: usize,
4725    /// Where the sorted order starts in the file. It is read a block at a time and only when
4726    /// something searches it, so a query that never compares this column against a literal never
4727    /// touches it at all.
4728    rank_at: u64,
4729    /// Where each block of the sorted order ends, as a byte offset from `rank_at`. A block is packed
4730    /// at whatever width its own heads need, so unlike the entries it replaced its length is not
4731    /// arithmetic on the block number.
4732    rank_ends: Vec<u64>,
4733    rank_hashes: Vec<u64>,
4734    rank_blocks: Vec<OnceLock<Result<Vec<u8>>>>,
4735    /// Bits one code is packed at, which is what the value count needs and is the same for every
4736    /// block of the column.
4737    code_bits: usize,
4738    /// The sorted order turned round, built the first time a reader asks for it.
4739    ///
4740    /// Four bytes per value against the four the offsets already hold, so a column that has this is
4741    /// carrying half again what it carried before rather than something of a new order. It is built
4742    /// only when something asks, which is a grouped min or max over this column and nothing else,
4743    /// and that reader was going to read the payload of this column once per row otherwise.
4744    code_ranks: OnceLock<Option<Vec<u32>>>,
4745    /// Where each block of the payload starts in the file, and how many stored bytes it is.
4746    ///
4747    /// Absolute rather than an offset from a base the blocks share, because a block is written the
4748    /// moment it fills and what comes after it in the file is whatever the load wrote next. A file
4749    /// old enough to have them back to back is read into these same two lists by adding the base to
4750    /// the ends it carries, so nothing below here knows which kind of file it came from.
4751    starts: Vec<u64>,
4752    lengths: Vec<u64>,
4753    hashes: Vec<u64>,
4754    /// Conservative four-byte substring signatures, read only by a compatible LIKE filter.
4755    grams: Option<NativeGrams>,
4756    /// The payload, read and decoded a block at a time and kept after that.
4757    blocks: Vec<OnceLock<Result<Vec<u8>>>>,
4758    /// The length in characters of every value of a block, worked out the first time `length` asks
4759    /// for a value in that block.
4760    ///
4761    /// Kept instead of the block it was counted out of. `length` reads every row of a column, and
4762    /// reading the bytes through [`Self::payload_block`] kept every block it touched, which is every
4763    /// distinct value of the column decoded: seven string columns of ClickBench held 13.9 GB to
4764    /// answer seven `max(length(...))`. The counts are four bytes a value, so the same scan keeps
4765    /// the counts and decodes each block once, the same number of times it did before.
4766    char_lens: Vec<OnceLock<Box<[u32]>>>,
4767    /// How many decoded payload bytes this column keeps before a sweep stops keeping what it reads.
4768    /// [`TEXT_KEEP_BUDGET`] everywhere but in the test of the ceiling.
4769    keep_budget: usize,
4770    /// Roughly how many decoded payload bytes are being kept, which is what [`TEXT_KEEP_BUDGET`]
4771    /// is measured against.
4772    ///
4773    /// Roughly, because two threads that keep the same block at the same time both add its length
4774    /// while [`OnceLock`] keeps one of the two. That makes the count read high and the budget bind
4775    /// a little early, which is the harmless direction, and it costs one relaxed add a block rather
4776    /// than a lock on the path every scan of a string column goes through.
4777    payload_kept: AtomicUsize,
4778    /// Which payload blocks a sweep has decoded before, one flag a block.
4779    ///
4780    /// A sweep keeps a block the second time it decodes it and not the first. A process that runs
4781    /// one statement, which is how a benchmark or a script uses the engine, sweeps each block once
4782    /// and so keeps nothing: on ten million rows a `URL LIKE` held 396 MB with every block kept and
4783    /// 97 MB with none, for the same processor time. A session that asks again pays the decode one
4784    /// more time and reads kept blocks from then on, under the same [`TEXT_KEEP_BUDGET`].
4785    swept: Vec<AtomicBool>,
4786    /// How many blocks [`TextSource::visit_at`] has decoded and dropped because the column was
4787    /// already holding its [`TEXT_KEEP_BUDGET`].
4788    ///
4789    /// A sweep reads the dictionary in order and touches a block once, so dropping what it reads
4790    /// past the budget costs one decode a block and bounds the column. A visit reads a vector of
4791    /// codes, and the codes of a scan land all over the dictionary: on ten million rows of
4792    /// ClickBench each vector of two thousand `URL`s touches about a hundred and forty of its two
4793    /// and a half thousand blocks, and so does the next one. A cache holding a tenth of the column
4794    /// still misses half of those, and dropping every block past the budget would decode the
4795    /// column hundreds of times over to answer one `lower(URL)`. So a visit drops past the budget
4796    /// only until it has dropped as many blocks as the column has, which is what a read whose codes
4797    /// are few or clustered never reaches, and keeps what it reads after that, the way a row at a
4798    /// time read always did. That bounds what a visit can cost over the old read at one more decode
4799    /// of the column.
4800    visit_dropped: AtomicUsize,
4801    /// The boundaries this dictionary has already been searched for, by the value searched for.
4802    ///
4803    /// A search is the expensive thing this type does. It settles a probe on the stored head where
4804    /// it can and reads a value where it cannot, and reading a value decodes the payload block it
4805    /// sits in, so one search can cost several blocks. The thing that makes remembering worth it is
4806    /// that the same search comes back: a top N asks once a chunk whether anything left can beat its
4807    /// worst candidate, and the worst candidate settles long before the chunks run out.
4808    ///
4809    /// Shared across the instances of a scan rather than kept per instance, because each of them has
4810    /// its own worst candidate and all of them are searching the same dictionary. One lock per chunk
4811    /// is nothing next to a probe of a file.
4812    ///
4813    /// Bounded by [`TEXT_SEARCH_MEMO`] and emptied rather than evicted when it is full. What fills
4814    /// it is a top N improving its bound, which happens a few dozen times and then stops, so the
4815    /// bound is there for the filter that searches for a different literal every chunk rather than
4816    /// for anything this is meant to help.
4817    searched: Mutex<HashMap<Vec<u8>, (usize, bool)>>,
4818}
4819
4820#[derive(Debug)]
4821struct NativeGrams {
4822    start: u64,
4823    length: usize,
4824    /// How long one block's signature is.
4825    width: usize,
4826    hash: u64,
4827    /// For each literal asked about lately, whether each block might hold it.
4828    ///
4829    /// The answer for every block at once, worked out by one pass over the signatures a window at a
4830    /// time, rather than the signatures read in and kept. On ClickBench `URL` they are 21 MB for
4831    /// ten million rows and a verdict is 2,650 flags, and a filter asks the same question of every
4832    /// block, so the pass is paid once and what stays resident is the flags.
4833    verdicts: Mutex<Vec<Verdict>>,
4834}
4835
4836/// A literal and whether each block might hold it.
4837type Verdict = (Vec<u8>, Arc<[bool]>);
4838
4839/// How many literals a column remembers the verdicts of.
4840const GRAM_VERDICTS: usize = 8;
4841
4842impl NativeGrams {
4843    /// Whether each block might hold `literal`, remembered or worked out now.
4844    ///
4845    /// The lock is held over the pass so that the threads of one scan, which all ask about the
4846    /// same literal at the start, read the signatures once between them.
4847    fn verdicts(&self, file: &File, literal: &[u8]) -> Result<Arc<[bool]>> {
4848        let mut held = self.verdicts.lock().map_err(|_| invalid("a poisoned signature verdict"))?;
4849        if let Some((_, verdict)) = held.iter().find(|(asked, _)| asked == literal) {
4850            return Ok(Arc::clone(verdict));
4851        }
4852        let wanted = literal.windows(4).map(|gram| gram_bits(gram, self.width)).collect::<Vec<_>>();
4853        let mut verdict = Vec::with_capacity(self.length / self.width);
4854        let window = GRAM_WINDOW / self.width * self.width;
4855        let hash = walk_checksummed(file, self.start, self.length, window, |bytes| {
4856            verdict.extend(bytes.chunks(self.width).map(|bits| {
4857                wanted
4858                    .iter()
4859                    .flatten()
4860                    .all(|&bit| bits.get(bit / 8).is_some_and(|byte| byte & (1 << (bit % 8)) != 0))
4861            }));
4862            Ok(())
4863        })?;
4864        if hash != self.hash {
4865            return Err(invalid("global dictionary substring signatures checksum differs"));
4866        }
4867        let verdict: Arc<[bool]> = verdict.into();
4868        if held.len() >= GRAM_VERDICTS {
4869            held.remove(0);
4870        }
4871        held.push((literal.to_vec(), Arc::clone(&verdict)));
4872        Ok(verdict)
4873    }
4874
4875    fn footprint(&self) -> usize {
4876        self.verdicts.lock().map_or(0, |held| {
4877            held.iter().map(|(asked, verdict)| asked.capacity() + verdict.len()).sum()
4878        })
4879    }
4880}
4881
4882/// How many searched for values a column's dictionary remembers the boundary of.
4883///
4884/// See [`NativeText::searched`]. Small because the case it is for repeats one value, not because a
4885/// larger one would be wrong.
4886const TEXT_SEARCH_MEMO: usize = 64;
4887
4888/// How many values of a dictionary go in one block of the payload.
4889///
4890/// The block is the unit the string cascade encodes, the unit a checksum covers, and the unit a
4891/// reader has to decode to get at a single value, so it is the one number the payload format turns
4892/// on. Blocking by values rather than by bytes is what keeps a value out of two blocks at once: the
4893/// block holding a code is `code / TEXT_PAYLOAD_VALUES` and nothing has to be stitched.
4894///
4895/// A probe on the five ClickBench columns that have a dictionary worth the name, written up on
4896/// #347, measured the ratio and the decode speed at 128, 256, 512, 1,024 and 4,096 values. Both get
4897/// better all the way up, because front coding and the LZ matcher have more to look back at and
4898/// because the per chunk setup is spread over more values. What stops it is the point read: a query
4899/// that wants ten values has to decode ten blocks, so the block is what a lookup costs. At 1,024
4900/// values a block is between 67 KB and 394 KB decoded across those five columns, and the ratios are
4901/// 2.3 to 4.5. Going up to 4,096 buys two to six percent more and makes a block as much as 1.5 MB.
4902/// Going down to 512 gives up five to nine percent.
4903const TEXT_PAYLOAD_VALUES: usize = 1024;
4904
4905/// Eight KiB per payload block, which is what makes a four-byte substring a useful negative test on
4906/// a column of URLs.
4907///
4908/// Two KiB was the first answer and on ClickBench `URL` it proved almost nothing. A block of 1,024
4909/// sorted URLs holds about seventeen thousand distinct four-byte grams, and at two bits each that
4910/// set nine in ten of the sixteen thousand bits there were, so `LIKE '%google%'` passed most blocks
4911/// it had no match in and decoded them. At eight KiB four bits in ten are set, and of the 2,650
4912/// blocks of `URL` in ten million rows a needle that is in none of them passes 36. The signatures
4913/// are not read into memory, see [`NativeGrams::verdicts`], so the width costs file and not
4914/// resident memory.
4915const TEXT_GRAM_BYTES: usize = 8192;
4916
4917/// The signature width of a format 28 file, which is still read.
4918const NARROW_GRAM_BYTES: usize = 2048;
4919
4920/// How much of a column's signatures a verdict reads at a time.
4921const GRAM_WINDOW: usize = 256 << 10;
4922
4923/// A fast mixing step for exactly four bytes, shared by load and query, into a signature of
4924/// `width` bytes.
4925fn gram_bits(bytes: &[u8], width: usize) -> [usize; 2] {
4926    let original = u32::from_le_bytes(bytes.try_into().expect("a four-byte gram"));
4927    let mut first = original ^ (original >> 16);
4928    first = first.wrapping_mul(0x7feb_352d);
4929    first ^= first >> 15;
4930    let mut second = original ^ (original >> 17);
4931    second = second.wrapping_mul(0x846c_a68b);
4932    second ^= second >> 16;
4933    let mask = width * 8 - 1;
4934    [(first as usize) & mask, (second as usize) & mask]
4935}
4936
4937/// How many decoded payload bytes one dictionary keeps before a sweep stops keeping what it reads.
4938///
4939/// A sweep of the whole dictionary decodes every block whatever it does, and the only question is
4940/// whether it hangs on to them. Keeping all of them is 4.2 GB on ClickBench `URL` at a hundred
4941/// million rows, which is what #997 was right to stop. Keeping none of them means the next query
4942/// asking the same thing decodes all of it again, and on the same column at a million rows that
4943/// took a `LIKE` from 2.7 ms to 16.2 ms, because the decode used to be paid once by a session and
4944/// is now paid by every statement in it. Neither end is the answer. A bound is.
4945///
4946/// So a sweep keeps what it decodes until the column is holding this much and decodes without
4947/// keeping after that. At a million rows the five ClickBench string columns decode to between 8 MB
4948/// and 85 MB, so they sit inside it and a repeated `LIKE` reads a decoded block rather than a
4949/// stored one. At a hundred million rows `URL` fills it and the rest of that column is read and
4950/// dropped, which is the old cost on the part that does not fit and none of the old footprint.
4951///
4952/// Two hundred and fifty six megabytes a column is a number and not a policy, and the policy is
4953/// what should replace it: this wants to be a buffer pool over the whole database, sized against
4954/// the memory limit the session was given, with the blocks of every column competing for it and the
4955/// least useful one evicted. That is F2 work. What is here is the part of it that can be written
4956/// without an eviction order, which is a ceiling.
4957const TEXT_KEEP_BUDGET: usize = 256 * 1024 * 1024;
4958
4959/// The length of every value of a column, as narrow as the longest of them allows.
4960///
4961/// The table is read at the codes a vector holds, which on a column the size of ClickBench `URL`
4962/// land all over it, so what a length costs is whether its line is in cache. Half a million URLs
4963/// are two megabytes at four bytes a length and one at two, which is the difference between the
4964/// table sitting in the second level cache or not.
4965#[derive(Debug)]
4966enum Lengths {
4967    /// Every length fits in sixteen bits.
4968    Narrow(Vec<u16>),
4969    /// Some value is longer than that.
4970    Wide(Vec<u32>),
4971}
4972
4973impl Lengths {
4974    /// The lengths at `indices`, appended to `into`, and zero for a position past the end, which
4975    /// is what a row at a time read says.
4976    fn extend_at(&self, indices: &[u32], into: &mut Vec<i64>) {
4977        match self {
4978            Lengths::Narrow(lens) => into.extend(
4979                indices
4980                    .iter()
4981                    .map(|&index| lens.get(index as usize).map_or(0, |&len| i64::from(len))),
4982            ),
4983            Lengths::Wide(lens) => into.extend(
4984                indices
4985                    .iter()
4986                    .map(|&index| lens.get(index as usize).map_or(0, |&len| i64::from(len))),
4987            ),
4988        }
4989    }
4990
4991    /// The bytes the table holds on to.
4992    fn footprint(&self) -> usize {
4993        match self {
4994            Lengths::Narrow(lens) => lens.capacity() * size_of::<u16>(),
4995            Lengths::Wide(lens) => lens.capacity() * size_of::<u32>(),
4996        }
4997    }
4998}
4999
5000/// The length of every value out of where each one ends inside its payload block, or `None` for
5001/// ends that go backwards somewhere inside a block.
5002///
5003/// A value that opens a block starts at zero and every other one starts where the value before it
5004/// ends, so a block is a run of differences.
5005///
5006/// Built at two bytes a length straight away, and built again at four only when some value turns
5007/// out too long for that, which is rare enough that the second pass is not worth avoiding.
5008fn lengths_of(ends: &[u32]) -> Option<Lengths> {
5009    match lengths_as::<u16>(ends)? {
5010        Some(narrow) => Some(Lengths::Narrow(narrow)),
5011        None => lengths_as::<u32>(ends)?.map(Lengths::Wide),
5012    }
5013}
5014
5015/// [`lengths_of`] at one width: `None` for ends that go backwards, and `Some(None)` for a length
5016/// that does not fit in `T`.
5017fn lengths_as<T: TryFrom<u32>>(ends: &[u32]) -> Option<Option<Vec<T>>> {
5018    let mut lens = Vec::with_capacity(ends.len());
5019    for block in ends.chunks(TEXT_PAYLOAD_VALUES) {
5020        let mut start = 0;
5021        for &end in block {
5022            let Ok(len) = T::try_from(end.checked_sub(start)?) else {
5023                return Some(None);
5024            };
5025            lens.push(len);
5026            start = end;
5027        }
5028    }
5029    Some(Some(lens))
5030}
5031
5032/// How many offsets go in one packed run.
5033///
5034/// A payload block holds 1,024 values and `bitpack::pack_tail` takes fewer than 1,024 at a time,
5035/// since a whole unit of that many belongs in the transposed layout instead. So the offsets of a
5036/// block go in two runs. Five hundred and twelve values at any width is a whole number of bytes, so
5037/// a run starts where a multiply says it does and nothing is padded.
5038const TEXT_OFFSET_RUN: usize = 512;
5039
5040/// Bytes at the front of a global dictionary index: the value count, the values a payload block
5041/// holds, the block count and the bits an offset is packed at.
5042const DICTIONARY_HEADER: usize = 16;
5043
5044/// Set beside the offset width in the fourth word of a global dictionary index, meaning each
5045/// payload block says where in the file it starts and how long it is, rather than sitting directly
5046/// behind the block before it.
5047///
5048/// In that word rather than in a word of its own because the width is at most 32 and lives in a
5049/// `u32`, so the top of it has never been anything. A build old enough not to know the flag reads
5050/// the file's format before it reads any of this and refuses it there, and if it somehow did get
5051/// here it would find an offset width of two billion and say so.
5052///
5053/// The point of the flag is that a block written the moment it fills does not know what will be
5054/// written after it, so the payload of a column cannot be one run of bytes unless the whole column
5055/// is held until the file is closed. That is the memory the load cannot afford. What it costs is
5056/// eight bytes a block, against the block being a thousand values.
5057const DICTIONARY_SCATTERED: u32 = 1 << 31;
5058/// The dictionary index carries one four-byte substring signature per payload block.
5059const DICTIONARY_GRAMS: u32 = 1 << 30;
5060/// Each signature is [`TEXT_GRAM_BYTES`] long rather than the [`NARROW_GRAM_BYTES`] a format 28
5061/// file wrote.
5062const DICTIONARY_WIDE_GRAMS: u32 = 1 << 29;
5063/// Every flag the width word of a dictionary can carry above the offset width.
5064const DICTIONARY_FLAGS: u32 = DICTIONARY_SCATTERED | DICTIONARY_GRAMS | DICTIONARY_WIDE_GRAMS;
5065
5066/// How many entries of a dictionary's sorted order sit in one block that is read and checked as a
5067/// unit.
5068///
5069/// Five hundred and twelve entries is between two and three kilobytes on the ClickBench string
5070/// columns, which is well under a page. A binary search over half a million entries makes nineteen
5071/// probes, and the first ten land in ten different blocks while the last nine land in the one block
5072/// that holds the answer, so the whole search reads about thirty kilobytes of a megabyte of order. A
5073/// smaller block would save a little on the early probes, cost a checksum and an end list four times
5074/// as long, and give the heads less to share a base with. A larger one would read more than it uses
5075/// on every probe.
5076const TEXT_RANK_BLOCK: usize = 512;
5077
5078/// Bytes at the front of a rank block, which is the base of its heads and the width they are packed
5079/// at.
5080///
5081/// An entry used to be twelve bytes flat, eight for the head and four for the code, and on the five
5082/// ClickBench columns that have a dictionary worth the name that was 744 MB of a 12.2 GB file. Both
5083/// halves of it are nearly empty. The heads are the first eight bytes of the values in sorted order,
5084/// so a block of five hundred and twelve of them spans a tiny slice of the column, and on a column of
5085/// URLs they are all `http://w` and the block holds one distinct head. The codes are positions in a
5086/// dictionary of eighteen million, which is twenty five bits and not thirty two.
5087///
5088/// So a block now writes the smallest head in it, the bits the largest is above that, and the heads
5089/// and the codes packed at the width each needs. A block where every head agrees costs nine bytes
5090/// and the codes.
5091const RANK_BLOCK_HEADER: usize = size_of::<u64>() + 1;
5092
5093impl NativeText {
5094    /// One block of the payload, read and decoded the first time anything asks for a value in it.
5095    ///
5096    /// The bytes handed back are the values of the block laid end to end, which is what the offsets
5097    /// describe, so a caller slices it with the offsets it already has. Where the block sits in the
5098    /// file is the only thing the caller cannot work out for itself, because the stored form is
5099    /// shorter than the decoded one and by a different amount in every block.
5100    fn payload_block(&self, block: usize) -> Result<Option<&[u8]>> {
5101        let Some(slot) = self.blocks.get(block) else { return Ok(None) };
5102        let bytes = slot.get_or_init(|| self.decode_block(block)).as_ref().map_err(Clone::clone)?;
5103        Ok(Some(bytes.as_slice()))
5104    }
5105
5106    /// The character length of every value in one block, counted the first time it is asked for.
5107    ///
5108    /// The block is read out of [`Self::blocks`] where something already kept it and decoded and
5109    /// dropped where nothing did, so counting never adds a block to what this column holds. Two
5110    /// threads asking for the same block at once both count it and one of the two answers is kept,
5111    /// which costs a decode and is cheaper than a lock on every lookup.
5112    fn block_chars(&self, block: usize) -> Result<&[u32]> {
5113        let slot = self
5114            .char_lens
5115            .get(block)
5116            .ok_or_else(|| invalid("a block past the global dictionary"))?;
5117        if let Some(lens) = slot.get() {
5118            return Ok(lens);
5119        }
5120        let decoded;
5121        let bytes: &[u8] = match self.blocks.get(block).and_then(OnceLock::get) {
5122            Some(Ok(kept)) => kept,
5123            _ => {
5124                decoded = self.decode_block(block)?;
5125                &decoded
5126            }
5127        };
5128        let first = block * TEXT_PAYLOAD_VALUES;
5129        let last = (first + TEXT_PAYLOAD_VALUES).min(self.values);
5130        let ends = self.ends_within(first, last)?;
5131        if ends.len() != last - first {
5132            return Err(invalid("global dictionary offsets are short"));
5133        }
5134        let mut lens = Vec::with_capacity(ends.len());
5135        let mut start = u64::from(self.start_within(first)?);
5136        for &end in &ends {
5137            let value = usize::try_from(start)
5138                .ok()
5139                .zip(usize::try_from(end).ok())
5140                .and_then(|(from, to)| bytes.get(from..to))
5141                .ok_or_else(|| invalid("global dictionary value is past its block"))?;
5142            // A continuation byte of UTF-8 is `0b10xx_xxxx` and every other byte starts a
5143            // character, so the bytes that are not continuations are the characters.
5144            let characters = value.iter().filter(|byte| (**byte as i8) >= -0x40).count();
5145            lens.push(u32::try_from(characters).unwrap_or(u32::MAX));
5146            start = end;
5147        }
5148        Ok(slot.get_or_init(|| lens.into_boxed_slice()))
5149    }
5150
5151    /// Reads and decodes one block of the payload, without deciding who keeps it.
5152    ///
5153    /// [`Self::payload_block`] keeps it forever, which is what a point read wants and what a walk
5154    /// of the whole dictionary must not do. Both call this and they differ in nothing else.
5155    fn decode_block(&self, block: usize) -> Result<Vec<u8>> {
5156        let len = self.lengths[block];
5157        let mut stored = vec![
5158            0;
5159            usize::try_from(len).map_err(|_| invalid(
5160                "global dictionary block does not fit in memory"
5161            ))?
5162        ];
5163        read_at(&self.file, self.starts[block], &mut stored)?;
5164        if checksum(&stored) != self.hashes[block] {
5165            return Err(invalid("global dictionary payload checksum differs"));
5166        }
5167        let first = block * TEXT_PAYLOAD_VALUES;
5168        let last = (first + TEXT_PAYLOAD_VALUES).min(self.values);
5169        let want = self.end_within(last - 1)? as usize;
5170        let values = string::decode_flat(&stored)?;
5171        if values.len() != last - first {
5172            return Err(invalid("global dictionary block holds the wrong value count"));
5173        }
5174        let bytes = values.into_bytes();
5175        if bytes.len() != want {
5176            return Err(invalid("global dictionary block decodes to the wrong length"));
5177        }
5178        Ok(bytes)
5179    }
5180
5181    /// The block holding a value that a read hands over on loan, kept or decoded for the call.
5182    ///
5183    /// A block something already kept is read where it is. One nothing kept is kept the second
5184    /// time a loaned read decodes it while the column is holding less than [`Self::keep_budget`],
5185    /// and decoded into `decoded` and dropped with it otherwise, which is the policy
5186    /// [`TextSource::sweep`] explains. `scattered` is a read by code rather than in order, which
5187    /// stops dropping once it has dropped a column's worth of blocks, for the reason
5188    /// [`Self::visit_dropped`] gives.
5189    fn loaned_block<'a>(
5190        &'a self,
5191        block: usize,
5192        decoded: &'a mut Vec<u8>,
5193        scattered: bool,
5194    ) -> Result<&'a [u8]> {
5195        let kept = self.blocks.get(block).and_then(OnceLock::get);
5196        if let Some(Ok(kept)) = kept {
5197            return Ok(kept);
5198        }
5199        let again = kept.is_none()
5200            && self.swept.get(block).is_some_and(|swept| swept.swap(true, Atomic::Relaxed));
5201        let keep = again
5202            && (self.payload_kept.load(Atomic::Relaxed) < self.keep_budget
5203                || (scattered && self.visit_dropped.load(Atomic::Relaxed) >= self.blocks.len()));
5204        if keep {
5205            let kept = self
5206                .payload_block(block)?
5207                .ok_or_else(|| invalid("global dictionary block is past the payload"))?;
5208            self.payload_kept.fetch_add(kept.len(), Atomic::Relaxed);
5209            return Ok(kept);
5210        }
5211        *decoded = self.decode_block(block)?;
5212        if scattered && again {
5213            self.visit_dropped.fetch_add(1, Atomic::Relaxed);
5214        }
5215        Ok(decoded)
5216    }
5217
5218    /// How many single offset reads make [`Self::value_ends`] worth building.
5219    ///
5220    /// As many reads as the dictionary has values. Building the table costs about thirty
5221    /// instructions a value once the fresh pages it lands in are counted, and a read out of it saves
5222    /// about thirty five, so it repays itself after roughly one read per value. The reads so far are
5223    /// the only guess there is at the reads to come, and waiting until they match the size of the
5224    /// dictionary is betting that a column read that much will be read that much again.
5225    ///
5226    /// A sixteenth was the first answer, from counting the unpacking alone at three instructions a
5227    /// value. ClickBench 38 showed what that missed: it reads about twenty thousand titles a
5228    /// statement out of a dictionary of three hundred and fifty thousand, crossed a sixteenth in its
5229    /// second statement and was two percent slower for a table it did not read enough to repay. A
5230    /// scan asking for the length of every row crosses it part way through its first statement on
5231    /// ClickBench, where a string column has about two rows for every value, and a filter that keeps
5232    /// a few thousand rows never does. The floor is there
5233    /// because a short dictionary would otherwise build a table for a handful of reads.
5234    fn ends_worth_unpacking(&self) -> usize {
5235        self.values.max(TEXT_PAYLOAD_VALUES)
5236    }
5237
5238    /// The unpacked ends, if they are built or if this read is the one that makes them worth it.
5239    fn value_ends(&self) -> Option<&[u32]> {
5240        if let Some(built) = self.value_ends.get() {
5241            return built.as_deref();
5242        }
5243        if self.ends_asked.fetch_add(1, Atomic::Relaxed) < self.ends_worth_unpacking() {
5244            return None;
5245        }
5246        self.value_ends.get_or_init(|| self.unpack_ends()).as_deref()
5247    }
5248
5249    /// Every end of the column, a run at a time.
5250    ///
5251    /// `None` rather than an error on anything wrong, because this is a cache in front of a reader
5252    /// that answers the same question. A column whose offsets are short or whose ends do not fit in
5253    /// four bytes gets no table and the same error it would have got, from the read that wanted it.
5254    fn unpack_ends(&self) -> Option<Vec<u32>> {
5255        let mut ends = vec![0u32; self.values];
5256        for (run, into) in ends.chunks_mut(TEXT_OFFSET_RUN).enumerate() {
5257            let bytes = self.packed().get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)?;
5258            bitpack::unpack_tail_into(bytes, self.offset_bits, into, |bits| {
5259                u32::try_from(bits).unwrap_or(u32::MAX)
5260            })
5261            .ok()?;
5262        }
5263        // An end that did not fit was stored as the sentinel, and a real one cannot reach it because
5264        // a payload block is far smaller than four gigabytes. So the column keeps the packed reader.
5265        if ends.contains(&u32::MAX) { None } else { Some(ends) }
5266    }
5267
5268    /// The packed offsets, which is the index past its header.
5269    fn packed(&self) -> &[u8] {
5270        self.offsets.get(DICTIONARY_HEADER..).unwrap_or_default()
5271    }
5272
5273    /// Where the value at `index` ends inside its payload block.
5274    fn end_within(&self, index: usize) -> Result<u32> {
5275        if let Some(ends) = self.value_ends() {
5276            return ends
5277                .get(index)
5278                .copied()
5279                .ok_or_else(|| invalid("global dictionary offsets are short"));
5280        }
5281        self.packed_end(index)
5282    }
5283
5284    /// [`Self::end_within`] read out of the packed offsets, whether or not the table is built.
5285    fn packed_end(&self, index: usize) -> Result<u32> {
5286        let run = index / TEXT_OFFSET_RUN;
5287        let bytes = self
5288            .packed()
5289            .get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)
5290            .ok_or_else(|| invalid("global dictionary offsets are short"))?;
5291        let end = bitpack::tail_at(bytes, self.offset_bits, index % TEXT_OFFSET_RUN)
5292            .map_err(|_| invalid("global dictionary offsets are short"))?;
5293        u32::try_from(end).map_err(|_| invalid("global dictionary offset is past the payload"))
5294    }
5295
5296    /// Where every value in `first..last` ends inside its payload block, in one pass over the runs.
5297    ///
5298    /// [`Self::end_within`] answers for one value and pays for it twice over: it shifts a window to
5299    /// the bit the value starts at, and the copy that fills that window is a length the compiler does
5300    /// not know, so it is a call to `memcpy` rather than a load. A sweep asked for two of those per
5301    /// value, one for the end and one for the start that is the end before it, and on the ClickBench
5302    /// `URL` dictionary of eighteen million that was most of the half second a `LIKE` over it took.
5303    ///
5304    /// [`bitpack::unpack_tail_into`] walks the run instead, which makes the window a fixed width and
5305    /// so an unaligned load, and reads the bit position off a counter. A run is five hundred and
5306    /// twelve values and a block is two of them, so a block of a thousand and twenty four values
5307    /// costs two calls here and nothing per value.
5308    ///
5309    /// The answer is written straight into the result. A run that is wanted from its first value,
5310    /// which is every run but the one the sweep starts in, unpacks into its own window of the result
5311    /// and is never copied. Only a run joined part way through needs the scratch buffer, and there is
5312    /// at most one of those per sweep, so the buffer is allocated the first time one turns up.
5313    fn ends_within(&self, first: usize, last: usize) -> Result<Vec<u64>> {
5314        let mut ends = vec![0u64; last.saturating_sub(first)];
5315        let mut scratch = Vec::new();
5316        let mut at = first;
5317        while at < last {
5318            let run = at / TEXT_OFFSET_RUN;
5319            let stop = ((run + 1) * TEXT_OFFSET_RUN).min(last);
5320            let held = self.values.saturating_sub(run * TEXT_OFFSET_RUN).min(TEXT_OFFSET_RUN);
5321            let bytes = self
5322                .packed()
5323                .get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)
5324                .ok_or_else(|| invalid("global dictionary offsets are short"))?;
5325            let from = at % TEXT_OFFSET_RUN;
5326            let upto = stop - run * TEXT_OFFSET_RUN;
5327            if upto > held || bytes.len() < bitpack::tail_len(held, self.offset_bits) {
5328                return Err(invalid("global dictionary offsets are short"));
5329            }
5330            let into = &mut ends[at - first..stop - first];
5331            if from == 0 {
5332                bitpack::unpack_tail_into(bytes, self.offset_bits, into, |bits| bits)
5333                    .map_err(|_| invalid("global dictionary offsets are short"))?;
5334            } else {
5335                scratch.resize(held, 0);
5336                bitpack::unpack_tail_into(bytes, self.offset_bits, &mut scratch, |bits| bits)
5337                    .map_err(|_| invalid("global dictionary offsets are short"))?;
5338                into.copy_from_slice(&scratch[from..upto]);
5339            }
5340            at = stop;
5341        }
5342        Ok(ends)
5343    }
5344
5345    /// Where the value at `index` starts inside its payload block, which is where the value before
5346    /// it ended unless it is the first of the block.
5347    fn start_within(&self, index: usize) -> Result<u32> {
5348        if index.is_multiple_of(TEXT_PAYLOAD_VALUES) { Ok(0) } else { self.end_within(index - 1) }
5349    }
5350
5351    /// Where the value at `index` starts and ends inside its payload block.
5352    ///
5353    /// The two offsets sit next to each other in the same run unless the value opens one, and a run
5354    /// of seventeen bit offsets, which is what a block of a thousand strings needs, puts a pair of
5355    /// them inside one eight byte load. So the common case reads the packed bytes once rather than
5356    /// twice and does the bounds arithmetic once. This is asked once per string a text column hands
5357    /// out, and on ClickBench 27 the two reads together were a quarter of the query.
5358    fn span_within(&self, index: usize) -> Result<(u32, u32)> {
5359        if let Some(ends) = self.value_ends() {
5360            let end =
5361                *ends.get(index).ok_or_else(|| invalid("global dictionary offsets are short"))?;
5362            // The value before it in the same block, and zero where there is no value before it.
5363            // `index` is inside the table, so the one under it is too.
5364            let start = if index.is_multiple_of(TEXT_PAYLOAD_VALUES) { 0 } else { ends[index - 1] };
5365            if start > end {
5366                return Err(invalid("global dictionary value ends before it starts"));
5367            }
5368            return Ok((start, end));
5369        }
5370        self.packed_span(index)
5371    }
5372
5373    /// [`Self::span_within`] read out of the packed offsets, whether or not the table is built.
5374    fn packed_span(&self, index: usize) -> Result<(u32, u32)> {
5375        let within = index % TEXT_OFFSET_RUN;
5376        let (start, end) = if within == 0 {
5377            let start = if index.is_multiple_of(TEXT_PAYLOAD_VALUES) {
5378                0
5379            } else {
5380                self.packed_end(index - 1)?
5381            };
5382            (start, self.packed_end(index)?)
5383        } else {
5384            let run = index / TEXT_OFFSET_RUN;
5385            let bytes = self
5386                .packed()
5387                .get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)
5388                .ok_or_else(|| invalid("global dictionary offsets are short"))?;
5389            let (start, end) = bitpack::tail_pair(bytes, self.offset_bits, within)
5390                .map_err(|_| invalid("global dictionary offsets are short"))?;
5391            let ends = u32::try_from(end)
5392                .map_err(|_| invalid("global dictionary offset is past the payload"))?;
5393            let starts = u32::try_from(start)
5394                .map_err(|_| invalid("global dictionary offset is past the payload"))?;
5395            (starts, ends)
5396        };
5397        if start > end {
5398            return Err(invalid("global dictionary value ends before it starts"));
5399        }
5400        Ok((start, end))
5401    }
5402
5403    /// The block of the sorted order that holds `rank`, and where in it that rank sits.
5404    ///
5405    /// The block is read from the file and checked against the hash the index carries for it the
5406    /// first time anything asks, and kept after that, the same way a payload block is. A search
5407    /// makes about as many probes as the order has bits, so the whole search reads a handful of
5408    /// these and never the rest.
5409    fn rank_parts(&self, rank: usize) -> Result<(&[u8], usize)> {
5410        let slot = self
5411            .rank_blocks
5412            .get(rank / TEXT_RANK_BLOCK)
5413            .ok_or_else(|| invalid("global dictionary rank is past the order"))?;
5414        let block = slot
5415            .get_or_init(|| {
5416                let mut bytes = Vec::new();
5417                self.read_rank_block(rank / TEXT_RANK_BLOCK, &mut bytes)?;
5418                Ok(bytes)
5419            })
5420            .as_ref()
5421            .map_err(Clone::clone)?;
5422        Ok((block.as_slice(), rank % TEXT_RANK_BLOCK))
5423    }
5424
5425    /// Reads block `which` of the sorted order into `bytes`, checked against the hash the index
5426    /// carries for it.
5427    fn read_rank_block(&self, which: usize, bytes: &mut Vec<u8>) -> Result<()> {
5428        let start = if which == 0 { 0 } else { self.rank_ends[which - 1] };
5429        let end = self.rank_ends[which];
5430        bytes.clear();
5431        bytes.resize((end - start) as usize, 0);
5432        read_at(&self.file, self.rank_at + start, bytes)?;
5433        let expected = self
5434            .rank_hashes
5435            .get(which)
5436            .ok_or_else(|| invalid("global dictionary rank block has no checksum"))?;
5437        if checksum(bytes) != *expected {
5438            return Err(invalid("global dictionary rank checksum differs"));
5439        }
5440        Ok(())
5441    }
5442
5443    /// The first eight bytes of the value at `rank`, as the integer a comparison reads.
5444    fn head_at(&self, rank: usize) -> Result<u64> {
5445        let (block, within) = self.rank_parts(rank)?;
5446        let (base, width, packed) = rank_heads(block)?;
5447        let above = bitpack::tail_at(packed, width, within)
5448            .map_err(|_| invalid("global dictionary rank block is short of heads"))?;
5449        Ok(base.wrapping_add(above))
5450    }
5451
5452    /// The packed codes of one rank block, which follow the heads on the next byte boundary.
5453    fn rank_codes<'block>(&self, block: &'block [u8], count: usize) -> Result<&'block [u8]> {
5454        let (_, width, packed) = rank_heads(block)?;
5455        packed
5456            .get(bitpack::tail_len(count, width)..)
5457            .ok_or_else(|| invalid("global dictionary rank block is short of codes"))
5458    }
5459
5460    /// How many entries the block holding `rank` has, which is a full block except at the end.
5461    fn rank_block_len(&self, rank: usize) -> usize {
5462        let first = rank / TEXT_RANK_BLOCK * TEXT_RANK_BLOCK;
5463        TEXT_RANK_BLOCK.min(self.ranks - first)
5464    }
5465}
5466
5467/// The base, the width and the packed bytes of one rank block's heads.
5468fn rank_heads(block: &[u8]) -> Result<(u64, usize, &[u8])> {
5469    let header = block
5470        .get(..RANK_BLOCK_HEADER)
5471        .ok_or_else(|| invalid("global dictionary rank block is short"))?;
5472    let base = u64::from_le_bytes(header[..8].try_into().expect("eight bytes"));
5473    let width = header[8] as usize;
5474    if width > 64 {
5475        return Err(invalid("global dictionary rank block packs heads past a word"));
5476    }
5477    Ok((base, width, &block[RANK_BLOCK_HEADER..]))
5478}
5479
5480/// Bits one offset of a dictionary takes, which is what its widest payload block spans.
5481///
5482/// One width for the whole column rather than one a block. A block is 1,024 values of the same
5483/// column, so the blocks of a column are within a factor of two of each other on every ClickBench
5484/// string column, and a width a block would save a fraction of a bit and cost a byte a block plus
5485/// the arithmetic that finds where a block starts.
5486fn offset_width(ends: &[u32]) -> usize {
5487    // The ends are already relative to the block the value is in, so the last end of a block is that
5488    // block's total and the largest end anywhere is the widest block. There is no subtraction left
5489    // to do and no need to walk the blocks to find where one starts.
5490    let span = ends.iter().copied().max().unwrap_or(0);
5491    (u32::BITS - span.leading_zeros()) as usize
5492}
5493
5494/// How many bytes `values` offsets take at `bits`, which is what the reader has to know before it
5495/// has read any of them.
5496fn offset_bytes(values: usize, bits: usize) -> usize {
5497    let full = values / TEXT_OFFSET_RUN;
5498    let rest = values % TEXT_OFFSET_RUN;
5499    full * TEXT_OFFSET_RUN / 8 * bits + bitpack::tail_len(rest, bits)
5500}
5501
5502/// The end of every value within its payload block, packed a run at a time.
5503/// A run never straddles a block, because [`TEXT_OFFSET_RUN`] divides [`TEXT_PAYLOAD_VALUES`], which
5504/// is what lets this be a walk of the ends rather than arithmetic against a per block base.
5505fn encode_offsets(ends: &[u32], bits: usize, out: &mut Vec<u8>) -> Result<()> {
5506    let mut run = Vec::with_capacity(TEXT_OFFSET_RUN);
5507    for chunk in ends.chunks(TEXT_OFFSET_RUN) {
5508        run.clear();
5509        run.extend(chunk.iter().map(|&end| u64::from(end)));
5510        bitpack::pack_tail(&run, bits, out)
5511            .map_err(|_| invalid("global dictionary offsets do not pack"))?;
5512    }
5513    Ok(())
5514}
5515
5516/// How many bits a code of a dictionary of `values` entries takes.
5517fn code_width(values: usize) -> usize {
5518    match u64::try_from(values).unwrap_or(u64::MAX) {
5519        0 | 1 => 0,
5520        last => (u64::BITS - (last - 1).leading_zeros()) as usize,
5521    }
5522}
5523
5524impl TextSource for NativeText {
5525    fn len(&self) -> usize {
5526        self.values
5527    }
5528
5529    fn might_contain(&self, first: usize, literal: &[u8]) -> Result<bool> {
5530        let Some(grams) = &self.grams else { return Ok(true) };
5531        if literal.len() < 4 || first >= self.values {
5532            return Ok(true);
5533        }
5534        let verdict = grams.verdicts(&self.file, literal)?;
5535        Ok(verdict.get(first / TEXT_PAYLOAD_VALUES).copied().unwrap_or(true))
5536    }
5537
5538    fn bytes_at(&self, index: usize) -> Result<Option<&[u8]>> {
5539        if index >= self.values {
5540            return Ok(None);
5541        }
5542        let (start, end) = self.span_within(index)?;
5543        if start == end {
5544            return Ok(Some(&[]));
5545        }
5546        // A block holds a fixed number of values rather than a fixed number of bytes, so the value
5547        // is in one block and the offsets already say where in it.
5548        let block = index / TEXT_PAYLOAD_VALUES;
5549        let Some(bytes) = self.payload_block(block)? else { return Ok(None) };
5550        Ok(bytes.get(start as usize..end as usize))
5551    }
5552
5553    fn bytes_len_at(&self, index: usize) -> Result<Option<usize>> {
5554        if index >= self.values {
5555            return Ok(None);
5556        }
5557        let (start, end) = self.span_within(index)?;
5558        Ok(Some((end - start) as usize))
5559    }
5560
5561    /// Every length out of the unpacked ends in one loop, which is the point of having them.
5562    ///
5563    /// The whole run of positions counts towards [`Self::ends_worth_unpacking`] at once, because a
5564    /// caller asking for a vector of lengths has said how many it wants, and a vector of them is
5565    /// usually enough on its own. Until the table is worth building this is the row at a time read,
5566    /// the same as the default.
5567    fn bytes_lens_at(&self, indices: &[u32], into: &mut Vec<i64>) -> Result<()> {
5568        into.reserve(indices.len());
5569        // Once the table is built the count has nothing left to decide, and every thread of a scan
5570        // adding to the one counter moves its cache line from core to core on every chunk.
5571        if let Some(Some(lens)) = self.value_lens.get() {
5572            lens.extend_at(indices, into);
5573            return Ok(());
5574        }
5575        // The lengths are built out of ends unpacked for the purpose and dropped, not out of the
5576        // table of ends. That table is four bytes a value and the lengths are two, and a column
5577        // that is only asked for lengths would keep both. On q29 that was 11 MB of `Referer` ends
5578        // held for `STRLEN` alone.
5579        let asked = self.ends_asked.fetch_add(indices.len(), Atomic::Relaxed) + indices.len();
5580        let lens = match self.value_ends.get() {
5581            Some(Some(ends)) => self.value_lens.get_or_init(|| lengths_of(ends)).as_ref(),
5582            _ if asked >= self.ends_worth_unpacking() => self
5583                .value_lens
5584                .get_or_init(|| self.unpack_ends().and_then(|ends| lengths_of(&ends)))
5585                .as_ref(),
5586            _ => None,
5587        };
5588        if let Some(lens) = lens {
5589            lens.extend_at(indices, into);
5590            return Ok(());
5591        }
5592        for &index in indices {
5593            let index = index as usize;
5594            // Past the end is no value and so no length, which is what a row at a time read says.
5595            if index >= self.values {
5596                into.push(0);
5597                continue;
5598            }
5599            let (start, end) = self.packed_span(index)?;
5600            into.push(i64::from(end - start));
5601        }
5602        Ok(())
5603    }
5604
5605    /// Every length in characters out of the counts kept a block at a time, which is what keeps a
5606    /// scan of `length` from holding the column decoded. See [`NativeText::char_lens`].
5607    fn chars_lens_at(&self, indices: &[u32], into: &mut Vec<i64>) -> Result<()> {
5608        into.reserve(indices.len());
5609        for &index in indices {
5610            let index = index as usize;
5611            // Past the end is no value and so no length, which is what a row at a time read says.
5612            if index >= self.values {
5613                into.push(0);
5614                continue;
5615            }
5616            let lens = self.block_chars(index / TEXT_PAYLOAD_VALUES)?;
5617            let len = lens
5618                .get(index % TEXT_PAYLOAD_VALUES)
5619                .ok_or_else(|| invalid("global dictionary block holds the wrong value count"))?;
5620            into.push(i64::from(*len));
5621        }
5622        Ok(())
5623    }
5624
5625    /// The rest of the block holding `first`, decoded into a buffer that may die with the call.
5626    ///
5627    /// A block is the unit this format decodes, so a walk that wants every value is going to decode
5628    /// every block whatever it does. The question is whether it keeps them, and both answers are
5629    /// wrong on their own. [`Self::payload_block`] keeps every block it is asked for, so a reader
5630    /// that walked the whole dictionary through `bytes_at` ended up holding the whole dictionary
5631    /// decoded, 4.2 GB on ClickBench `URL`. Keeping none of them makes the next statement asking
5632    /// the same question decode all of it again, which on the same column at a million rows is a
5633    /// `LIKE` going from 2.7 ms to 16.2 ms.
5634    ///
5635    /// So a sweep keeps what it decodes for the second time while the column is under
5636    /// [`TEXT_KEEP_BUDGET`] and drops it after that. A block already in hand is used where it is there and costs nothing either way.
5637    fn sweep(
5638        &self,
5639        first: usize,
5640        limit: usize,
5641        body: &mut dyn FnMut(usize, &[u8]) -> Result<()>,
5642    ) -> Result<usize> {
5643        let limit = limit.min(self.values);
5644        if first >= limit {
5645            return Ok(first);
5646        }
5647        let block = first / TEXT_PAYLOAD_VALUES;
5648        let last = ((block + 1) * TEXT_PAYLOAD_VALUES).min(limit);
5649        let mut decoded = Vec::new();
5650        let bytes = self.loaned_block(block, &mut decoded, false)?;
5651        let ends = self.ends_within(first, last)?;
5652        if ends.len() != last - first {
5653            return Err(invalid("global dictionary offsets are short"));
5654        }
5655        let mut start = u64::from(self.start_within(first)?);
5656        // row at a time: the caller is handed one value after another, and what it does with one is
5657        // its own business, so there is no shape here for anything but a walk.
5658        for (index, &end) in (first..last).zip(&ends) {
5659            let value = usize::try_from(start)
5660                .ok()
5661                .zip(usize::try_from(end).ok())
5662                .and_then(|(from, to)| bytes.get(from..to))
5663                .ok_or_else(|| invalid("global dictionary value is past its block"))?;
5664            body(index, value)?;
5665            start = end;
5666        }
5667        Ok(last)
5668    }
5669
5670    /// The values at `indices` a block at a time, each block read once for the call.
5671    ///
5672    /// The positions are put in code order first, because the codes of a vector are in row order
5673    /// and land all over the dictionary, and read in that order each block a vector touches would
5674    /// be looked up once for every row in it. Whether a block is kept is
5675    /// [`NativeText::loaned_block`]'s decision, which keeps at most the budget of this column
5676    /// until the reads have shown they come back to the same blocks too often for dropping them to
5677    /// be cheap.
5678    fn visit_at(
5679        &self,
5680        indices: &[u32],
5681        body: &mut dyn FnMut(usize, &[u8]) -> Result<()>,
5682    ) -> Result<()> {
5683        let mut order = (0..indices.len()).collect::<Vec<_>>();
5684        order.sort_unstable_by_key(|&at| indices[at]);
5685        let block_of = |at: usize| {
5686            let index = indices[at] as usize;
5687            (index < self.values).then_some(index / TEXT_PAYLOAD_VALUES)
5688        };
5689        let mut decoded = Vec::new();
5690        let mut run = 0;
5691        while run < order.len() {
5692            let Some(block) = block_of(order[run]) else {
5693                // Past the end is no value, and every position after this one is past it too.
5694                for &at in &order[run..] {
5695                    body(at, &[])?;
5696                }
5697                break;
5698            };
5699            let upto = run + order[run..].partition_point(|&at| block_of(at) == Some(block));
5700            let bytes = self.loaned_block(block, &mut decoded, true)?;
5701            for &at in &order[run..upto] {
5702                let (start, end) = self.span_within(indices[at] as usize)?;
5703                let value = bytes
5704                    .get(start as usize..end as usize)
5705                    .ok_or_else(|| invalid("global dictionary value is past its block"))?;
5706                body(at, value)?;
5707            }
5708            run = upto;
5709        }
5710        Ok(())
5711    }
5712
5713    /// Each block the indices land in, decoded once and dropped, or read where it is already kept.
5714    ///
5715    /// Never kept, unlike [`Self::sweep`] under its budget, because a scattered read is a one off:
5716    /// a synopsis turned into values is turned once and remembered by the reader as values, a few
5717    /// kilobytes, where the blocks it went through are megabytes nobody asks for again.
5718    fn visit(
5719        &self,
5720        indices: &[usize],
5721        body: &mut dyn FnMut(usize, &[u8]) -> Result<()>,
5722    ) -> Result<()> {
5723        let mut at = 0;
5724        while at < indices.len() {
5725            let block = indices[at] / TEXT_PAYLOAD_VALUES;
5726            let upto =
5727                at + indices[at..].partition_point(|&index| index / TEXT_PAYLOAD_VALUES == block);
5728            let wanted = &indices[at..upto];
5729            if wanted.iter().any(|&index| index >= self.values) {
5730                return Err(invalid("a visited value is past the global dictionary"));
5731            }
5732            let decoded;
5733            let bytes: &[u8] = match self.blocks.get(block).and_then(OnceLock::get) {
5734                Some(Ok(kept)) => kept,
5735                _ => {
5736                    decoded = self.decode_block(block)?;
5737                    &decoded
5738                }
5739            };
5740            for (offset, &index) in wanted.iter().enumerate() {
5741                let (start, end) = self.span_within(index)?;
5742                let value = bytes
5743                    .get(start as usize..end as usize)
5744                    .ok_or_else(|| invalid("global dictionary value is past its block"))?;
5745                body(at + offset, value)?;
5746            }
5747            at = upto;
5748        }
5749        Ok(())
5750    }
5751
5752    fn ranks(&self) -> Option<usize> {
5753        (self.ranks > 0).then_some(self.ranks)
5754    }
5755
5756    /// The boundary for `wanted`, out of [`Self::searched`] where it is there and put there where
5757    /// it is not.
5758    ///
5759    /// The lock is held over the search rather than dropped and taken again, so that two threads
5760    /// asking for the same value at the same time do the work once between them. That is the shape
5761    /// the scan actually arrives in: sixteen instances of a top N, all reading the same column, all
5762    /// improving their bound over the same early chunks.
5763    fn below(&self, ranks: usize, wanted: &[u8]) -> Result<(usize, bool)> {
5764        let mut memo = self.searched.lock().map_err(|_| invalid("a poisoned dictionary search"))?;
5765        if let Some(&answer) = memo.get(wanted) {
5766            return Ok(answer);
5767        }
5768        let answer = search_below(self, ranks, wanted)?;
5769        if memo.len() >= TEXT_SEARCH_MEMO {
5770            memo.clear();
5771        }
5772        memo.insert(wanted.to_vec(), answer);
5773        Ok(answer)
5774    }
5775
5776    fn compare_rank(&self, rank: usize, wanted: &[u8]) -> Result<Ordering> {
5777        // The head settles the probe unless the two values start with the same eight bytes, and
5778        // only then is a value read. On a column of URLs that is the difference between a search
5779        // that touches one block of the payload and a search that touches nineteen of them.
5780        let settled = self.head_at(rank)?.cmp(&head(wanted));
5781        if settled != Ordering::Equal {
5782            return Ok(settled);
5783        }
5784        let code = self.code_at_rank(rank)?;
5785        let bytes = self
5786            .bytes_at(code as usize)?
5787            .ok_or_else(|| invalid("global dictionary order names a code it does not have"))?;
5788        Ok(bytes.cmp(wanted))
5789    }
5790
5791    fn code_at_rank(&self, rank: usize) -> Result<u32> {
5792        let (block, within) = self.rank_parts(rank)?;
5793        let codes = self.rank_codes(block, self.rank_block_len(rank))?;
5794        let code = bitpack::tail_at(codes, self.code_bits, within)
5795            .map_err(|_| invalid("global dictionary rank block is short of codes"))?;
5796        let code = u32::try_from(code)
5797            .map_err(|_| invalid("global dictionary order names a code it does not have"))?;
5798        if code as usize >= self.len() {
5799            return Err(invalid("global dictionary order names a code it does not have"));
5800        }
5801        Ok(code)
5802    }
5803
5804    fn code_ranks(&self) -> Option<&[u32]> {
5805        // The order is a permutation of the positions, so inverting it needs every position to be
5806        // named exactly once. Anything else and the slice would have holes, and a caller indexing
5807        // it by a code would read a rank that belongs to nothing.
5808        if self.ranks == 0 || self.ranks != self.len() {
5809            return None;
5810        }
5811        self.code_ranks
5812            .get_or_init(|| {
5813                let mut ranks = vec![u32::MAX; self.ranks];
5814                // A block at a time rather than a rank at a time, because reading it per rank pays
5815                // for the bounds check, the division and the lock on every one of them.
5816                //
5817                // A block nothing has read yet is read into one buffer that is reused, rather than
5818                // through `rank_parts`, which would keep every block of the order once this is
5819                // done with it. The inverse is all anything wants after this, and on the `Referer`
5820                // column of the ClickBench file the blocks are tens of megabytes held for nothing.
5821                let mut scratch = Vec::new();
5822                let mut codes = vec![0u64; TEXT_RANK_BLOCK];
5823                for first in (0..self.ranks).step_by(TEXT_RANK_BLOCK) {
5824                    let which = first / TEXT_RANK_BLOCK;
5825                    let block = match self.rank_blocks.get(which)?.get() {
5826                        Some(kept) => kept.as_ref().ok()?.as_slice(),
5827                        None => {
5828                            self.read_rank_block(which, &mut scratch).ok()?;
5829                            scratch.as_slice()
5830                        }
5831                    };
5832                    let count = self.rank_block_len(first);
5833                    let packed = self.rank_codes(block, count).ok()?;
5834                    let codes = codes.get_mut(..count)?;
5835                    bitpack::unpack_tail_into(packed, self.code_bits, codes, |bits| bits).ok()?;
5836                    for (within, &code) in codes.iter().enumerate() {
5837                        let code = usize::try_from(code).ok()?;
5838                        *ranks.get_mut(code)? = u32::try_from(first + within).ok()?;
5839                    }
5840                }
5841                if ranks.contains(&u32::MAX) {
5842                    return None;
5843                }
5844                Some(ranks)
5845            })
5846            .as_deref()
5847    }
5848
5849    fn footprint(&self) -> usize {
5850        self.offsets.capacity()
5851            + self
5852                .value_ends
5853                .get()
5854                .and_then(Option::as_ref)
5855                .map_or(0, |ends| ends.capacity() * size_of::<u32>())
5856            + self.value_lens.get().and_then(Option::as_ref).map_or(0, Lengths::footprint)
5857            + self
5858                .code_ranks
5859                .get()
5860                .and_then(Option::as_ref)
5861                .map_or(0, |ranks| ranks.capacity() * size_of::<u32>())
5862            + self.rank_hashes.capacity() * size_of::<u64>()
5863            + self.rank_ends.capacity() * size_of::<u64>()
5864            + self.rank_blocks.capacity() * size_of::<OnceLock<Result<Vec<u8>>>>()
5865            + self
5866                .rank_blocks
5867                .iter()
5868                .filter_map(OnceLock::get)
5869                .filter_map(|result| result.as_ref().ok())
5870                .map(Vec::capacity)
5871                .sum::<usize>()
5872            + self.blocks.capacity() * size_of::<OnceLock<Result<Vec<u8>>>>()
5873            + self.char_lens.capacity() * size_of::<OnceLock<Box<[u32]>>>()
5874            + self
5875                .char_lens
5876                .iter()
5877                .filter_map(OnceLock::get)
5878                .map(|lens| lens.len() * size_of::<u32>())
5879                .sum::<usize>()
5880            + self.hashes.capacity() * size_of::<u64>()
5881            + self.starts.capacity() * size_of::<u64>()
5882            + self.lengths.capacity() * size_of::<u64>()
5883            + self.grams.as_ref().map_or(0, NativeGrams::footprint)
5884            + self
5885                .blocks
5886                .iter()
5887                .filter_map(OnceLock::get)
5888                .filter_map(|result| result.as_ref().ok())
5889                .map(Vec::capacity)
5890                .sum::<usize>()
5891    }
5892}
5893
5894/// Every table wide part number in order, with the stripe it belongs to.
5895fn places(table: &Table) -> Result<Vec<Place>> {
5896    let mut places = Vec::with_capacity(table.stripes.len().saturating_mul(STRIPE_PARTS));
5897    for (at, stripe) in table.stripes.iter().enumerate() {
5898        let index = u32::try_from(at).map_err(|_| invalid("too many stripes"))?;
5899        for (part, &rows) in stripe.parts.iter().enumerate() {
5900            places.push(Place {
5901                stripe: index,
5902                part: u32::try_from(part).map_err(|_| invalid("too many parts in a stripe"))?,
5903                rows,
5904            });
5905        }
5906    }
5907    Ok(places)
5908}
5909
5910/// Reads one column's section of a stripe's index page.
5911///
5912/// The section carries its own checksum, so a reader that wants one column out of a hundred and
5913/// five preads a few hundred bytes and still knows that what it got is what was written.
5914fn read_index<F: Positional + ?Sized>(
5915    file: &F,
5916    stripe: &Stripe,
5917    column: usize,
5918) -> Result<Vec<PartSpan>> {
5919    let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
5920    read_index_span(file, stripe.index, *page, stripe.parts.len(), column)
5921}
5922
5923fn read_index_span<F: Positional + ?Sized>(
5924    file: &F,
5925    index: Span,
5926    page: Span,
5927    parts: usize,
5928    column: usize,
5929) -> Result<Vec<PartSpan>> {
5930    let section = index_section(parts)?;
5931    let at = column.checked_mul(section).ok_or_else(|| invalid("index page offset overflow"))?;
5932    let end = at.checked_add(section).ok_or_else(|| invalid("index page offset overflow"))?;
5933    if end > index.length as usize {
5934        return Err(invalid("index page is shorter than its columns"));
5935    }
5936    let mut bytes = vec![0; section];
5937    let offset =
5938        index.offset.checked_add(at as u64).ok_or_else(|| invalid("index page offset overflow"))?;
5939    read_at(file, offset, &mut bytes)?;
5940    let entries = section - size_of::<u64>();
5941    let stored = u64::from_le_bytes(bytes[entries..].try_into().expect("eight bytes"));
5942    if checksum(&bytes[..entries]) != stored {
5943        // With where it was read from, because the two ways this fires look identical from the
5944        // message alone: a file somebody damaged, and a file we wrote to the wrong offset.
5945        return Err(invalid(&format!(
5946            "index page section checksum differs, column {column} of {parts} parts at {offset}, \
5947             wanted {stored:016x} and got {:016x}",
5948            checksum(&bytes[..entries]),
5949        )));
5950    }
5951    let mut spans = Vec::with_capacity(parts);
5952    let mut start = 0_usize;
5953    for part in 0..parts {
5954        let at = part * INDEX_ENTRY;
5955        let length = u32::from_le_bytes(bytes[at..at + 4].try_into().expect("four bytes")) as usize;
5956        let hash = u64::from_le_bytes(bytes[at + 4..at + 12].try_into().expect("eight bytes"));
5957        spans.push(PartSpan { start, length, hash });
5958        start = start.checked_add(length).ok_or_else(|| invalid("column page length overflow"))?;
5959    }
5960    if start != page.length as usize {
5961        return Err(invalid("column page length differs from its index"));
5962    }
5963    Ok(spans)
5964}
5965
5966/// One part's bytes out of a whole column page.
5967fn part_bytes(page: &[u8], span: PartSpan) -> Result<&[u8]> {
5968    let end = span.start.checked_add(span.length).ok_or_else(|| invalid("part range overflow"))?;
5969    page.get(span.start..end).ok_or_else(|| invalid("part exceeds its column page"))
5970}
5971
5972/// Marks part `part` of a stripe of `parts` parts read, and says whether it had been read before and
5973/// whether every part of the stripe had been before this one was asked for again.
5974fn touch(bits: &mut Vec<u64>, part: usize, parts: usize) -> (bool, bool) {
5975    if bits.is_empty() {
5976        bits.resize(parts.div_ceil(64).max(1), 0);
5977    }
5978    let (word, bit) = (part / 64, 1_u64 << (part % 64));
5979    let Some(held) = bits.get_mut(word) else { return (false, false) };
5980    let again = *held & bit != 0;
5981    *held |= bit;
5982    let through = bits.iter().map(|word| word.count_ones() as usize).sum::<usize>() >= parts;
5983    (again, again && through)
5984}
5985
5986/// Puts one stripe of one column in the cache, and hands back the page for the pool to count when
5987/// it is a page the column did not already hold.
5988///
5989/// The index goes in its own slot and stays. Only the page is under the budget, and the pool is
5990/// what enforces it, once the caller has let go of the column's lock.
5991fn remember(cached: &mut Cached, held: &CachedColumn) -> Option<(usize, Arc<AtomicBool>)> {
5992    if let Some(slot) = cached.index.get_mut(held.stripe)
5993        && slot.is_none()
5994    {
5995        *slot = Some(Arc::clone(&held.index));
5996    }
5997    let page = held.page.clone()?;
5998    let slot = cached.pages.get_mut(held.stripe)?;
5999    if slot.is_some() {
6000        return None;
6001    }
6002    let bytes = page.bytes().len();
6003    // Set, so that the page a worker has just paid to read is not the one the pass it pays for
6004    // lets go of before the worker has read a part out of it.
6005    let used = Arc::new(AtomicBool::new(true));
6006    *slot = Some(Resident { page, used: Arc::clone(&used) });
6007    Some((bytes, used))
6008}
6009
6010/// Every table a native file holds, without the directory of any of them.
6011///
6012/// This is what opening a database reads. It is the small level of the directory, so the cost is
6013/// proportional to how many tables there are rather than to how much data they hold, and a session
6014/// that touches two tables of eight decodes two table directories.
6015///
6016/// The file handle is shared with every reader this hands out. Eight tables in one file is one open
6017/// file descriptor, not eight, which is the other thing one file buys over a file per table.
6018#[derive(Debug, Clone)]
6019pub struct Catalog {
6020    file: Arc<File>,
6021    size: u64,
6022    /// The file as it was at open, mapped, so a reader takes a page's bytes where the page cache
6023    /// holds them instead of copying them out. `None` where the file cannot be mapped.
6024    map: Option<Arc<Mapped>>,
6025    entries: Arc<Vec<Entry>>,
6026    /// The views the file holds, whole, since a view has no second level to read later.
6027    views: Arc<Vec<ViewEntry>>,
6028    /// How much of the log the file holds.
6029    anchor: Option<Arc<LogAnchor>>,
6030    opening: Opening,
6031    /// Where every reader this hands out counts its pages.
6032    pool: PagePool,
6033}
6034
6035/// Signed integer sums and non-null counts for selected columns, plus total table rows.
6036#[derive(Debug, Clone, PartialEq, Eq)]
6037pub struct CertifiedSums {
6038    pub columns: Vec<(i128, u64)>,
6039    pub rows: u64,
6040}
6041
6042/// Exact ends of an integer or date column, including a certified all-null column.
6043#[derive(Debug, Clone, Copy, PartialEq, Eq)]
6044pub enum IntegerExtremes {
6045    Null,
6046    Values { low: i128, high: i128 },
6047}
6048
6049/// A complete numeric value-to-row-count synopsis; `None` represents SQL NULL.
6050pub type NumericFrequencies = Vec<(Option<i128>, u64)>;
6051
6052impl Catalog {
6053    /// Reads the highest valid catalog slot and nothing under it.
6054    ///
6055    /// The readers it hands out keep pages in a pool of their own with no budget, so each column
6056    /// holds its floor of four stripes and no more. A database opens with [`Catalog::open_in`].
6057    ///
6058    /// # Errors
6059    ///
6060    /// If the file has no valid committed catalog or a catalog pointer is out of bounds.
6061    pub fn open(path: impl AsRef<Path>) -> Result<Self> {
6062        Self::open_in(path, &PagePool::default())
6063    }
6064
6065    /// The same, with every reader it hands out keeping its pages in `pool`.
6066    ///
6067    /// # Errors
6068    ///
6069    /// If the file has no valid committed catalog or a catalog pointer is out of bounds.
6070    pub fn open_in(path: impl AsRef<Path>, pool: &PagePool) -> Result<Self> {
6071        let path = path.as_ref();
6072        let (file, size, _, bytes, opening) = slot_bytes(path)?;
6073        let (entries, views, card, anchor) = decode_catalog(&bytes, size)?;
6074        remember_card(path, card.as_ref());
6075        let map = Mapped::open(&file, size).map(Arc::new);
6076        Ok(Self {
6077            anchor: anchor.map(Arc::new),
6078            file: Arc::new(file),
6079            size,
6080            map,
6081            entries: Arc::new(entries),
6082            views: Arc::new(views),
6083            opening,
6084            pool: pool.clone(),
6085        })
6086    }
6087
6088    /// The tables in the file, in the order they were written.
6089    pub fn names(&self) -> impl ExactSizeIterator<Item = &str> {
6090        self.entries.iter().map(|entry| entry.name.as_str())
6091    }
6092
6093    /// The same tables with how many rows each of them holds.
6094    ///
6095    /// The names alone answer which tables the file has, which is what a checkpoint needs to know.
6096    /// A load asks a second question: whether a table already in the file is really in the way of
6097    /// the one it wants to write. A table with no rows is not, because it has no pages the next
6098    /// generation would have to carry, so the count has to come out of the catalog beside the name.
6099    pub fn rows(&self) -> impl ExactSizeIterator<Item = (&str, usize)> {
6100        self.entries.iter().map(|entry| (entry.name.as_str(), entry.rows))
6101    }
6102
6103    /// The views in the file, in the order they were written.
6104    ///
6105    /// Whole, unlike [`Catalog::names`], which hands back names and makes the caller ask for a table
6106    /// by one. A view is a few strings and a column list and it was all read at open, so there is
6107    /// nothing left to go and fetch and no reason to make the caller ask twice.
6108    pub fn views(&self) -> impl ExactSizeIterator<Item = &ViewEntry> {
6109        self.views.iter()
6110    }
6111
6112    /// How much of the log the file holds, or `None` for a file no log was ever anchored in.
6113    #[must_use]
6114    pub fn log_anchor(&self) -> Option<&LogAnchor> {
6115        self.anchor.as_deref()
6116    }
6117
6118    /// How many tables the file holds.
6119    #[must_use]
6120    pub fn len(&self) -> usize {
6121        self.entries.len()
6122    }
6123
6124    /// Whether the file holds no table at all, which is what [`Writer::empty`] writes and what a
6125    /// database somebody dropped the last table out of comes back as.
6126    #[must_use]
6127    pub fn is_empty(&self) -> bool {
6128        self.entries.is_empty()
6129    }
6130
6131    /// Opens one table by name, decoding its directory now.
6132    ///
6133    /// # Errors
6134    ///
6135    /// If there is no table by that name, or its directory is torn or points outside the file.
6136    pub fn table(&self, name: &str) -> Result<Reader> {
6137        let entry = self
6138            .entries
6139            .iter()
6140            .find(|entry| entry.name == name)
6141            .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
6142        // Checked and then decoded a window at a time, so that the directory's own bytes are never
6143        // all in memory beside the table they decode into. It is read twice, and the second read
6144        // comes out of the page cache the first one filled.
6145        let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
6146        if file_checksum(&self.file, offset, length)? != entry.directory.hash {
6147            return Err(invalid(&format!("the directory of table {name} does not checksum")));
6148        }
6149        let mut opening = self.opening;
6150        opening.reads += 1;
6151        opening.bytes += u64::from(entry.directory.length);
6152        Reader::build(
6153            Arc::clone(&self.file),
6154            self.map.clone(),
6155            self.size,
6156            read_directory(Cursor::over(&self.file, offset, length), self.size, Some(offset))?,
6157            u64::from(entry.directory.length),
6158            opening,
6159            self.pool.clone(),
6160        )
6161    }
6162
6163    /// Counts one signed integer column from its encoded parts without building metadata for
6164    /// unrelated columns. The counts are computed from row encodings when this is called.
6165    /// Nullable and non-cascade parts use the ordinary decoder for that part.
6166    ///
6167    /// # Errors
6168    ///
6169    /// If the directory, selected page index, checksum, or encoded integer is invalid.
6170    pub fn integer_tally(&self, name: &str, column: usize) -> Result<Option<Vec<(i64, u64)>>> {
6171        let mut counts = BTreeMap::<i64, u64>::new();
6172        let Some(()) = self.integer_fold(name, column, |value, count| {
6173            let held = counts.entry(value).or_default();
6174            *held = held.checked_add(count).ok_or_else(|| invalid("integer count overflow"))?;
6175            Ok(())
6176        })?
6177        else {
6178            return Ok(None);
6179        };
6180        Ok(Some(counts.into_iter().collect()))
6181    }
6182
6183    /// Visits a signed integer column's row values without building per-part or table-wide count
6184    /// maps. The caller combines the emitted counts for its query at runtime.
6185    ///
6186    /// # Errors
6187    ///
6188    /// If the selected file data is invalid or the callback rejects a count.
6189    pub fn integer_fold(
6190        &self,
6191        name: &str,
6192        column: usize,
6193        mut emit: impl FnMut(i64, u64) -> Result<()>,
6194    ) -> Result<Option<()>> {
6195        let entry = self
6196            .entries
6197            .iter()
6198            .find(|entry| entry.name == name)
6199            .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
6200        let field =
6201            entry.fields.get(column).ok_or_else(|| invalid("integer column index out of range"))?;
6202        if !signed_integer(&field.ty) {
6203            return Ok(None);
6204        }
6205        let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
6206        if file_checksum(&self.file, offset, length)? != entry.directory.hash {
6207            return Err(invalid(&format!("the directory of table {name} does not checksum")));
6208        }
6209        quick_integer_fold(
6210            &self.file,
6211            Cursor::over(&self.file, offset, length),
6212            entry,
6213            self.size,
6214            column,
6215            &mut emit,
6216        )?;
6217        Ok(Some(()))
6218    }
6219
6220    /// Counts non-null, nonzero values from generic column frequencies when complete. For an
6221    /// older file or a partial catalog synopsis, reads the validated native directory without
6222    /// building a reader for every stripe. Returns `None` when the bounded frequency synopsis
6223    /// cannot prove the count, so callers can use the ordinary query path.
6224    pub fn nonzero_count(&self, name: &str, column: usize) -> Result<Option<u64>> {
6225        let entry = self
6226            .entries
6227            .iter()
6228            .find(|entry| entry.name == name)
6229            .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
6230        let Some(field) = entry.fields.get(column) else {
6231            return Err(invalid("frequency column index out of range"));
6232        };
6233        if !matches!(
6234            field.ty,
6235            LogicalType::TinyInt
6236                | LogicalType::SmallInt
6237                | LogicalType::Integer
6238                | LogicalType::BigInt
6239                | LogicalType::UTinyInt
6240                | LogicalType::USmallInt
6241                | LogicalType::UInteger
6242                | LogicalType::UBigInt
6243        ) {
6244            return Ok(None);
6245        }
6246        let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
6247        if file_checksum(&self.file, offset, length)? != entry.directory.hash {
6248            return Err(invalid(&format!("the directory of table {name} does not checksum")));
6249        }
6250        if let Some(Some(frequencies)) = entry.frequencies.get(column) {
6251            return frequencies
6252                .iter()
6253                .filter(|(value, _)| value.is_some_and(|value| value != 0))
6254                .try_fold(0_u64, |total, (_, count)| total.checked_add(*count))
6255                .map(Some)
6256                .ok_or_else(|| invalid("numeric frequency count overflow"));
6257        }
6258        quick_nonzero(
6259            Cursor::over(&self.file, offset, length),
6260            &entry.name,
6261            &entry.fields,
6262            entry.rows,
6263            column,
6264        )
6265    }
6266
6267    /// Exact signed-integer sums and non-null counts from the small catalog. The table directory
6268    /// checksum is still checked once before any certificate can answer a query.
6269    pub fn aggregate_sums(&self, name: &str, columns: &[usize]) -> Result<Option<CertifiedSums>> {
6270        let entry = self
6271            .entries
6272            .iter()
6273            .find(|entry| entry.name == name)
6274            .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
6275        let mut sums = Vec::with_capacity(columns.len());
6276        for &column in columns {
6277            let Some(field) = entry.fields.get(column) else {
6278                return Err(invalid("aggregate column index out of range"));
6279            };
6280            if !signed_integer(&field.ty) {
6281                return Ok(None);
6282            }
6283            let Some(sum) = entry.aggregates[column] else {
6284                return Ok(None);
6285            };
6286            sums.push(sum);
6287        }
6288        let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
6289        if file_checksum(&self.file, offset, length)? != entry.directory.hash {
6290            return Err(invalid(&format!("the directory of table {name} does not checksum")));
6291        }
6292        Ok(Some(CertifiedSums { columns: sums, rows: entry.rows as u64 }))
6293    }
6294
6295    /// Exact non-null distinct count from the small catalog, after checking the table directory.
6296    pub fn distinct_count(&self, name: &str, column: usize) -> Result<Option<u64>> {
6297        let entry = self
6298            .entries
6299            .iter()
6300            .find(|entry| entry.name == name)
6301            .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
6302        let Some(count) = entry.distincts.get(column).copied() else {
6303            return Err(invalid("distinct column index out of range"));
6304        };
6305        let Some(count) = count else { return Ok(None) };
6306        let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
6307        if file_checksum(&self.file, offset, length)? != entry.directory.hash {
6308            return Err(invalid(&format!("the directory of table {name} does not checksum")));
6309        }
6310        Ok(Some(count))
6311    }
6312
6313    /// Exact integer or date ends from the small catalog after checking the table directory.
6314    pub fn integer_extremes(&self, name: &str, column: usize) -> Result<Option<IntegerExtremes>> {
6315        let entry = self
6316            .entries
6317            .iter()
6318            .find(|entry| entry.name == name)
6319            .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
6320        let Some(extremes) = entry.extremes.get(column).copied() else {
6321            return Err(invalid("extremes column index out of range"));
6322        };
6323        let Some(extremes) = extremes else { return Ok(None) };
6324        let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
6325        if file_checksum(&self.file, offset, length)? != entry.directory.hash {
6326            return Err(invalid(&format!("the directory of table {name} does not checksum")));
6327        }
6328        Ok(Some(match extremes {
6329            None => IntegerExtremes::Null,
6330            Some((low, high)) => IntegerExtremes::Values { low, high },
6331        }))
6332    }
6333
6334    /// Complete numeric frequencies from the small catalog, after checking the table directory.
6335    pub fn exact_numeric_frequencies(
6336        &self,
6337        name: &str,
6338        column: usize,
6339    ) -> Result<Option<NumericFrequencies>> {
6340        let entry = self
6341            .entries
6342            .iter()
6343            .find(|entry| entry.name == name)
6344            .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
6345        let Some(frequencies) = entry.frequencies.get(column).cloned() else {
6346            return Err(invalid("numeric frequency column index out of range"));
6347        };
6348        let Some(frequencies) = frequencies else { return Ok(None) };
6349        let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
6350        if file_checksum(&self.file, offset, length)? != entry.directory.hash {
6351            return Err(invalid(&format!("the directory of table {name} does not checksum")));
6352        }
6353        Ok(Some(frequencies))
6354    }
6355
6356    /// The schema copied into the small file catalog, available without opening the table directory.
6357    pub fn table_fields(&self, name: &str) -> Option<&[Field]> {
6358        self.entries.iter().find(|entry| entry.name == name).map(|entry| entry.fields.as_slice())
6359    }
6360}
6361
6362/// Where the slot naming `generation` goes, which is the one the generation before it did not use.
6363///
6364/// Generation 1 takes the slot at 16, so a file written once is byte for byte the file this wrote
6365/// before there was a second generation to write.
6366fn slot_offset(generation: u64) -> u64 {
6367    16 + (generation - 1) % 2 * SLOT_BYTES as u64
6368}
6369
6370/// The header and the bytes the highest valid slot points at.
6371///
6372/// Both levels of the directory are reached this way, so the magic check, the version check and the
6373/// choice between the two slots live here rather than being written out twice.
6374fn slot_bytes(path: impl AsRef<Path>) -> Result<(File, u64, Slot, Vec<u8>, Opening)> {
6375    let file = File::open(path).map_err(io)?;
6376    let size = file.metadata().map_err(io)?.len();
6377    let (slot, bytes, opening) = committed_slot(&file, size)?;
6378    Ok((file, size, slot, bytes, opening))
6379}
6380
6381/// The committed slot of a file that is `size` bytes long, and the catalog it points at.
6382///
6383/// The half of [`slot_bytes`] that does not care how the file was opened. A reader comes here with
6384/// the `std::fs::File` it goes on to share between its threads, and a writer with the `rudb_io`
6385/// file it is about to append to.
6386fn committed_slot<F: Positional + ?Sized>(file: &F, size: u64) -> Result<(Slot, Vec<u8>, Opening)> {
6387    if size < HEADER {
6388        return Err(invalid("file is shorter than its header"));
6389    }
6390    let mut header = [0; HEADER as usize];
6391    read_at(file, 0, &mut header)?;
6392    let mut opening = Opening { reads: 1, bytes: HEADER };
6393    let version = u32::from_le_bytes([header[8], header[9], header[10], header[11]]);
6394    // The two halves are worth telling apart. A wrong magic is a file that was never ours and
6395    // the answer is to look at the path. A wrong version is our own file from another build,
6396    // and the number this build wants is the only thing that tells the reader whether to
6397    // rebuild the file or to go back to the binary that wrote it.
6398    if &header[..8] != MAGIC {
6399        return Err(invalid("the header does not begin with a rudb native magic"));
6400    }
6401    if !READABLE.contains(&version) {
6402        return Err(invalid(&format!(
6403            "the file is format {version} and this build reads format {FORMAT}, so it has to \
6404                 be written again"
6405        )));
6406    }
6407    let mut selected = None;
6408    for start in [16, 16 + SLOT_BYTES] {
6409        let slot = Slot::read(&header[start..start + SLOT_BYTES]);
6410        if slot.generation == 0 || slot.length == 0 || slot.length as usize > MAX_DIRECTORY {
6411            continue;
6412        }
6413        let Some(end) = slot.offset.checked_add(u64::from(slot.length)) else { continue };
6414        if slot.offset < HEADER || end > size {
6415            continue;
6416        }
6417        let mut bytes = vec![0; slot.length as usize];
6418        read_at(file, slot.offset, &mut bytes)?;
6419        opening.reads += 1;
6420        opening.bytes += u64::from(slot.length);
6421        if checksum(&bytes) == slot.hash
6422            && selected
6423                .as_ref()
6424                .is_none_or(|(old, _): &(Slot, Vec<u8>)| old.generation < slot.generation)
6425        {
6426            selected = Some((slot, bytes));
6427        }
6428    }
6429    let (slot, bytes) = selected.ok_or_else(|| invalid("no committed directory slot is valid"))?;
6430    Ok((slot, bytes, opening))
6431}
6432
6433impl Reader {
6434    /// Opens a file that holds exactly one table.
6435    ///
6436    /// # Errors
6437    ///
6438    /// If the file has no valid committed directory, a directory pointer is out of bounds, or the
6439    /// file holds more than one table, which is a file that has to be opened by name.
6440    pub fn open(path: impl AsRef<Path>) -> Result<Self> {
6441        let catalog = Catalog::open(path)?;
6442        let mut names = catalog.names();
6443        let name = names.next().ok_or_else(|| invalid("the file holds no table"))?.to_string();
6444        if names.next().is_some() {
6445            return Err(invalid(
6446                "the file holds more than one table, so it has to be opened by name",
6447            ));
6448        }
6449        catalog.table(&name)
6450    }
6451
6452    /// Builds a reader over one decoded table directory.
6453    fn build(
6454        file: Arc<File>,
6455        map: Option<Arc<Mapped>>,
6456        size: u64,
6457        table: Table,
6458        directory: u64,
6459        opening: Opening,
6460        pool: PagePool,
6461    ) -> Result<Self> {
6462        let places = places(&table)?;
6463        let dictionaries = (0..table.fields.len()).map(|_| OnceLock::new()).collect();
6464        let table_fields = table.fields.len();
6465        let columns = (0..table_fields).map(|_| Mutex::new(Cached::default())).collect::<Vec<_>>();
6466        let cache = Shelf {
6467            columns,
6468            held: (0..table_fields).map(|_| AtomicUsize::new(0)).collect(),
6469            kept: AtomicUsize::new(CACHED_STRIPES_PER_COLUMN),
6470        };
6471        let sieves = (0..table_fields).map(|_| OnceLock::new()).collect();
6472        let part_ranges = (0..table_fields).map(|_| OnceLock::new()).collect();
6473        let verified = (places.len() * table_fields).div_ceil(64);
6474        let unreleased = table
6475            .stripes
6476            .iter()
6477            .flat_map(|stripe| {
6478                let parts = u32::try_from(stripe.parts.len()).unwrap_or(u32::MAX);
6479                (0..table_fields).map(move |_| AtomicU32::new(parts))
6480            })
6481            .collect();
6482        let firsts = places
6483            .iter()
6484            .scan(0, |first, place| {
6485                let at = *first;
6486                *first += place.rows as usize;
6487                Some(at)
6488            })
6489            .collect();
6490        Ok(Self {
6491            file,
6492            map,
6493            table: Arc::new(table),
6494            dictionaries: Arc::new(dictionaries),
6495            loading: Arc::new((0..table_fields).map(|_| Mutex::new(())).collect()),
6496            frequency_values: Arc::new((0..table_fields).map(|_| OnceLock::new()).collect()),
6497            frequency_summaries: Arc::new((0..table_fields).map(|_| OnceLock::new()).collect()),
6498            frequency_heads: Arc::new((0..table_fields).map(|_| OnceLock::new()).collect()),
6499            summaries: Arc::new((0..table_fields).map(|_| OnceLock::new()).collect()),
6500            facts: Arc::new(OnceLock::new()),
6501            opened: Arc::new(AtomicUsize::new(0)),
6502            sieves: Arc::new(sieves),
6503            part_ranges: Arc::new(part_ranges),
6504            places: Arc::new(places),
6505            cache: Arc::new(cache),
6506            pool,
6507            pages: Arc::new(AtomicUsize::new(0)),
6508            indexes: Arc::new(AtomicUsize::new(0)),
6509            verified: Arc::new((0..verified).map(|_| AtomicU64::new(0)).collect()),
6510            unreleased: Arc::new(unreleased),
6511            text_grams: Arc::new((0..table_fields).map(|_| OnceLock::new()).collect()),
6512            firsts: Arc::new(firsts),
6513            graph: Arc::default(),
6514            size,
6515            directory,
6516            opening,
6517        })
6518    }
6519
6520    /// What this reader has read so far, and what opening it cost.
6521    ///
6522    /// Public because the claim of `spec/stats/04-in-memory.md` section 4.2 is about this number
6523    /// and a claim nobody can check is a comment. A caller that wants to know whether opening a
6524    /// file touched the data asks here, and gets an answer that does not depend on what the page
6525    /// cache happened to hold.
6526    #[must_use]
6527    pub fn reads(&self) -> Reads {
6528        Reads {
6529            opening: self.opening,
6530            pages: self.pages.load(Atomic::Relaxed),
6531            indexes: self.indexes.load(Atomic::Relaxed),
6532            dictionaries: self.opened.load(Atomic::Relaxed),
6533        }
6534    }
6535
6536    /// Where the file's bytes went, from the directory alone.
6537    ///
6538    /// No page is read, so this costs the same on a 45 GB table as on an empty one. See [`Layout`]
6539    /// for what is charged where and for why the three things that are not columns stay separate.
6540    #[must_use]
6541    pub fn layout(&self) -> Layout {
6542        let table = &self.table;
6543        let stripes = table.stripes.as_slice();
6544        let columns = table
6545            .fields
6546            .iter()
6547            .enumerate()
6548            .map(|(at, field)| ColumnLayout {
6549                name: field.name.clone(),
6550                kind: field.ty.to_string(),
6551                pages: sum(stripes.iter().map(|stripe| span_bytes(&stripe.pages, at))),
6552                memberships: sum(stripes.iter().map(|stripe| stripe.memberships.bytes(at))),
6553                sieves: sum(stripes.iter().map(|stripe| stripe.sieves.bytes(at))),
6554                part_ranges: sum(stripes.iter().map(|stripe| stripe.part_ranges.bytes(at))),
6555                dictionary: dictionary_bytes(table, at),
6556            })
6557            .collect();
6558        Layout {
6559            file: self.size,
6560            rows: table.rows,
6561            stripes: stripes.len(),
6562            parts: self.places.len(),
6563            columns,
6564            indexes: sum(stripes.iter().map(|stripe| u64::from(stripe.index.length))),
6565            directory: self.directory,
6566            header: HEADER,
6567        }
6568    }
6569
6570    /// What every part of one column is stored as, which is what `pragma_storage_info` reports.
6571    ///
6572    /// Unlike [`Self::layout`] this reads the data, because the encoder's choice is in the page and
6573    /// nowhere else. The directory says how many bytes a column took and says nothing about what
6574    /// shape they are in, and the shape is the question worth asking: the same rows in a different
6575    /// order come back bit packed on one file and plain on another, and that is the difference a
6576    /// clustered load makes to a scan.
6577    ///
6578    /// One read per stripe rather than one per part. A part is a few kilobytes out of a page that
6579    /// is a quarter of a megabyte, so asking part by part would read the same page sixty four
6580    /// times. Nothing is put in the page cache, because a caller asking what a file looks like is
6581    /// not about to scan it and evicting the pages a real query wants would be a poor trade.
6582    ///
6583    /// # Errors
6584    ///
6585    /// If the column is outside the schema, or a page, index section or checksum is invalid.
6586    pub fn stored(&self, column: usize) -> Result<Vec<StoredPart>> {
6587        let field = self
6588            .table
6589            .fields
6590            .get(column)
6591            .ok_or_else(|| invalid("stored column index out of range"))?;
6592        let mut stored = Vec::with_capacity(self.places.len());
6593        let mut row = 0;
6594        for (at, stripe) in self.table.stripes.iter().enumerate() {
6595            let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
6596            let index = read_index(&self.file, stripe, column)?;
6597            let mut bytes = vec![0; page.length as usize];
6598            read_at(&self.file, page.offset, &mut bytes)?;
6599            let ranges = self.stripe_part_ranges(at, column);
6600            for (part, &rows) in stripe.parts.iter().enumerate() {
6601                let span = *index.get(part).ok_or_else(|| invalid("part index out of range"))?;
6602                let held = part_bytes(&bytes, span)?;
6603                let range = ranges.and_then(|held| held.get(part));
6604                stored.push(StoredPart {
6605                    stripe: at,
6606                    part,
6607                    row,
6608                    rows: rows as usize,
6609                    encoding: page_encoding(&field.ty, rows as usize, held),
6610                    bytes: span.length as u64,
6611                    page: page.offset,
6612                    offset: span.start as u64,
6613                    low: range
6614                        .and_then(|range| range.low.clone())
6615                        .and_then(|bound| bound.into_value(&field.ty)),
6616                    high: range
6617                        .and_then(|range| range.high.clone())
6618                        .and_then(|bound| bound.into_value(&field.ty)),
6619                    nulls: range.map(|range| range.nulls),
6620                });
6621                row += rows as usize;
6622            }
6623        }
6624        Ok(stored)
6625    }
6626
6627    /// How many parts the table has, which is how many chunks a scan of it reads.
6628    #[must_use]
6629    pub fn parts(&self) -> usize {
6630        self.places.len()
6631    }
6632
6633    /// The parts of each stripe, in table wide part numbers.
6634    ///
6635    /// A scan that wants one worker to own the page it reads hands work out in these runs. The
6636    /// stripes are contiguous in part numbering and all but the last hold sixty four parts, but a
6637    /// stripe can be flushed early when rows arrive out of order, so the runs are read off the
6638    /// directory rather than worked out from a constant.
6639    #[must_use]
6640    pub fn stripe_parts(&self) -> Vec<std::ops::Range<usize>> {
6641        let mut runs = Vec::with_capacity(self.table.stripes.len());
6642        let mut start = 0;
6643        for stripe in &self.table.stripes {
6644            let end = start + stripe.parts.len();
6645            runs.push(start..end);
6646            start = end;
6647        }
6648        runs
6649    }
6650
6651    /// How many rows one stripe holds, in the numbering [`Self::stripe_parts`] hands back.
6652    ///
6653    /// Off the directory, which is already in memory, rather than by the caller asking for each
6654    /// part in turn through the catalog. Nothing past the end holds any rows.
6655    #[must_use]
6656    pub fn stripe_rows(&self, stripe: usize) -> usize {
6657        self.table.stripes.get(stripe).map_or(0, |held| held.rows)
6658    }
6659
6660    /// Asks the page cache to keep `stripes` stripes of every column instead of the default.
6661    ///
6662    /// This only ever raises the number. A scan that gives each worker a whole stripe has one page
6663    /// per column per worker open at once, and a cache smaller than that is worse than no cache at
6664    /// all: every worker's page is evicted by the others before it has finished its stripe, so it
6665    /// reads a quarter of a megabyte for every part it takes out of it.
6666    pub fn keep_stripes(&self, stripes: usize) {
6667        self.cache.kept.fetch_max(stripes, Atomic::Relaxed);
6668    }
6669
6670    /// Rows in one part, or zero when the part number is past the table.
6671    #[must_use]
6672    pub fn part_rows(&self, at: usize) -> usize {
6673        self.places.get(at).map_or(0, |place| place.rows as usize)
6674    }
6675
6676    /// The committed table directory.
6677    #[must_use]
6678    pub fn table(&self) -> &Table {
6679        &self.table
6680    }
6681
6682    /// Exact leading frequencies when the stored synopsis proves a count-descending prefix.
6683    ///
6684    /// The returned list can be longer than `top`. Keeping the stored tail lets a later TopN apply
6685    /// additional ordering keys without losing a value tied with the requested boundary.
6686    ///
6687    /// # Errors
6688    ///
6689    /// If the column is outside the schema or a stored value does not fit its declared type.
6690    pub fn top_frequencies(&self, column: usize, top: usize) -> Result<Option<Vec<(Value, u64)>>> {
6691        let field = self
6692            .table
6693            .fields
6694            .get(column)
6695            .ok_or_else(|| invalid("frequency column index out of range"))?;
6696        let Some((entries, omitted_max)) = self.frequency_head(column)? else {
6697            return Ok(None);
6698        };
6699        if top == 0 || entries.len() < top {
6700            return Ok(None);
6701        }
6702        let boundary = entries[top - 1].count;
6703        if boundary <= omitted_max {
6704            return Ok(None);
6705        }
6706        self.decode_frequencies(column, &field.ty, &entries).map(|values| Some(Vec::clone(&values)))
6707    }
6708
6709    /// Exact leading counts for a numeric key paired with a stable-dictionary string key.
6710    ///
6711    /// Legacy pair summaries are parsed for file compatibility but never used as query output.
6712    ///
6713    /// # Errors
6714    ///
6715    /// If either column is outside the schema.
6716    pub fn top_pair_frequencies(
6717        &self,
6718        first: usize,
6719        second: usize,
6720        _top: usize,
6721    ) -> Result<Option<PairFrequencyCounts>> {
6722        if first >= self.table.fields.len() || second >= self.table.fields.len() {
6723            return Err(invalid("pair frequency column index out of range"));
6724        }
6725        Ok(None)
6726    }
6727
6728    /// Every value of one column with the number of rows holding it, when the synopsis is complete.
6729    ///
6730    /// The heavy hitter pass keeps a bounded set of candidates and decrements them all when it runs
6731    /// out of room, so what it usually ends with is the leading values and a bound on everything it
6732    /// dropped. `omitted_max` of zero says that never happened: no candidate was ever decremented and
6733    /// the entries did not overflow the stored budget, so the list is every distinct value of the
6734    /// column with an exact count, and a null counts as a value of its own rather than being skipped.
6735    ///
6736    /// That makes a whole class of question answerable without reading a row. How many rows hold a
6737    /// value, how many do not, and what a `GROUP BY` of that column with a count over it produces are
6738    /// all in here. It is only ever true of a column with few enough distinct values, which is the
6739    /// case worth having, because that is exactly the column a grouping or an equality filter would
6740    /// otherwise walk every row to answer.
6741    ///
6742    /// `None` when the column has no synopsis, or has one that dropped anything.
6743    ///
6744    /// # Errors
6745    ///
6746    /// If the column is outside the schema or a stored value does not fit its declared type.
6747    pub fn exact_frequencies(&self, column: usize) -> Result<Option<Vec<(Value, u64)>>> {
6748        let Some(prefix) = self.frequency_prefix(column)? else {
6749            return Ok(None);
6750        };
6751        Ok((prefix.omitted_max == 0).then_some(prefix.entries))
6752    }
6753
6754    /// Every value the synopsis lists with the number of rows holding it, and a bound on the rest.
6755    ///
6756    /// The counts are exact whether or not the list is complete. The heavy hitter pass keeps a
6757    /// bounded candidate set and then recounts only the candidates that survived it, so a value that
6758    /// made it into the list carries the number of rows that really hold it rather than whatever the
6759    /// pass had left over. What the pass loses is values, not counts.
6760    ///
6761    /// `omitted_max` is how many rows the most common value left out can hold, and zero says nothing
6762    /// was left out at all, which is what [`exact_frequencies`] asks for. Above zero the list is the
6763    /// leading values of the column and everything else is somewhere between no rows and that bound.
6764    ///
6765    /// That prefix is worth reading on its own. A column with a value in half its rows and a long
6766    /// tail behind it has no complete synopsis and never will, and it is the column where dividing
6767    /// the rows by the distinct count is furthest from the truth.
6768    ///
6769    /// `None` when the column has no synopsis.
6770    ///
6771    /// # Errors
6772    ///
6773    /// If the column is outside the schema or a stored value does not fit its declared type.
6774    ///
6775    /// [`exact_frequencies`]: Self::exact_frequencies
6776    pub fn frequency_prefix(&self, column: usize) -> Result<Option<FrequencyPrefix>> {
6777        Ok(self.held_prefix(column)?.map(|(entries, omitted_max)| FrequencyPrefix {
6778            entries: Vec::clone(&entries),
6779            omitted_max,
6780        }))
6781    }
6782
6783    /// [`Self::frequency_prefix`] as the reader holds it, shared rather than copied.
6784    ///
6785    /// The planner asks for a column's synopsis for every estimate that touches it, on every
6786    /// statement, and the list of a string column is a few hundred strings, so copying it each time
6787    /// was a hundred allocations for an answer nothing changes.
6788    pub(crate) fn held_prefix(&self, column: usize) -> Result<Option<(Synopsis, u64)>> {
6789        let field = self
6790            .table
6791            .fields
6792            .get(column)
6793            .ok_or_else(|| invalid("frequency column index out of range"))?;
6794        let Some((entries, omitted_max)) = self.frequency_head(column)? else {
6795            return Ok(None);
6796        };
6797        let entries = self.decode_frequencies(column, &field.ty, &entries)?;
6798        Ok(Some((entries, omitted_max)))
6799    }
6800
6801    /// One column's synopsis entries and the bound on what they leave out. A synopsis left in the
6802    /// file is read only as far as its entries go, unless all of it was already read.
6803    fn frequency_head(&self, column: usize) -> Result<Option<(Cow<'_, [FrequencyEntry]>, u64)>> {
6804        let (span, count) = match self.table.frequencies.get(column) {
6805            None | Some(None) => return Ok(None),
6806            Some(Some(Frequencies::Held(summary))) => {
6807                return Ok(Some((Cow::Borrowed(&summary.entries), summary.omitted_max)));
6808            }
6809            Some(Some(Frequencies::Stored { span, entries, .. })) => (span, *entries),
6810        };
6811        if let Some(summary) = self.frequency_summaries.get(column).and_then(OnceLock::get) {
6812            return Ok(Some((Cow::Borrowed(&summary.entries), summary.omitted_max)));
6813        }
6814        let slot = self
6815            .frequency_heads
6816            .get(column)
6817            .ok_or_else(|| invalid("frequency column index out of range"))?;
6818        if slot.get().is_none() {
6819            let field = self
6820                .table
6821                .fields
6822                .get(column)
6823                .ok_or_else(|| invalid("frequency column index out of range"))?;
6824            // A tag, the bound, the entry count, and then at most a tag, sixteen value bytes and
6825            // an eight byte count for each entry.
6826            let length = (span.length as usize).min(13 + count * 25);
6827            let mut bytes = vec![0; length];
6828            read_at(&self.file, span.offset, &mut bytes)?;
6829            let head = decode_summary_head(&mut Cursor::new(&bytes), field, self.table.rows)?
6830                .ok_or_else(|| invalid("a stored synopsis is missing"))?;
6831            if head.0.len() != count {
6832                return Err(invalid("a stored synopsis differs from its directory span"));
6833            }
6834            let _ = slot.set(Arc::new(head));
6835        }
6836        let (entries, omitted_max) = slot.get().expect("the synopsis head was stored").as_ref();
6837        Ok(Some((Cow::Borrowed(entries), *omitted_max)))
6838    }
6839
6840    /// One column's synopsis, read back from the file when the directory left it there.
6841    fn frequency_summary(&self, column: usize) -> Result<Option<Cow<'_, FrequencySummary>>> {
6842        Ok(match self.table.frequencies.get(column) {
6843            None | Some(None) => None,
6844            Some(Some(Frequencies::Held(summary))) => Some(Cow::Borrowed(summary)),
6845            Some(Some(Frequencies::Stored { span, values, entries })) => {
6846                let slot = self
6847                    .frequency_summaries
6848                    .get(column)
6849                    .ok_or_else(|| invalid("frequency column index out of range"))?;
6850                if let Some(summary) = slot.get() {
6851                    return Ok(Some(Cow::Borrowed(summary.as_ref())));
6852                }
6853                let field = self
6854                    .table
6855                    .fields
6856                    .get(column)
6857                    .ok_or_else(|| invalid("frequency column index out of range"))?;
6858                let mut bytes = vec![0; span.length as usize];
6859                read_at(&self.file, span.offset, &mut bytes)?;
6860                let mut cur = Cursor::new(&bytes);
6861                let summary = decode_summary(&mut cur, field, self.table.rows, *values)?;
6862                let summary = summary.ok_or_else(|| invalid("a stored synopsis is missing"))?;
6863                if !cur.done() || summary.entries.len() != *entries {
6864                    return Err(invalid("a stored synopsis differs from its directory span"));
6865                }
6866                let _ = slot.set(Arc::new(summary));
6867                Some(Cow::Borrowed(slot.get().expect("the decoded summary was stored").as_ref()))
6868            }
6869        })
6870    }
6871
6872    /// Turns stored frequency entries into values of the column's own type.
6873    ///
6874    /// Remembered per column, because the planner asks once for every estimate that touches the
6875    /// column and the executor asks again, and the answer is a few hundred values. The codes of a
6876    /// string column are read through [`Vector::try_values_visited`], which does not keep the blocks
6877    /// it decodes, so what a query answered out of the synopsis holds is those values and not the
6878    /// hundred or so dictionary blocks they are scattered over.
6879    fn decode_frequencies(
6880        &self,
6881        column: usize,
6882        ty: &LogicalType,
6883        entries: &[FrequencyEntry],
6884    ) -> Result<Synopsis> {
6885        if let Some(values) = self.frequency_values.get(column).and_then(OnceLock::get) {
6886            return Ok(Arc::clone(values));
6887        }
6888        let values = Arc::new(self.decode_frequencies_once(column, ty, entries)?);
6889        if let Some(slot) = self.frequency_values.get(column) {
6890            let _ = slot.set(Arc::clone(&values));
6891        }
6892        Ok(values)
6893    }
6894
6895    fn decode_frequencies_once(
6896        &self,
6897        column: usize,
6898        ty: &LogicalType,
6899        entries: &[FrequencyEntry],
6900    ) -> Result<Vec<(Value, u64)>> {
6901        let stored_texts = self.table.frequency_texts.get(column).filter(|texts| !texts.is_empty());
6902        if stored_texts.is_some_and(|texts| texts.len() != entries.len()) {
6903            return Err(invalid("frequency text count differs from its synopsis"));
6904        }
6905        let dictionary =
6906            if coded_type(ty) && stored_texts.is_none() { self.dictionary(column)? } else { None };
6907        let mut codes = entries
6908            .iter()
6909            .filter_map(|entry| match entry.value {
6910                FrequencyValue::Code(code) => Some(code as usize),
6911                _ => None,
6912            })
6913            .collect::<Vec<_>>();
6914        codes.sort_unstable();
6915        codes.dedup();
6916        let texts = match &dictionary {
6917            Some(dictionary) if !codes.is_empty() => dictionary.try_values_visited(&codes)?,
6918            _ => Vec::new(),
6919        };
6920        let mut out = Vec::with_capacity(entries.len());
6921        for (entry_at, entry) in entries.iter().enumerate() {
6922            let value = match entry.value {
6923                FrequencyValue::Null => {
6924                    if stored_texts.and_then(|texts| texts[entry_at].as_ref()).is_some() {
6925                        return Err(invalid("a null frequency entry has text"));
6926                    }
6927                    Value::Null
6928                }
6929                FrequencyValue::Integer(value) => match *ty {
6930                    LogicalType::TinyInt => Value::TinyInt(
6931                        i8::try_from(value)
6932                            .map_err(|_| invalid("frequency TINYINT is out of range"))?,
6933                    ),
6934                    LogicalType::UTinyInt => Value::UTinyInt(
6935                        u8::try_from(value)
6936                            .map_err(|_| invalid("frequency UTINYINT is out of range"))?,
6937                    ),
6938                    LogicalType::USmallInt => Value::USmallInt(
6939                        u16::try_from(value)
6940                            .map_err(|_| invalid("frequency USMALLINT is out of range"))?,
6941                    ),
6942                    LogicalType::UInteger => Value::UInteger(
6943                        u32::try_from(value)
6944                            .map_err(|_| invalid("frequency UINTEGER is out of range"))?,
6945                    ),
6946                    LogicalType::UBigInt => Value::UBigInt(
6947                        u64::try_from(value)
6948                            .map_err(|_| invalid("frequency UBIGINT is out of range"))?,
6949                    ),
6950                    LogicalType::SmallInt => Value::SmallInt(
6951                        i16::try_from(value)
6952                            .map_err(|_| invalid("frequency SMALLINT is out of range"))?,
6953                    ),
6954                    LogicalType::Integer => Value::Integer(
6955                        i32::try_from(value)
6956                            .map_err(|_| invalid("frequency INTEGER is out of range"))?,
6957                    ),
6958                    LogicalType::BigInt => Value::BigInt(
6959                        i64::try_from(value)
6960                            .map_err(|_| invalid("frequency BIGINT is out of range"))?,
6961                    ),
6962                    LogicalType::Date => Value::Date(
6963                        i32::try_from(value)
6964                            .map_err(|_| invalid("frequency DATE is out of range"))?,
6965                    ),
6966                    LogicalType::Timestamp => Value::Timestamp(
6967                        i64::try_from(value)
6968                            .map_err(|_| invalid("frequency TIMESTAMP is out of range"))?,
6969                    ),
6970                    _ => return Err(invalid("integer frequency belongs to another type")),
6971                },
6972                FrequencyValue::Code(code) => {
6973                    if let Some(text) = stored_texts.and_then(|texts| texts[entry_at].as_ref()) {
6974                        if *ty == LogicalType::Blob {
6975                            Value::Blob(text.clone())
6976                        } else {
6977                            Value::Varchar(
6978                                String::from_utf8(text.clone())
6979                                    .map_err(|_| invalid("frequency text is not UTF-8"))?,
6980                            )
6981                        }
6982                    } else {
6983                        if dictionary.is_none() {
6984                            return Err(invalid("frequency code has no dictionary or stored text"));
6985                        }
6986                        let at = codes
6987                            .binary_search(&(code as usize))
6988                            .map_err(|_| invalid("frequency code was not among the codes read"))?;
6989                        texts[at].clone()
6990                    }
6991                }
6992            };
6993            out.push((value, entry.count));
6994        }
6995        Ok(out)
6996    }
6997
6998    /// Sparse rows belonging to the bounded numeric frequency candidate set.
6999    ///
7000    /// The list is omitted when collecting it would exceed the fixed storage budget. A composite
7001    /// aggregate may accept a result over these rows only when its requested boundary is strictly
7002    /// greater than `omitted_max`.
7003    ///
7004    /// # Errors
7005    ///
7006    /// If the column is outside the schema.
7007    pub fn frequency_occurrences(&self, column: usize) -> Result<Option<FrequencyOccurrences>> {
7008        let field = self
7009            .table
7010            .fields
7011            .get(column)
7012            .ok_or_else(|| invalid("frequency column index out of range"))?;
7013        let Some(summary) = self.frequency_summary(column)? else {
7014            return Ok(None);
7015        };
7016        if summary.ordinals.is_empty() {
7017            return Ok(None);
7018        }
7019        let (anchors, anchor_indices) = if summary.ordinal_entries.len() == summary.ordinals.len() {
7020            let entries = self.decode_frequencies(column, &field.ty, &summary.entries)?;
7021            (
7022                entries.iter().map(|(value, _)| value.clone()).collect(),
7023                summary.ordinal_entries.clone(),
7024            )
7025        } else {
7026            (Vec::new(), Vec::new())
7027        };
7028        // The rows kept may be those of the leading entries alone, and then a value outside them is
7029        // bounded by the first entry left out rather than by the synopsis.
7030        let stored = self.table.ordinal_bounds.get(column).copied().unwrap_or(0);
7031        Ok(Some(FrequencyOccurrences {
7032            omitted_max: summary.omitted_max.max(summary.ordinal_bound).max(stored),
7033            ordinals: summary.ordinals.clone(),
7034            anchors,
7035            anchor_indices,
7036        }))
7037    }
7038
7039    /// How many distinct values one column holds, counting a null as no value.
7040    ///
7041    /// A string column of this format is written against one dictionary that covers the whole table.
7042    /// A code is handed out the first time a value is seen and nothing ever removes one, so the
7043    /// number of codes is the number of distinct values exactly rather than an estimate. That makes
7044    /// `COUNT(DISTINCT column)` over a whole table a question the directory already knows the answer
7045    /// to, and the alternative is a hash table with a row per distinct value built from a pass over
7046    /// every row.
7047    ///
7048    /// A null in the column used to make this `None` and no longer does. A null row is written as
7049    /// the code for the empty string, so a nullable column's dictionary can hold an empty string
7050    /// that no row of it actually has, and the dictionary on its own does not say which case it is.
7051    /// The writer does know, because it counts the non-null rows that use each code on its way to
7052    /// the frequency summary, so it records how many codes any row holds and the directory carries
7053    /// that number. This reads it rather than the size of the dictionary, which also means the
7054    /// dictionary page is not opened to answer.
7055    ///
7056    /// An integer column has no dictionary, and its count comes from the set the writer keeps on its
7057    /// numeric frequency pass instead, which is exact up to a cap. `None` for a column past that cap
7058    /// and for every column that is neither, where a sketch would answer approximately and SQL asked
7059    /// for the exact number.
7060    ///
7061    /// # Errors
7062    ///
7063    /// If the column is outside the schema.
7064    pub fn distinct_values(&self, column: usize) -> Result<Option<u64>> {
7065        self.table
7066            .distincts
7067            .get(column)
7068            .copied()
7069            .ok_or_else(|| invalid("distinct column index out of range"))
7070    }
7071
7072    /// How many rows of one column are null, added up over the stripes.
7073    ///
7074    /// Every stripe records this exactly when it is written, because a null count is not a bound
7075    /// that is allowed to be wide the way a minimum and a maximum are: a filter that reads one too
7076    /// many is slow and a `COUNT` that reads one too many is wrong. Adding up a few hundred numbers
7077    /// already in memory is what makes `COUNT(column)` over a whole table free.
7078    ///
7079    /// # Errors
7080    ///
7081    /// If the column is outside the schema.
7082    pub fn null_count(&self, column: usize) -> Result<u64> {
7083        if column >= self.table.fields.len() {
7084            return Err(invalid("null count column index out of range"));
7085        }
7086        let mut nulls = 0_u64;
7087        for stripe in &self.table.stripes {
7088            let range = stripe
7089                .zone
7090                .column(column)
7091                .ok_or_else(|| invalid("stripe zone is narrower than the schema"))?;
7092            nulls = nulls
7093                .checked_add(range.nulls as u64)
7094                .ok_or_else(|| invalid("null count overflow"))?;
7095        }
7096        Ok(nulls)
7097    }
7098
7099    /// The smallest and the largest value of one string column, from the order beside its values.
7100    ///
7101    /// The dictionary holds exactly the values the column holds, so the first and the last of them
7102    /// in sorted order are the column's minimum and maximum. Two reads of a rank block settle what
7103    /// otherwise walks a million rows.
7104    ///
7105    /// `None` when the column is not a string, when the file was written before version 9 and so has
7106    /// no order, when the column has no values at all, or when it has a null in it, which is the
7107    /// placeholder again: the empty string a null is written as would sort ahead of every real
7108    /// value and be reported as the minimum.
7109    ///
7110    /// # Errors
7111    ///
7112    /// If the column is outside the schema, or a rank names a code the dictionary does not have.
7113    pub fn text_extremes(&self, column: usize) -> Result<Option<(Value, Value)>> {
7114        if self.null_count(column)? > 0 || self.demoted(column) {
7115            return Ok(None);
7116        }
7117        let Some(dictionary) = self.dictionary(column)? else { return Ok(None) };
7118        let Some(ranks) = dictionary.ranks() else { return Ok(None) };
7119        if ranks == 0 {
7120            return Ok(None);
7121        }
7122        let low = text_at_rank(&dictionary, 0)?;
7123        let high = text_at_rank(&dictionary, ranks - 1)?;
7124        Ok(Some((low, high)))
7125    }
7126
7127    /// The smallest and the largest value of one column, when every stripe wrote exact ends.
7128    ///
7129    /// A stripe's ends are allowed to be wider than the truth, because a bound that rules out a
7130    /// chunk that could not match is still correct when it rules out nothing. That is what makes
7131    /// them cheap to write for a bit packed or a dictionary column, and it is also what stops them
7132    /// answering a `MIN`. So each stripe says which of the two it wrote, and this answers only when
7133    /// all of them walked their rows.
7134    ///
7135    /// `None` for a column with no ends, for an empty table, and for a column any stripe of which
7136    /// guessed. Nulls need no special case, because the ends skip them the same way `MIN` does.
7137    ///
7138    /// One case is given up on that did not have to be. A stripe merges the ends of its sixty four
7139    /// parts, and a part with no ends at all erases the merged ones, because a part whose rows are
7140    /// not covered by the stripe's ends is a stripe that would skip rows it should keep. A part of
7141    /// nothing but nulls has no rows to cover and so did not need to erase anything, but the merge
7142    /// cannot tell that part from a part whose layout it could not read. So a column with a chunk
7143    /// of nothing but nulls in the middle of it goes and reads the rows. That is slow and right,
7144    /// and the fix is a row count per part rather than anything here.
7145    ///
7146    /// # Errors
7147    ///
7148    /// If the column is outside the schema.
7149    pub fn exact_extremes(&self, column: usize) -> Result<Option<(Bound, Bound)>> {
7150        if column >= self.table.fields.len() {
7151            return Err(invalid("extremes column index out of range"));
7152        }
7153        let mut low: Option<Bound> = None;
7154        let mut high: Option<Bound> = None;
7155        for stripe in &self.table.stripes {
7156            let range = stripe
7157                .zone
7158                .column(column)
7159                .ok_or_else(|| invalid("stripe zone is narrower than the schema"))?;
7160            if !range.exact {
7161                return Ok(None);
7162            }
7163            // A stripe of nothing but nulls has no ends and says nothing about the column's, which
7164            // is why this skips it rather than giving up on the whole column. A stripe that has
7165            // rows and still has no end is a layout whose values this cannot see, and skipping that
7166            // one would answer with an end taken from the other stripes, so it gives up instead.
7167            let (Some(small), Some(large)) = (range.low.as_ref(), range.high.as_ref()) else {
7168                if stripe.rows > range.nulls {
7169                    return Ok(None);
7170                }
7171                continue;
7172            };
7173            low = Some(low.map_or_else(|| small.clone(), |held| held.smaller(small.clone())));
7174            high = Some(high.map_or_else(|| large.clone(), |held| held.larger(large.clone())));
7175        }
7176        Ok(low.zip(high))
7177    }
7178
7179    /// The sum of one integer column and how many rows went into it, when every stripe wrote one.
7180    ///
7181    /// The count beside the sum is the non-null rows, because that is what a `SUM` adds up and what
7182    /// an `AVG` divides by, and a caller that had to work it out from the row count and the null
7183    /// count would be doing the same walk twice.
7184    ///
7185    /// `None` for anything that is not an integer column, for a file written by something that did
7186    /// not record it, and when adding the stripes together would overflow.
7187    ///
7188    /// # Errors
7189    ///
7190    /// If the column is outside the schema.
7191    pub fn exact_sum(&self, column: usize) -> Result<Option<(i128, u64)>> {
7192        if column >= self.table.fields.len() {
7193            return Err(invalid("sum column index out of range"));
7194        }
7195        let mut total = 0_i128;
7196        let mut rows = 0_u64;
7197        for stripe in &self.table.stripes {
7198            let range = stripe
7199                .zone
7200                .column(column)
7201                .ok_or_else(|| invalid("stripe zone is narrower than the schema"))?;
7202            let Some(part) = range.sum else { return Ok(None) };
7203            let Some(sum) = total.checked_add(part) else { return Ok(None) };
7204            total = sum;
7205            rows = rows.saturating_add(stripe.rows as u64 - range.nulls as u64);
7206        }
7207        Ok(Some((total, rows)))
7208    }
7209
7210    /// Legacy derived host groups are parsed for file compatibility but never used as query output.
7211    pub fn host_groups(
7212        &self,
7213        column: usize,
7214        _minimum_count: u64,
7215    ) -> Result<Option<Vec<host::HostEntry>>> {
7216        if column >= self.table.fields.len() {
7217            return Err(invalid("host group column index out of range"));
7218        }
7219        Ok(None)
7220    }
7221
7222    /// Whether the column's dictionary stopped taking values partway through the load, and so
7223    /// decodes the stripes written before that and says nothing about the column as a whole. See
7224    /// `DEMOTED`.
7225    #[must_use]
7226    pub fn demoted(&self, column: usize) -> bool {
7227        self.table.demoted.get(column).copied().unwrap_or(false)
7228    }
7229
7230    /// The global dictionary of a column, opened once however many workers ask for it at once.
7231    ///
7232    /// The unlocked look is first because it is the answer every time after the first and it costs a
7233    /// load. Everybody who misses it queues on [`Self::loading`] and looks again on the way in, so
7234    /// the one who arrived first does the reading and the rest take what it left. Waiting is the
7235    /// cheaper thing to do: the work behind the lock is a page read, a checksum and the decode of a
7236    /// dictionary that can hold half a million entries, and the alternative is every worker of the
7237    /// scan doing all of it and all but one dropping the result on the floor.
7238    fn dictionary(&self, column: usize) -> Result<Option<Arc<Vector>>> {
7239        let Some(page) = self.table.dictionaries[column] else { return Ok(None) };
7240        if let Some(dictionary) = self.dictionaries[column].get() {
7241            return Ok(Some(Arc::clone(dictionary)));
7242        }
7243        let _queued = self.loading[column].lock().map_err(|_| invalid("a poisoned dictionary"))?;
7244        if let Some(dictionary) = self.dictionaries[column].get() {
7245            return Ok(Some(Arc::clone(dictionary)));
7246        }
7247        self.opened.fetch_add(1, Atomic::Relaxed);
7248        let dictionary = Arc::new(open_global_dictionary(
7249            Arc::clone(&self.file),
7250            page,
7251            &self.table.fields[column].ty,
7252            TEXT_KEEP_BUDGET,
7253        )?);
7254        let _ = self.dictionaries[column].set(Arc::clone(&dictionary));
7255        Ok(Some(dictionary))
7256    }
7257
7258    /// Reads one section's extent table and checks it against the entry that names it.
7259    ///
7260    /// # Errors
7261    ///
7262    /// If the entry points outside the file, the table does not checksum, or it does not decode as
7263    /// a run of extents in element order.
7264    pub fn extents(&self, of: &Section) -> Result<Vec<section::Extent>> {
7265        if of.extent_bytes == 0 {
7266            return Ok(Vec::new());
7267        }
7268        let mut bytes = vec![0; of.extent_bytes as usize];
7269        read_at(&self.file, of.extent_page, &mut bytes)?;
7270        if checksum(&bytes) != of.hash {
7271            return Err(invalid("a section's extent table does not checksum"));
7272        }
7273        let extents = section::decode_extents(&bytes)?;
7274        if extents.len() != of.extents as usize {
7275            return Err(invalid("a section's extent table is not the length the entry says"));
7276        }
7277        Ok(extents)
7278    }
7279
7280    /// Reads and verifies one extent of a section.
7281    ///
7282    /// This is what section 3.2's second rule is for. A reduction that needs one extent of a two
7283    /// gigabyte forward link reads and checksums that extent and nothing else, which is the whole
7284    /// difference between a structure that works at SF100 and issue #745.
7285    ///
7286    /// # Errors
7287    ///
7288    /// If the extent points outside the file, or its bytes do not checksum.
7289    pub fn extent(&self, of: &section::Extent) -> Result<Vec<u8>> {
7290        let mut bytes = Vec::new();
7291        self.extent_into(of, &mut bytes)?;
7292        Ok(bytes)
7293    }
7294
7295    /// Read a verified extent into a caller-owned buffer so repeated extents can reuse its pages.
7296    fn extent_into(&self, of: &section::Extent, bytes: &mut Vec<u8>) -> Result<()> {
7297        bytes.resize(of.length as usize, 0);
7298        self.extent_in_place(of, bytes)
7299    }
7300
7301    /// Reads and verifies one extent into `bytes`, which is exactly its length.
7302    fn extent_in_place(&self, of: &section::Extent, bytes: &mut [u8]) -> Result<()> {
7303        let end = of
7304            .offset
7305            .checked_add(u64::from(of.length))
7306            .ok_or_else(|| invalid("an extent overflows the file"))?;
7307        if of.offset < HEADER || end > self.size || bytes.len() != of.length as usize {
7308            return Err(invalid("an extent is outside the file"));
7309        }
7310        read_at(&self.file, of.offset, bytes)?;
7311        if checksum(bytes) != of.hash {
7312            return Err(invalid("an extent does not checksum"));
7313        }
7314        Ok(())
7315    }
7316
7317    /// Reads the first `len` bytes of a section's payload, or all of it when it is shorter, without
7318    /// checking them.
7319    ///
7320    /// Only the extent table is checked, because an extent's checksum is over the whole extent and
7321    /// checking it is reading the whole of it, which is what this is here to avoid. It is for a
7322    /// kind-specific header that a planner reads to decide what to plan, and never for bytes a
7323    /// query's answer is made of: a reader that goes on to use the structure reads it again through
7324    /// [`Self::payload`], and a header that was torn is found there.
7325    ///
7326    /// # Errors
7327    ///
7328    /// If the extent table fails its check or the first extent points outside the file.
7329    pub fn payload_head(&self, of: &Section, len: usize) -> Result<Vec<u8>> {
7330        let extents = self.extents(of)?;
7331        let Some(first) = extents.first() else { return Ok(Vec::new()) };
7332        let end = first
7333            .offset
7334            .checked_add(u64::from(first.length))
7335            .ok_or_else(|| invalid("an extent overflows the file"))?;
7336        if first.offset < HEADER || end > self.size {
7337            return Err(invalid("an extent is outside the file"));
7338        }
7339        let mut bytes = vec![0; len.min(first.length as usize)];
7340        read_at(&self.file, first.offset, &mut bytes)?;
7341        Ok(bytes)
7342    }
7343
7344    /// Reads a whole section's payload, every extent of it, in order.
7345    ///
7346    /// For a structure that is resident anyway, which a key map is. Anything large enough that the
7347    /// split matters should be walking [`Reader::extents`] and taking the one it needs.
7348    ///
7349    /// Each extent is read where it goes in the payload. Read into a buffer of its own and copied
7350    /// over, every byte of a section went to fresh memory twice, and in TPC-H q21 loading the
7351    /// sections was half the page faults of a query whose system time was as large as its user time.
7352    ///
7353    /// # Errors
7354    ///
7355    /// If the extent table or any extent fails its check.
7356    pub fn payload(&self, of: &Section) -> Result<Vec<u8>> {
7357        let extents = self.extents(of)?;
7358        let total = usize::try_from(sum(extents.iter().map(|one| u64::from(one.length))))
7359            .map_err(|_| invalid("a section longer than fits in memory"))?;
7360        let mut bytes = vec![0; total];
7361        let mut at = 0;
7362        for one in &extents {
7363            if one.first != at as u64 {
7364                return Err(invalid("a section's extents do not join up"));
7365            }
7366            let end = at + one.length as usize;
7367            self.extent_in_place(one, &mut bytes[at..end])?;
7368            at = end;
7369        }
7370        // The same exception `write_section` makes: a budget record has no bytes, so its
7371        // `header_bytes` is a size rather than a header and there is nothing for it to run past.
7372        if !bytes.is_empty() && of.header_bytes as usize > bytes.len() {
7373            return Err(invalid("a section's header is longer than its payload"));
7374        }
7375        Ok(bytes)
7376    }
7377
7378    /// Reads only the named columns from one part.
7379    ///
7380    /// The whole stripe page each column lives in is read and kept once a scan has been through the
7381    /// stripe before, because a session that scans a table again asks for the parts of a stripe one
7382    /// after another and this is what turns sixty four reads into one. The first time through, the
7383    /// part is read alone. See `Cached`.
7384    ///
7385    /// # Errors
7386    ///
7387    /// If a part, column, page, or checksum is invalid.
7388    pub fn read(&self, part: usize, columns: &[usize]) -> Result<Chunk> {
7389        self.read_impl(part, columns, true, None)
7390    }
7391
7392    /// Reads named columns from one part without keeping the stripe page it came out of.
7393    ///
7394    /// This is for sparse row fetches after a selective TopN or filter, which reach a few parts of
7395    /// a stripe rather than all of them. A caller that will read most of a stripe should use
7396    /// [`Self::read`] instead, because this reads and discards the page index every time.
7397    ///
7398    /// # Errors
7399    ///
7400    /// If a part, column, page, or checksum is invalid.
7401    pub fn read_sparse(&self, part: usize, columns: &[usize]) -> Result<Chunk> {
7402        self.read_impl(part, columns, false, None)
7403    }
7404
7405    /// Counts one signed integer part from its encoded row values when it uses an all-valid
7406    /// cascade. Sparse and run-length cascades are folded without expanding their rows. Other
7407    /// page forms return `None` so the caller can use the ordinary reader.
7408    ///
7409    /// # Errors
7410    ///
7411    /// If a part, column, page checksum, or encoded integer is invalid.
7412    pub fn integer_tally(&self, part: usize, column: usize) -> Result<Option<Vec<(i64, u64)>>> {
7413        let place = *self.places.get(part).ok_or_else(|| invalid("part index out of range"))?;
7414        let field =
7415            self.table.fields.get(column).ok_or_else(|| invalid("column index out of range"))?;
7416        if !matches!(
7417            field.ty,
7418            LogicalType::TinyInt
7419                | LogicalType::SmallInt
7420                | LogicalType::Integer
7421                | LogicalType::BigInt
7422        ) {
7423            return Ok(None);
7424        }
7425        let (rows, counts) = match self.with_part(place, column, |bytes| {
7426            if bytes.first() != Some(&5) || bytes.get(1) != Some(&0) {
7427                return Ok(None);
7428            }
7429            integer::tally(&bytes[2..]).map(Some)
7430        })? {
7431            Some(tallied) => tallied,
7432            None => return Ok(None),
7433        };
7434        if rows != place.rows as usize {
7435            return Err(invalid("encoded integer part holds the wrong number of rows"));
7436        }
7437        for &(value, _) in &counts {
7438            let fits = match field.ty {
7439                LogicalType::TinyInt => i8::try_from(value).is_ok(),
7440                LogicalType::SmallInt => i16::try_from(value).is_ok(),
7441                LogicalType::Integer => i32::try_from(value).is_ok(),
7442                LogicalType::BigInt => true,
7443                _ => false,
7444            };
7445            if !fits {
7446                return Err(invalid("encoded integer value is outside its column type"));
7447            }
7448        }
7449        Ok(Some(counts))
7450    }
7451
7452    /// The rows of one text part that hold `sequence`'s pieces in order, or with `negated` the rows
7453    /// that do not, answered on the compressed page without decompressing it. Nulls are in neither.
7454    /// `None` for a part that is not compressed text, which the caller reads the usual way.
7455    ///
7456    /// For a scan whose filter is the only thing that reads the column, which then never has the
7457    /// strings at all. In TPC-H q13 that is `o_comment NOT LIKE '%special%requests%'`, and
7458    /// decompressing the comments and searching them was most of the orders scan.
7459    ///
7460    /// # Errors
7461    ///
7462    /// If a part, column, page, or checksum is invalid.
7463    pub fn rows_holding(
7464        &self,
7465        part: usize,
7466        column: usize,
7467        sequence: &Sequence,
7468        negated: bool,
7469    ) -> Result<Option<Vec<u32>>> {
7470        let place = *self.places.get(part).ok_or_else(|| invalid("part index out of range"))?;
7471        let field =
7472            self.table.fields.get(column).ok_or_else(|| invalid("column index out of range"))?;
7473        if field.ty != LogicalType::Varchar {
7474            return Ok(None);
7475        }
7476        let rows = place.rows as usize;
7477        self.with_part(place, column, |bytes| {
7478            if bytes.first() != Some(&6) {
7479                return Ok(None);
7480            }
7481            let mut cur = Cursor::new(bytes);
7482            cur.u8()?;
7483            let mask = match cur.u8()? {
7484                0 => None,
7485                1 => return Ok(Some(Vec::new())),
7486                2 => {
7487                    let from = cur.at;
7488                    cur.take(rows.div_ceil(8))?;
7489                    Some(&bytes[from..cur.at])
7490                }
7491                _ => return Err(invalid("page validity tag differs")),
7492            };
7493            // A row whose sketch lacks a bit the pieces need cannot hold them, so only the rest
7494            // are walked. See `grams`.
7495            let needs = sequence.needs();
7496            let first = self.firsts.get(part).copied().unwrap_or_default();
7497            let sketch = self
7498                .text_grams
7499                .get(column)
7500                .and_then(|slot| slot.get_or_init(|| grams::text_grams(self, column)).as_deref())
7501                .and_then(|words| words.get(first..first + rows));
7502            let maybe = |row: usize| sketch.is_none_or(|words| words[row] & needs == needs);
7503            let Some(held) = string::holds_in_where(&bytes[cur.at..], sequence, maybe)? else {
7504                return Ok(None);
7505            };
7506            if held.len() != rows {
7507                return Err(invalid("compressed text page holds the wrong number of rows"));
7508            }
7509            let valid = |row: usize| mask.is_none_or(|mask| mask[row / 8] >> (row % 8) & 1 == 1);
7510            Ok(Some(
7511                (0..rows)
7512                    .filter(|&row| held[row] != negated && valid(row))
7513                    .map(|row| row as u32)
7514                    .collect(),
7515            ))
7516        })
7517    }
7518
7519    /// Whether part and column `bit`, numbered as [`Reader::verified`] numbers them, has matched its
7520    /// checksum since this reader was opened.
7521    fn is_verified(&self, bit: usize) -> bool {
7522        self.verified
7523            .get(bit / 64)
7524            .is_some_and(|word| word.load(Atomic::Relaxed) >> (bit % 64) & 1 == 1)
7525    }
7526
7527    /// Remembers that part and column `bit` matched its checksum.
7528    fn set_verified(&self, bit: usize) {
7529        if let Some(word) = self.verified.get(bit / 64) {
7530            word.fetch_or(1 << (bit % 64), Atomic::Relaxed);
7531        }
7532    }
7533
7534    /// Runs `read` over the stored bytes of one column of one part, out of the stripe's page when
7535    /// it is held and read off the file on their own when it is not.
7536    fn with_part<T>(
7537        &self,
7538        place: Place,
7539        column: usize,
7540        read: impl FnOnce(&[u8]) -> Result<T>,
7541    ) -> Result<T> {
7542        let stripe_index = place.stripe as usize;
7543        let stripe = self
7544            .table
7545            .stripes
7546            .get(stripe_index)
7547            .ok_or_else(|| invalid("stripe index out of range"))?;
7548        let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
7549        let held = self.held(stripe_index, place.part as usize, stripe, column, true)?;
7550        let span = *held
7551            .index
7552            .get(place.part as usize)
7553            .ok_or_else(|| invalid("part index out of range"))?;
7554        match &held.page {
7555            Some(page) => read(page.part(place.part as usize, span)?),
7556            None => {
7557                let offset = page
7558                    .offset
7559                    .checked_add(span.start as u64)
7560                    .ok_or_else(|| invalid("part range overflow"))?;
7561                let mut bytes = vec![0; span.length];
7562                read_at(&self.file, offset, &mut bytes)?;
7563                verify_part(&bytes, span)?;
7564                read(&bytes)
7565            }
7566        }
7567    }
7568
7569    /// Reads named columns from one part, only at the rows `positions` names.
7570    ///
7571    /// For a scan that already knows which rows of the part it keeps, from the columns it read
7572    /// first. A compressed string page decompresses only those rows, and every other page is
7573    /// decoded whole and gathered, which is what reading it and narrowing it costs anyway. With
7574    /// `whole` the stripe's pages are kept the way [`Self::read`] keeps them, and without it they
7575    /// are not, the way [`Self::read_sparse`] does.
7576    ///
7577    /// # Errors
7578    ///
7579    /// If a part, column, page, or checksum is invalid, or the positions do not rise or run past
7580    /// the end of the part.
7581    pub fn read_rows(
7582        &self,
7583        part: usize,
7584        columns: &[usize],
7585        positions: &[u32],
7586        whole: bool,
7587    ) -> Result<Chunk> {
7588        self.read_impl(part, columns, whole, Some(positions))
7589    }
7590
7591    /// Whether an exact global-code membership index proves that the stripe holding a part cannot
7592    /// contain any of the sorted candidate codes.
7593    ///
7594    /// # Errors
7595    ///
7596    /// If the part, column, index page, checksum, or delta stream is invalid.
7597    pub fn skips_codes(&self, part: usize, column: usize, candidates: &[u32]) -> Result<bool> {
7598        // A demoted column's later stripes hold values the dictionary never coded, so no list of
7599        // codes can prove a stripe of it holds none of a value.
7600        if self.demoted(column) {
7601            return Ok(false);
7602        }
7603        if candidates.is_empty() {
7604            return Ok(true);
7605        }
7606        if candidates.windows(2).any(|pair| pair[0] >= pair[1]) {
7607            return Err(Error::internal("native code candidates are not sorted and unique"));
7608        }
7609        let stripe = self.stripe_of(part)?;
7610        let Some(page) = stripe.memberships.get(column) else {
7611            return Ok(false);
7612        };
7613        let mut bytes = vec![0; page.length as usize];
7614        read_at(&self.file, page.offset, &mut bytes)?;
7615        if checksum(&bytes) != page.hash {
7616            return Err(invalid("membership page checksum differs"));
7617        }
7618        let codes = decode_membership(&bytes)?;
7619        let mut left = 0;
7620        let mut right = 0;
7621        while left < codes.len() && right < candidates.len() {
7622            match codes[left].cmp(&candidates[right]) {
7623                Ordering::Less => left += 1,
7624                Ordering::Greater => right += 1,
7625                Ordering::Equal => return Ok(false),
7626            }
7627        }
7628        Ok(true)
7629    }
7630
7631    fn stripe_of(&self, part: usize) -> Result<&Stripe> {
7632        let place = self.places.get(part).ok_or_else(|| invalid("part index out of range"))?;
7633        self.table
7634            .stripes
7635            .get(place.stripe as usize)
7636            .ok_or_else(|| invalid("stripe index out of range"))
7637    }
7638
7639    /// The page index of one column of one stripe, and its page when the caller wants all of it.
7640    ///
7641    /// A scan hands parts out in order, so every worker on a column crosses into a new stripe within
7642    /// a few parts of the others and they all want the same page at the same moment. This used to
7643    /// let all of them read it, which cost the scan as many copies of every page as it had workers.
7644    /// On the full ClickBench file a `MIN(EventDate), MAX(EventDate)` moved 3.2 GB off the disk to
7645    /// look at 400 MB of column.
7646    ///
7647    /// A worker that finds the page it wants already being read neither waits for it nor reads it
7648    /// again. It comes back with the index alone, which sends [`Reader::read_impl`] down the path
7649    /// that reads the one part it came for, a few kilobytes against a quarter of a megabyte, and it
7650    /// picks the page up from the cache on its next part. Waiting would be the other way to avoid
7651    /// the duplicate read and it is worse: the pages that matter are the wide string ones, they take
7652    /// milliseconds to copy even warm, and every other worker would be stopped for all of it.
7653    ///
7654    /// The file is never read under the lock.
7655    fn held(
7656        &self,
7657        at: usize,
7658        part: usize,
7659        stripe: &Stripe,
7660        column: usize,
7661        whole: bool,
7662    ) -> Result<CachedColumn> {
7663        let cache =
7664            self.cache.columns.get(column).ok_or_else(|| invalid("column index out of range"))?;
7665        let mut cached = cache.lock().map_err(|_| invalid("column page cache is poisoned"))?;
7666        if cached.index.is_empty() {
7667            let stripes = self.table.stripes.len();
7668            cached.pages = (0..stripes).map(|_| None).collect();
7669            cached.index = vec![None; stripes];
7670            cached.touched = vec![Vec::new(); stripes];
7671        }
7672        let known = cached.index.get(at).and_then(Clone::clone);
7673        let page = cached.pages.get(at).and_then(Option::as_ref).map(|slot| {
7674            slot.used.store(true, Atomic::Relaxed);
7675            Arc::clone(&slot.page)
7676        });
7677        // Whole only for a part asked for before, see [`Cached`].
7678        let (again, through) = match cached.touched.get_mut(at) {
7679            Some(bits) if whole && page.is_none() => touch(bits, part, stripe.parts.len()),
7680            _ => (false, false),
7681        };
7682        let whole = whole && again;
7683        if let Some(index) = known.clone()
7684            && (!whole || page.is_some())
7685        {
7686            return Ok(CachedColumn { stripe: at, index, page });
7687        }
7688        if cached.loading.contains(&at) {
7689            drop(cached);
7690            // The index is almost always already here, because somebody read this stripe to get
7691            // into the loading list in the first place, so this branch usually costs no read at
7692            // all and the one part read in `read_impl` is all the losing worker pays for.
7693            if let Some(index) = known {
7694                return Ok(CachedColumn { stripe: at, index, page: None });
7695            }
7696            let held = self.page_of(stripe, column, at, false, None)?;
7697            let mut cached = cache.lock().map_err(|_| invalid("column page cache is poisoned"))?;
7698            remember(&mut cached, &held);
7699            return Ok(held);
7700        }
7701        cached.loading.push(at);
7702        drop(cached);
7703
7704        let read = self.page_of(stripe, column, at, whole, known);
7705
7706        // The stripe leaves the loading list and its page enters the cache under one lock. Doing
7707        // them separately would leave a moment where another worker sees neither and reads the
7708        // page a second time, which is the whole thing this is here to stop.
7709        let mut cached = cache.lock().map_err(|_| invalid("column page cache is poisoned"))?;
7710        if let Some(position) = cached.loading.iter().position(|loading| *loading == at) {
7711            cached.loading.remove(position);
7712        }
7713        let held = read?;
7714        let taken = remember(&mut cached, &held);
7715        if taken.is_some() && !through {
7716            let floor = self.cache.kept.load(Atomic::Relaxed).max(1);
7717            cached.passing.push_back(at);
7718            while cached.passing.len() > floor {
7719                let Some(old) = cached.passing.pop_front() else { break };
7720                if let Some(slot) = cached.pages.get_mut(old) {
7721                    *slot = None;
7722                }
7723            }
7724            return Ok(held);
7725        }
7726        drop(cached);
7727        if let Some((bytes, used)) = taken {
7728            self.cache.held[column].fetch_add(1, Atomic::Relaxed);
7729            self.pool.admit(Held {
7730                shelf: Arc::downgrade(&self.cache),
7731                column,
7732                stripe: at,
7733                bytes,
7734                used,
7735            });
7736        }
7737        Ok(held)
7738    }
7739
7740    /// Reads one stripe's index for a column, and its page when the caller wants all of it.
7741    ///
7742    /// `known` is the index when the reader has already read it, which after the first worker
7743    /// through a stripe it always has, because [`remember`] keeps every index for the life of the
7744    /// reader. Without that a scan reads the index again on every part that misses the page cache.
7745    fn page_of(
7746        &self,
7747        stripe: &Stripe,
7748        column: usize,
7749        at: usize,
7750        whole: bool,
7751        known: Option<Arc<Vec<PartSpan>>>,
7752    ) -> Result<CachedColumn> {
7753        let index = match known {
7754            Some(index) => index,
7755            None => {
7756                self.indexes.fetch_add(1, Atomic::Relaxed);
7757                Arc::new(read_index(&self.file, stripe, column)?)
7758            }
7759        };
7760        let page = if whole {
7761            self.pages.fetch_add(1, Atomic::Relaxed);
7762            let span = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
7763            let length = span.length as usize;
7764            let bytes = match &self.map {
7765                Some(map) if map.get(span.offset, length).is_some() => {
7766                    PageBytes::Mapped { map: Arc::clone(map), offset: span.offset, length }
7767                }
7768                _ => {
7769                    let mut bytes = vec![0; length];
7770                    read_at(&self.file, span.offset, &mut bytes)?;
7771                    PageBytes::Read(bytes)
7772                }
7773            };
7774            let checked = index.iter().map(|_| AtomicBool::new(false)).collect();
7775            Some(Arc::new(HeldPage { bytes, checked }))
7776        } else {
7777            None
7778        };
7779        Ok(CachedColumn { stripe: at, index, page })
7780    }
7781
7782    fn read_impl(
7783        &self,
7784        at: usize,
7785        columns: &[usize],
7786        whole: bool,
7787        positions: Option<&[u32]>,
7788    ) -> Result<Chunk> {
7789        let place = *self.places.get(at).ok_or_else(|| invalid("part index out of range"))?;
7790        let index = place.stripe as usize;
7791        let stripe =
7792            self.table.stripes.get(index).ok_or_else(|| invalid("stripe index out of range"))?;
7793        let rows = place.rows as usize;
7794        let mut picked = Vec::with_capacity(columns.len());
7795        for &column in columns {
7796            let field = self
7797                .table
7798                .fields
7799                .get(column)
7800                .ok_or_else(|| invalid("column index out of range"))?;
7801            let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
7802            let held = self.held(index, place.part as usize, stripe, column, whole)?;
7803            let span = *held
7804                .index
7805                .get(place.part as usize)
7806                .ok_or_else(|| invalid("part index out of range"))?;
7807            let owned;
7808            let mut mapped = false;
7809            let bit = at * self.table.fields.len() + column;
7810            let bytes = match &held.page {
7811                Some(held) if self.is_verified(bit) => part_bytes(held.bytes(), span),
7812                Some(held) => {
7813                    held.part(place.part as usize, span).inspect(|_| self.set_verified(bit))
7814                }
7815                None => {
7816                    let offset = page
7817                        .offset
7818                        .checked_add(span.start as u64)
7819                        .ok_or_else(|| invalid("part range overflow"))?;
7820                    let bytes =
7821                        match self.map.as_deref().and_then(|map| map.get(offset, span.length)) {
7822                            Some(bytes) => {
7823                                mapped = true;
7824                                bytes
7825                            }
7826                            None => {
7827                                let mut bytes = vec![0; span.length];
7828                                read_at(&self.file, offset, &mut bytes)?;
7829                                owned = bytes;
7830                                owned.as_slice()
7831                            }
7832                        };
7833                    if self.is_verified(bit) {
7834                        Ok(bytes)
7835                    } else {
7836                        verify_part(bytes, span).map(|()| {
7837                            self.set_verified(bit);
7838                            bytes
7839                        })
7840                    }
7841                }
7842            }
7843            .map_err(|error| {
7844                invalid(&format!(
7845                    "{}, column {column} part {} of the page at {}",
7846                    error.message(),
7847                    place.part,
7848                    page.offset,
7849                ))
7850            })?;
7851            let dictionary = self.dictionary(column)?;
7852            // Held as a page, because a column that came out of a file is handed out more than
7853            // once. A group by clones its key columns out of the chunk so the keys outlive it, a
7854            // projection of a bare column name does the same, and a cut of a flat run copies unless
7855            // the run is a page. One `Arc` per column per part buys all of those, and it moves the
7856            // run into the `Arc` without touching a value.
7857            let mut vector = match positions {
7858                None => decode(&field.ty, rows, bytes, dictionary)?,
7859                Some(positions) => decode_at(&field.ty, rows, bytes, dictionary, positions)?,
7860            };
7861            // What was decoded is in memory of its own now, so once every part of the page has
7862            // been, nothing this scan does reads the page again. A part read twice counts twice
7863            // and lets the page go early, which costs the reads after it a fault and no more.
7864            if mapped
7865                && let Some(map) = self.map.as_deref()
7866                && let Some(left) = self.unreleased.get(index * self.table.fields.len() + column)
7867                && left.fetch_update(Atomic::Relaxed, Atomic::Relaxed, |left| left.checked_sub(1))
7868                    == Ok(1)
7869            {
7870                map.release(page.offset, page.length as usize);
7871            }
7872            // A demoted column's codes are not the column's codes, only the codes of the stripes
7873            // written before the demotion, so they are not handed out as if they were. See
7874            // [`DEMOTED`].
7875            if self.demoted(column) && vector.stable_dictionary_parts().is_some() {
7876                vector = vector.flatten()?;
7877            }
7878            picked.push(vector.into_pages());
7879        }
7880        Chunk::with_rows(picked, positions.map_or(rows, <[u32]>::len))
7881    }
7882
7883    /// Whether persisted statistics prove that a part cannot match the predicates.
7884    ///
7885    /// Three of them, asked cheapest first.
7886    ///
7887    /// The stripe's bounds are in memory already, so they are free, and they are also the coarsest:
7888    /// every part of a stripe gets the same answer and a scan that skips one part that way skips all
7889    /// sixty four. Then the part's own bounds, which are a read of one page per column per stripe
7890    /// and are sixty four times finer. Then the sieves, which are per part and answer equality, the
7891    /// test bounds are worst at: a column of identifiers has every stripe and nearly every part
7892    /// covering the whole of its type, so bounds keep them all and the sieve keeps the ones that
7893    /// really hold the value.
7894    ///
7895    /// The middle one is what an ordered comparison on a column the rows are not sorted by needs. On
7896    /// ClickBench 24 the stripe bounds leave eight stripes of sixteen alive, which is half the file,
7897    /// and the part bounds leave thirty parts of nine hundred and seventy four.
7898    #[must_use]
7899    pub fn skips(&self, part: usize, probes: &[Probe]) -> bool {
7900        let Some(place) = self.places.get(part).copied() else { return false };
7901        let Some(stripe) = self.table.stripes.get(place.stripe as usize) else { return false };
7902        if stripe.zone.skips(probes) {
7903            return true;
7904        }
7905        probes.iter().any(|probe| self.outside(place, probe) || self.sifted(place, probe))
7906    }
7907
7908    /// Whether `rule` rules out a part from what the stored range of one column says about it.
7909    ///
7910    /// The same two steps as [`Self::skips`] without the sieve, for a test no [`Probe`] can write.
7911    /// A probe is one comparison against one constant, and the keys a join's build side holds are a
7912    /// set, which rules a part out when none of them falls inside the part's two ends. Handing the
7913    /// range to the caller is what lets the set stay with the join that knows what it is.
7914    #[must_use]
7915    pub fn ruled_by(&self, part: usize, column: usize, rule: impl Fn(&Range) -> bool) -> bool {
7916        let Some(place) = self.places.get(part).copied() else { return false };
7917        let Some(stripe) = self.table.stripes.get(place.stripe as usize) else { return false };
7918        if stripe.zone.column(column).is_some_and(&rule) {
7919            return true;
7920        }
7921        self.stripe_part_ranges(place.stripe as usize, column)
7922            .and_then(|ranges| ranges.get(place.part as usize))
7923            .is_some_and(rule)
7924    }
7925
7926    /// The stored range of one column over one part, the part's own where its stripe kept one and
7927    /// the stripe's where it did not, which is wider but still holds every row of the part.
7928    #[must_use]
7929    pub fn part_range(&self, part: usize, column: usize) -> Option<Range> {
7930        let place = self.places.get(part).copied()?;
7931        let own = self
7932            .stripe_part_ranges(place.stripe as usize, column)
7933            .and_then(|ranges| ranges.get(place.part as usize));
7934        own.or_else(|| self.table.stripes.get(place.stripe as usize)?.zone.column(column)).cloned()
7935    }
7936
7937    /// The half of [`Self::ruled_by`] that reads nothing, asked about a whole stripe.
7938    #[must_use]
7939    pub fn stripe_ruled_by(
7940        &self,
7941        stripe: usize,
7942        column: usize,
7943        rule: impl Fn(&Range) -> bool,
7944    ) -> bool {
7945        self.table.stripes.get(stripe).and_then(|held| held.zone.column(column)).is_some_and(rule)
7946    }
7947
7948    /// Whether the bounds of one part rule out one probe.
7949    ///
7950    /// The part's own two ends, which are narrower than the stripe's and cost a page read the first
7951    /// time this is asked about a column. A column with no page here answers `false`, which is the
7952    /// answer a caller got before there were any.
7953    fn outside(&self, place: Place, probe: &Probe) -> bool {
7954        match self.stripe_part_ranges(place.stripe as usize, probe.column) {
7955            Some(ranges) => ranges
7956                .get(place.part as usize)
7957                .is_some_and(|range| range.excludes(probe.op, &probe.value)),
7958            None => false,
7959        }
7960    }
7961
7962    /// The per part ranges of one stripe of one column, read once and kept.
7963    ///
7964    /// `None` when the column has no page in that stripe and when the page is damaged, on the same
7965    /// reasoning as the sieves: this is an index over data that is still there, so a caller that
7966    /// cannot read one reads the rows and gets the right answer slowly.
7967    fn stripe_part_ranges(&self, stripe: usize, column: usize) -> Option<&[Range]> {
7968        let slot = self
7969            .part_ranges
7970            .get(column)?
7971            .get_or_init(|| self.table.stripes.iter().map(|_| OnceLock::new()).collect())
7972            .get(stripe)?;
7973        if let Some(held) = slot.get() {
7974            return Some(held);
7975        }
7976        let page = self.table.stripes.get(stripe)?.part_ranges.get(column)?;
7977        let mut bytes = vec![0; page.length as usize];
7978        read_at(&self.file, page.offset, &mut bytes).ok()?;
7979        if checksum(&bytes) != page.hash {
7980            return None;
7981        }
7982        let ranges = Arc::new(decode_part_ranges(&bytes).ok()?);
7983        let _ = slot.set(ranges);
7984        slot.get().map(|held| held.as_slice())
7985    }
7986
7987    /// Whether persisted statistics prove that every row of a part matches the predicates.
7988    ///
7989    /// Only the bounds. The sieves say nothing here, because a sieve that holds a value is a sieve
7990    /// that may be holding somebody else's hash, so it can rule a part out and can never wave one
7991    /// through.
7992    ///
7993    /// The stripe first and the part after it, the same two steps and in the same order as
7994    /// [`Self::skips`]. The stripe's bounds are in memory already and its null count covers sixty
7995    /// four parts rather than one, so a stripe that answers is an answer for nothing, and the part's
7996    /// own bounds are only read for the probes it could not settle. Both directions are safe: a
7997    /// stretch where everything passes contains no narrower stretch where something fails, and a
7998    /// stripe with no nulls has no nulls in any of its parts.
7999    ///
8000    /// A string end a part recorded is cut down to its first few bytes, so a part's stretch can be
8001    /// wider than its rows really are as well. That is the same safe direction for the same reason,
8002    /// and it is why this asks the two ends rather than anything `exact` says.
8003    #[must_use]
8004    pub fn certain(&self, part: usize, probes: &[Probe]) -> bool {
8005        let Some(place) = self.places.get(part).copied() else { return false };
8006        let Some(stripe) = self.table.stripes.get(place.stripe as usize) else { return false };
8007        if stripe.zone.certain(probes) {
8008            return true;
8009        }
8010        probes
8011            .iter()
8012            .all(|probe| stripe.zone.certain(slice::from_ref(probe)) || self.inside(place, probe))
8013    }
8014
8015    /// Whether one part's own two ends prove that every row of it passes `probe`.
8016    ///
8017    /// The mirror of [`Self::outside`], reading the same page. `false` for a part whose stripe wrote
8018    /// no range page, which is a stripe of one part, because there the stripe's own bounds are the
8019    /// part's and the caller has already asked them.
8020    fn inside(&self, place: Place, probe: &Probe) -> bool {
8021        match self.stripe_part_ranges(place.stripe as usize, probe.column) {
8022            Some(ranges) => ranges
8023                .get(place.part as usize)
8024                .is_some_and(|range| range.certain(probe.op, &probe.value)),
8025            None => false,
8026        }
8027    }
8028
8029    /// Whether the bounds of one stripe prove that none of its parts can match the predicates.
8030    ///
8031    /// The cheap half of [`Self::skips`], asked about a whole stripe at once. The bounds live in the
8032    /// directory and are already in memory, so this answers without touching the file, and that is
8033    /// the reason it is worth having on its own: a caller that wants to know roughly where the work
8034    /// is before it starts any workers can ask this about sixteen stripes for nothing, where asking
8035    /// [`Self::skips`] about nine hundred parts would read and decode a sieve page per stripe first.
8036    ///
8037    /// It keeps stripes that [`Self::skips`] would rule out part by part, which is the right way for
8038    /// it to be wrong: the parts are still checked when they are read.
8039    #[must_use]
8040    pub fn stripe_skips(&self, stripe: usize, probes: &[Probe]) -> bool {
8041        self.table.stripes.get(stripe).is_some_and(|held| held.zone.skips(probes))
8042    }
8043
8044    /// Whether the sieve of one part rules out one probe.
8045    ///
8046    /// Only equality. An ordered comparison is what the bounds are for and a sieve says nothing
8047    /// about it, and a read that cannot answer keeps the part, which is the answer a caller with no
8048    /// sieve gets anyway.
8049    fn sifted(&self, place: Place, probe: &Probe) -> bool {
8050        if probe.op != Op::Equal {
8051            return false;
8052        }
8053        match self.stripe_sieves(place.stripe as usize, probe.column) {
8054            Some(sieves) => sieves
8055                .get(place.part as usize)
8056                .and_then(Option::as_ref)
8057                .is_some_and(|sieve| sieve.excludes(&probe.value)),
8058            None => false,
8059        }
8060    }
8061
8062    /// The sieves of one stripe of one column, read once and kept.
8063    ///
8064    /// `None` when the column has no sieves in that stripe, when the page is damaged, and when the
8065    /// bytes are not a page this version can read. A sieve is an index over data that is still there
8066    /// and a caller that cannot read one reads the rows, so this is the one place in the file where
8067    /// a bad checksum is a slow query rather than an error.
8068    fn stripe_sieves(&self, stripe: usize, column: usize) -> Option<&[Option<Sieve>]> {
8069        let slot = self
8070            .sieves
8071            .get(column)?
8072            .get_or_init(|| self.table.stripes.iter().map(|_| OnceLock::new()).collect())
8073            .get(stripe)?;
8074        if let Some(held) = slot.get() {
8075            return Some(held);
8076        }
8077        let page = self.table.stripes.get(stripe)?.sieves.get(column)?;
8078        let mut bytes = vec![0; page.length as usize];
8079        read_at(&self.file, page.offset, &mut bytes).ok()?;
8080        if checksum(&bytes) != page.hash {
8081            return None;
8082        }
8083        let sieves = Arc::new(decode_sieves(&bytes).ok()?);
8084        let _ = slot.set(sieves);
8085        slot.get().map(|held| held.as_slice())
8086    }
8087}
8088
8089/// The value sitting at one position of a dictionary's sorted order.
8090fn text_at_rank(dictionary: &Vector, rank: usize) -> Result<Value> {
8091    let code = dictionary.code_at_rank(rank)? as usize;
8092    if dictionary.logical_type() == &LogicalType::Blob {
8093        let bytes = dictionary
8094            .try_bytes_at(code)?
8095            .ok_or_else(|| invalid("global dictionary order names a code it does not have"))?;
8096        return Ok(Value::Blob(bytes.to_vec()));
8097    }
8098    let text = dictionary
8099        .try_text_at(code)?
8100        .ok_or_else(|| invalid("global dictionary order names a code it does not have"))?;
8101    Ok(Value::Varchar(text.into()))
8102}
8103
8104/// Reads one span of a file at an offset, without moving a cursor anybody else can see.
8105///
8106/// Every reader of a table shares one [`File`] behind an [`Arc`], and a grouped aggregate reads its
8107/// pages from several threads at once, so this has to be positional. Seeking and then reading is
8108/// two calls with a gap in the middle, and in that gap another thread's seek lands and the read
8109/// comes back with somebody else's bytes.
8110///
8111/// The writer reads back through here too, out of the `rudb_io` file it writes through, which is
8112/// why this takes anything [`Positional`] rather than a [`File`].
8113fn read_at<F: Positional + ?Sized>(file: &F, offset: u64, bytes: &mut [u8]) -> Result<()> {
8114    file.fill_at(offset, bytes)
8115}
8116
8117/// Something a span of bytes can be read out of by offset.
8118///
8119/// There are two of these. The reader holds a `std::fs::File`, because it shares it between its
8120/// threads behind an [`Arc`] and every read it makes is on the hot path of a scan. The writer holds
8121/// an `rudb_io::File`, because everything it does to the file has to be something the simulated
8122/// filesystem can stop and crash. The few helpers both of them use, [`read_index`] and the choice
8123/// of committed slot, are written once over this rather than once for each.
8124trait Positional {
8125    /// Fills `bytes` from `offset`, or fails if the file ends first.
8126    ///
8127    /// Both kinds can come back short, so both loop. A read of zero bytes before the span is filled
8128    /// means the file stops earlier than the directory said it does.
8129    fn fill_at(&self, offset: u64, bytes: &mut [u8]) -> Result<()>;
8130}
8131
8132impl<T: Positional + ?Sized> Positional for &T {
8133    fn fill_at(&self, offset: u64, bytes: &mut [u8]) -> Result<()> {
8134        (**self).fill_at(offset, bytes)
8135    }
8136}
8137
8138impl<T: Positional + ?Sized> Positional for Arc<T> {
8139    fn fill_at(&self, offset: u64, bytes: &mut [u8]) -> Result<()> {
8140        (**self).fill_at(offset, bytes)
8141    }
8142}
8143
8144impl<T: Positional + ?Sized> Positional for Box<T> {
8145    fn fill_at(&self, offset: u64, bytes: &mut [u8]) -> Result<()> {
8146        (**self).fill_at(offset, bytes)
8147    }
8148}
8149
8150impl Positional for dyn rudb_io::File + '_ {
8151    fn fill_at(&self, mut offset: u64, mut bytes: &mut [u8]) -> Result<()> {
8152        while !bytes.is_empty() {
8153            let read = self.read_at(offset, bytes)?;
8154            if read == 0 {
8155                return Err(invalid("column page ends before its declared length"));
8156            }
8157            offset += read as u64;
8158            bytes = &mut bytes[read..];
8159        }
8160        Ok(())
8161    }
8162}
8163
8164impl Positional for File {
8165    #[cfg(unix)]
8166    fn fill_at(&self, mut offset: u64, mut bytes: &mut [u8]) -> Result<()> {
8167        use std::os::unix::fs::FileExt;
8168        while !bytes.is_empty() {
8169            let read = self.read_at(bytes, offset).map_err(io)?;
8170            if read == 0 {
8171                return Err(invalid("column page ends before its declared length"));
8172            }
8173            offset += read as u64;
8174            bytes = &mut bytes[read..];
8175        }
8176        Ok(())
8177    }
8178
8179    /// The same read, on the call Windows spells differently.
8180    ///
8181    /// `seek_read` is one `ReadFile` carrying the offset with it, so two of them cannot interleave
8182    /// the way a seek and a read can. It does leave the shared cursor somewhere afterwards, which is
8183    /// why nothing in this file may read that cursor.
8184    #[cfg(windows)]
8185    fn fill_at(&self, mut offset: u64, mut bytes: &mut [u8]) -> Result<()> {
8186        use std::os::windows::fs::FileExt;
8187        while !bytes.is_empty() {
8188            let read = self.seek_read(bytes, offset).map_err(io)?;
8189            if read == 0 {
8190                return Err(invalid("column page ends before its declared length"));
8191            }
8192            offset += read as u64;
8193            bytes = &mut bytes[read..];
8194        }
8195        Ok(())
8196    }
8197
8198    /// Somewhere that is neither, where the cursor is all there is.
8199    ///
8200    /// This one does race, and there is no way to write it so it does not. Nothing we build for
8201    /// runs here, so it exists to keep the crate compiling rather than to be correct under threads.
8202    #[cfg(not(any(unix, windows)))]
8203    fn fill_at(&self, offset: u64, bytes: &mut [u8]) -> Result<()> {
8204        use std::io::{Read, Seek, SeekFrom};
8205        let mut file = self.try_clone().map_err(io)?;
8206        file.seek(SeekFrom::Start(offset)).map_err(io)?;
8207        file.read_exact(bytes).map_err(io)
8208    }
8209}
8210
8211/// Overwrites one span of a file in place, which is how the tests damage a file on purpose.
8212///
8213/// The writer does not come through here. It writes through `rudb_io`, and this is a
8214/// `std::fs::File` opened by a test beside it.
8215#[cfg(test)]
8216fn write_at(file: &File, offset: u64, bytes: &[u8]) -> Result<()> {
8217    use std::io::{Seek, SeekFrom, Write};
8218    let mut file = file;
8219    file.seek(SeekFrom::Start(offset)).map_err(io)?;
8220    file.write_all(bytes).map_err(io)
8221}
8222
8223/// What a column type is called in the directory.
8224///
8225/// A tag is a number in a file somebody else wrote, so a tag that has been used is used forever and
8226/// the only thing that may happen to this list is that it grows. 1 to 13 are the tags the format
8227/// had when it could store thirteen types, and 14 to 27 are the rest, in the order they were added
8228/// rather than in an order that means anything.
8229fn type_tag(ty: &LogicalType) -> Result<u8> {
8230    match ty {
8231        LogicalType::SmallInt => Ok(1),
8232        LogicalType::Integer => Ok(2),
8233        LogicalType::BigInt => Ok(3),
8234        LogicalType::Varchar => Ok(4),
8235        LogicalType::Date => Ok(5),
8236        LogicalType::Timestamp => Ok(6),
8237        LogicalType::Boolean => Ok(7),
8238        LogicalType::TinyInt => Ok(8),
8239        LogicalType::UTinyInt => Ok(9),
8240        LogicalType::USmallInt => Ok(10),
8241        LogicalType::UInteger => Ok(11),
8242        LogicalType::UBigInt => Ok(12),
8243        LogicalType::Decimal { .. } => Ok(13),
8244        LogicalType::Float => Ok(14),
8245        LogicalType::Double => Ok(15),
8246        LogicalType::HugeInt => Ok(16),
8247        LogicalType::UHugeInt => Ok(17),
8248        LogicalType::Time => Ok(18),
8249        LogicalType::TimeTz => Ok(19),
8250        LogicalType::TimestampTz => Ok(20),
8251        LogicalType::Interval => Ok(21),
8252        LogicalType::Uuid => Ok(22),
8253        LogicalType::Blob => Ok(23),
8254        LogicalType::Bit => Ok(24),
8255        LogicalType::TimestampS => Ok(25),
8256        LogicalType::TimestampMs => Ok(26),
8257        LogicalType::TimestampNs => Ok(27),
8258        _ => Err(Error::not_implemented(format!("native storage for {ty}"))),
8259    }
8260}
8261
8262/// The tag of a column type, and the parameters of the ones that have any.
8263///
8264/// Only `DECIMAL` has parameters today. Width and scale go after the tag rather than into it
8265/// because they are what says how wide a value is on disk, and a reader that guessed would read the
8266/// wrong number of bytes per row rather than the wrong number of digits.
8267fn put_type(out: &mut Vec<u8>, ty: &LogicalType) -> Result<()> {
8268    out.push(type_tag(ty)?);
8269    if let LogicalType::Decimal { width, scale } = ty {
8270        out.push(*width);
8271        out.push(*scale);
8272    }
8273    Ok(())
8274}
8275
8276/// The other half of [`put_type`], reading the parameters the tag says are there.
8277fn read_type(cur: &mut Cursor<'_>) -> Result<LogicalType> {
8278    let tag = cur.u8()?;
8279    if tag == 13 {
8280        let width = cur.u8()?;
8281        let scale = cur.u8()?;
8282        return LogicalType::decimal(width, scale)
8283            .map_err(|_| invalid("decimal column width and scale are not a decimal"));
8284    }
8285    tag_type(tag)
8286}
8287
8288fn tag_type(tag: u8) -> Result<LogicalType> {
8289    match tag {
8290        1 => Ok(LogicalType::SmallInt),
8291        2 => Ok(LogicalType::Integer),
8292        3 => Ok(LogicalType::BigInt),
8293        4 => Ok(LogicalType::Varchar),
8294        5 => Ok(LogicalType::Date),
8295        6 => Ok(LogicalType::Timestamp),
8296        7 => Ok(LogicalType::Boolean),
8297        8 => Ok(LogicalType::TinyInt),
8298        9 => Ok(LogicalType::UTinyInt),
8299        10 => Ok(LogicalType::USmallInt),
8300        11 => Ok(LogicalType::UInteger),
8301        12 => Ok(LogicalType::UBigInt),
8302        14 => Ok(LogicalType::Float),
8303        15 => Ok(LogicalType::Double),
8304        16 => Ok(LogicalType::HugeInt),
8305        17 => Ok(LogicalType::UHugeInt),
8306        18 => Ok(LogicalType::Time),
8307        19 => Ok(LogicalType::TimeTz),
8308        20 => Ok(LogicalType::TimestampTz),
8309        21 => Ok(LogicalType::Interval),
8310        22 => Ok(LogicalType::Uuid),
8311        23 => Ok(LogicalType::Blob),
8312        24 => Ok(LogicalType::Bit),
8313        25 => Ok(LogicalType::TimestampS),
8314        26 => Ok(LogicalType::TimestampMs),
8315        27 => Ok(LogicalType::TimestampNs),
8316        _ => Err(invalid("column type tag is unknown")),
8317    }
8318}
8319
8320fn put_u16(out: &mut Vec<u8>, value: u16) {
8321    out.extend_from_slice(&value.to_le_bytes());
8322}
8323fn put_u32(out: &mut Vec<u8>, value: u32) {
8324    out.extend_from_slice(&value.to_le_bytes());
8325}
8326fn put_u64(out: &mut Vec<u8>, value: u64) {
8327    out.extend_from_slice(&value.to_le_bytes());
8328}
8329fn put_var_u64(out: &mut Vec<u8>, mut value: u64) {
8330    while value >= 0x80 {
8331        out.push((value as u8 & 0x7f) | 0x80);
8332        value >>= 7;
8333    }
8334    out.push(value as u8);
8335}
8336
8337fn frequency_order(left: FrequencyValue, right: FrequencyValue) -> Ordering {
8338    match (left, right) {
8339        (FrequencyValue::Null, FrequencyValue::Null) => Ordering::Equal,
8340        (FrequencyValue::Null, _) => Ordering::Less,
8341        (_, FrequencyValue::Null) => Ordering::Greater,
8342        (FrequencyValue::Integer(left), FrequencyValue::Integer(right)) => left.cmp(&right),
8343        (FrequencyValue::Code(left), FrequencyValue::Code(right)) => left.cmp(&right),
8344        (FrequencyValue::Integer(_), FrequencyValue::Code(_)) => Ordering::Less,
8345        (FrequencyValue::Code(_), FrequencyValue::Integer(_)) => Ordering::Greater,
8346    }
8347}
8348
8349/// Leaves the [`FREQUENCY_ENTRIES`] commonest entries in order and says what the next one counted.
8350///
8351/// There is one entry a distinct value, so on `URL` this is handed two and a quarter million of
8352/// them and keeps five hundred and twelve. Sorting all of them to throw almost all of them away is
8353/// the whole of what counting a dictionary column used to cost, 2.13 seconds of it on `URL` at eight
8354/// million rows against 11.93 for compressing the same column's values.
8355///
8356/// Partitioning answers both questions instead. It puts the five hundred and thirteenth entry where
8357/// it belongs and everything commoner in front of it, which is the entries to keep and the count to
8358/// report as the largest one omitted, and then only the part that survives is sorted. The order that
8359/// comes out is the order the sort gave, because the tie break makes the comparison total: two
8360/// entries never hold the same value.
8361fn keep_most_frequent(entries: &mut Vec<FrequencyEntry>) -> u64 {
8362    let order = |left: &FrequencyEntry, right: &FrequencyEntry| {
8363        right.count.cmp(&left.count).then_with(|| frequency_order(left.value, right.value))
8364    };
8365    let omitted_max = if entries.len() > FREQUENCY_ENTRIES {
8366        let (_, next, _) = entries.select_nth_unstable_by(FREQUENCY_ENTRIES, order);
8367        let omitted_max = next.count;
8368        entries.truncate(FREQUENCY_ENTRIES);
8369        // The summary lives until the table is written, and what it was cut down from can be
8370        // millions of entries long.
8371        entries.shrink_to_fit();
8372        omitted_max
8373    } else {
8374        0
8375    };
8376    entries.sort_unstable_by(order);
8377    omitted_max
8378}
8379
8380fn code_frequency(
8381    dictionary: &GlobalDictionary,
8382    flat: &[u8],
8383    bases: &[u64],
8384) -> Result<(FrequencySummary, Vec<Option<Vec<u8>>>)> {
8385    // Every distinct value is a candidate and only [`FREQUENCY_ENTRIES`] of them are kept, so the
8386    // candidates are a count and a code rather than a whole entry each, which is a third of the
8387    // size. On the 10 million row `hits` load the entries of `URL` and `Referer` were about 200 MB
8388    // each at the moment they were cut down, and the close ran both at once.
8389    let seen = dictionary.counts.iter().filter(|count| **count != 0).count();
8390    let mut candidates = Vec::with_capacity(seen + usize::from(dictionary.nulls != 0));
8391    candidates.extend(
8392        dictionary
8393            .counts
8394            .iter()
8395            .enumerate()
8396            .filter(|(_, count)| **count != 0)
8397            .map(|(code, &count)| (count, Some(code as u32))),
8398    );
8399    if dictionary.nulls != 0 {
8400        candidates.push((dictionary.nulls, None));
8401    }
8402    // The order of `keep_most_frequent`, where a null sorts before any code as `None` does.
8403    let order = |left: &(u64, Option<u32>), right: &(u64, Option<u32>)| {
8404        right.0.cmp(&left.0).then_with(|| left.1.cmp(&right.1))
8405    };
8406    let omitted_max = if candidates.len() > FREQUENCY_ENTRIES {
8407        let (_, next, _) = candidates.select_nth_unstable_by(FREQUENCY_ENTRIES, order);
8408        let omitted_max = next.0;
8409        candidates.truncate(FREQUENCY_ENTRIES);
8410        omitted_max
8411    } else {
8412        0
8413    };
8414    candidates.sort_unstable_by(order);
8415    let entries = candidates
8416        .into_iter()
8417        .map(|(count, code)| FrequencyEntry {
8418            value: code.map_or(FrequencyValue::Null, FrequencyValue::Code),
8419            count,
8420        })
8421        .collect::<Vec<_>>();
8422    let mut spans = Vec::with_capacity(entries.len());
8423    let mut text_bytes = 0_usize;
8424    for entry in &entries {
8425        let span = match entry.value {
8426            FrequencyValue::Code(code) => {
8427                let span = GlobalDictionary::value_span(&dictionary.ends, bases, code as usize);
8428                let bytes = flat
8429                    .get(span.0..span.1)
8430                    .ok_or_else(|| invalid("a frequency code is outside its dictionary"))?;
8431                text_bytes = text_bytes.saturating_add(bytes.len());
8432                Some(span)
8433            }
8434            FrequencyValue::Null | FrequencyValue::Integer(_) => None,
8435        };
8436        spans.push(span);
8437    }
8438    let texts = if text_bytes > FREQUENCY_TEXT_BUDGET {
8439        Vec::new()
8440    } else {
8441        spans.into_iter().map(|span| span.map(|(from, to)| flat[from..to].to_vec())).collect()
8442    };
8443    Ok((
8444        FrequencySummary {
8445            entries,
8446            omitted_max,
8447            ordinals: Vec::new(),
8448            ordinal_entries: Vec::new(),
8449            ordinal_bound: 0,
8450        },
8451        texts,
8452    ))
8453}
8454
8455fn encode_directory(table: &Table) -> Result<Vec<u8>> {
8456    let mut out = DIRECTORY.to_vec();
8457    let name = table.name.as_bytes();
8458    put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("table name too long"))?);
8459    out.extend_from_slice(name);
8460    put_u16(&mut out, u16::try_from(table.fields.len()).map_err(|_| invalid("too many columns"))?);
8461    for field in &table.fields {
8462        let name = field.name.as_bytes();
8463        put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("column name too long"))?);
8464        out.extend_from_slice(name);
8465        put_type(&mut out, &field.ty)?;
8466        out.push(u8::from(field.not_null));
8467    }
8468    for (field, dictionary) in table.fields.iter().zip(&table.dictionaries) {
8469        match dictionary {
8470            None => out.push(0),
8471            Some(page) => {
8472                out.push(dictionary_tag(&field.ty));
8473                put_u64(&mut out, page.offset);
8474                put_u32(&mut out, page.length);
8475                put_u64(&mut out, page.hash);
8476            }
8477        }
8478    }
8479    for distinct in &table.distincts {
8480        match distinct {
8481            None => out.push(0),
8482            Some(count) => {
8483                out.push(1);
8484                put_u64(&mut out, *count);
8485            }
8486        }
8487    }
8488    put_u64(&mut out, u64::try_from(table.rows).map_err(|_| invalid("row count overflow"))?);
8489    put_u32(&mut out, u32::try_from(table.stripes.len()).map_err(|_| invalid("too many stripes"))?);
8490    for stripe in &table.stripes {
8491        put_u32(
8492            &mut out,
8493            u32::try_from(stripe.parts.len()).map_err(|_| invalid("too many parts in a stripe"))?,
8494        );
8495        for &rows in &stripe.parts {
8496            put_u32(&mut out, rows);
8497        }
8498        put_u64(&mut out, stripe.index.offset);
8499        put_u32(&mut out, stripe.index.length);
8500        for page in &stripe.pages {
8501            put_u64(&mut out, page.offset);
8502            put_u32(&mut out, page.length);
8503        }
8504        // A membership index says which of a dictionary's codes a part holds, so a column the writer
8505        // decided against giving a dictionary has nothing for it to be about and writes none. Every
8506        // file written before that decision existed has a dictionary on every varchar column, so
8507        // this reads those files byte for byte the way it always did.
8508        for (column, ((field, dictionary), membership)) in
8509            table.fields.iter().zip(&table.dictionaries).zip(stripe.memberships.slots()).enumerate()
8510        {
8511            if !coded_type(&field.ty) || dictionary.is_none() {
8512                continue;
8513            }
8514            let page = match membership {
8515                Some(page) => page,
8516                None if table.demoted.get(column).copied().unwrap_or(false) => {
8517                    Page { offset: HEADER, length: 0, hash: 0 }
8518                }
8519                None => return Err(invalid("string page has no code membership index")),
8520            };
8521            put_u64(&mut out, page.offset);
8522            put_u32(&mut out, page.length);
8523            put_u64(&mut out, page.hash);
8524        }
8525        for sieve in stripe.sieves.slots() {
8526            match sieve {
8527                None => out.push(0),
8528                Some(page) => {
8529                    out.push(1);
8530                    put_u64(&mut out, page.offset);
8531                    put_u32(&mut out, page.length);
8532                    put_u64(&mut out, page.hash);
8533                }
8534            }
8535        }
8536        for held in stripe.part_ranges.slots() {
8537            match held {
8538                None => out.push(0),
8539                Some(page) => {
8540                    out.push(1);
8541                    put_u64(&mut out, page.offset);
8542                    put_u32(&mut out, page.length);
8543                    put_u64(&mut out, page.hash);
8544                }
8545            }
8546        }
8547        for range in stripe.zone.columns() {
8548            put_bound(&mut out, range.low.as_ref())?;
8549            put_bound(&mut out, range.high.as_ref())?;
8550            put_u32(
8551                &mut out,
8552                u32::try_from(range.nulls).map_err(|_| invalid("null count overflow"))?,
8553            );
8554            out.push(u8::from(range.exact));
8555            match range.sum {
8556                None => out.push(0),
8557                Some(total) => {
8558                    out.push(1);
8559                    out.extend_from_slice(&total.to_le_bytes());
8560                }
8561            }
8562        }
8563    }
8564    out.extend_from_slice(FREQUENCIES_SPANS);
8565    put_u16(
8566        &mut out,
8567        u16::try_from(table.frequencies.len())
8568            .map_err(|_| invalid("too many frequency columns"))?,
8569    );
8570    for summary in &table.frequencies {
8571        let summary = match summary {
8572            None => {
8573                put_u32(&mut out, 0);
8574                put_u32(&mut out, 0);
8575                continue;
8576            }
8577            Some(Frequencies::Held(summary)) => summary,
8578            // Only a reader leaves a synopsis in the file, and nothing writes a reader's table back.
8579            Some(Frequencies::Stored { .. }) => {
8580                return Err(invalid("a synopsis left in the file cannot be written back"));
8581            }
8582        };
8583        let length_at = out.len();
8584        put_u32(&mut out, 0);
8585        put_u32(
8586            &mut out,
8587            u32::try_from(summary.entries.len())
8588                .map_err(|_| invalid("too many frequency entries"))?,
8589        );
8590        let start = out.len();
8591        out.push(1);
8592        put_u64(&mut out, summary.omitted_max);
8593        put_u32(
8594            &mut out,
8595            u32::try_from(summary.entries.len())
8596                .map_err(|_| invalid("too many frequency entries"))?,
8597        );
8598        for entry in &summary.entries {
8599            match entry.value {
8600                FrequencyValue::Null => out.push(0),
8601                FrequencyValue::Integer(value) => {
8602                    out.push(1);
8603                    out.extend_from_slice(&value.to_le_bytes());
8604                }
8605                FrequencyValue::Code(value) => {
8606                    out.push(2);
8607                    put_u32(&mut out, value);
8608                }
8609            }
8610            put_u64(&mut out, entry.count);
8611        }
8612        put_u32(
8613            &mut out,
8614            u32::try_from(summary.ordinals.len())
8615                .map_err(|_| invalid("too many frequency ordinals"))?,
8616        );
8617        let mut previous = 0_u64;
8618        for (at, &ordinal) in summary.ordinals.iter().enumerate() {
8619            let delta = if at == 0 {
8620                ordinal
8621            } else {
8622                ordinal
8623                    .checked_sub(previous)
8624                    .ok_or_else(|| invalid("frequency ordinals are not ordered"))?
8625            };
8626            if at != 0 && delta == 0 {
8627                return Err(invalid("frequency ordinals are not unique"));
8628            }
8629            put_var_u64(&mut out, delta);
8630            previous = ordinal;
8631        }
8632        if summary.ordinal_entries.len() != summary.ordinals.len() {
8633            return Err(invalid("frequency ordinal values have a different length"));
8634        }
8635        for &entry in &summary.ordinal_entries {
8636            if entry as usize >= summary.entries.len() {
8637                return Err(invalid("frequency ordinal value is outside its entries"));
8638            }
8639            put_u16(&mut out, entry);
8640        }
8641        let length = u32::try_from(out.len() - start)
8642            .map_err(|_| invalid("a frequency synopsis is too long"))?;
8643        out[length_at..length_at + 4].copy_from_slice(&length.to_le_bytes());
8644    }
8645    let bounds = table
8646        .frequencies
8647        .iter()
8648        .enumerate()
8649        .filter_map(|(column, summary)| match summary {
8650            Some(Frequencies::Held(summary)) if summary.ordinal_bound != 0 => {
8651                Some((column, summary.ordinal_bound))
8652            }
8653            _ => None,
8654        })
8655        .collect::<Vec<_>>();
8656    if !bounds.is_empty() {
8657        out.extend_from_slice(ORDINAL_BOUNDS);
8658        put_u16(&mut out, u16::try_from(bounds.len()).map_err(|_| invalid("too many bounds"))?);
8659        for (column, bound) in bounds {
8660            put_u16(
8661                &mut out,
8662                u16::try_from(column).map_err(|_| invalid("bound column overflows"))?,
8663            );
8664            put_u64(&mut out, bound);
8665        }
8666    }
8667    if !table.pair_frequencies.is_empty() {
8668        out.extend_from_slice(PAIR_FREQUENCIES);
8669        put_u16(
8670            &mut out,
8671            u16::try_from(table.pair_frequencies.len())
8672                .map_err(|_| invalid("too many pair frequency summaries"))?,
8673        );
8674        for summary in &table.pair_frequencies {
8675            put_u16(&mut out, summary.first);
8676            put_u16(&mut out, summary.second);
8677            put_u64(&mut out, summary.omitted_max);
8678            put_u16(
8679                &mut out,
8680                u16::try_from(summary.entries.len())
8681                    .map_err(|_| invalid("too many pair frequency entries"))?,
8682            );
8683            for entry in &summary.entries {
8684                put_u16(&mut out, entry.first_entry);
8685                match entry.second {
8686                    None => out.push(0),
8687                    Some(code) => {
8688                        out.push(1);
8689                        put_u32(&mut out, code);
8690                    }
8691                }
8692                put_u64(&mut out, entry.count);
8693            }
8694        }
8695    }
8696    let text_columns = table.frequency_texts.iter().filter(|texts| !texts.is_empty()).count();
8697    if text_columns != 0 {
8698        out.extend_from_slice(FREQUENCY_TEXTS);
8699        put_u16(
8700            &mut out,
8701            u16::try_from(text_columns)
8702                .map_err(|_| invalid("too many string frequency columns"))?,
8703        );
8704        for (column, texts) in table.frequency_texts.iter().enumerate() {
8705            if texts.is_empty() {
8706                continue;
8707            }
8708            put_u16(
8709                &mut out,
8710                u16::try_from(column).map_err(|_| invalid("frequency text column overflows"))?,
8711            );
8712            put_u16(
8713                &mut out,
8714                u16::try_from(texts.len())
8715                    .map_err(|_| invalid("too many frequency text entries"))?,
8716            );
8717            for text in texts {
8718                match text {
8719                    None => out.push(0),
8720                    Some(text) => {
8721                        out.push(1);
8722                        put_u32(
8723                            &mut out,
8724                            u32::try_from(text.len())
8725                                .map_err(|_| invalid("frequency text is too long"))?,
8726                        );
8727                        out.extend_from_slice(text);
8728                    }
8729                }
8730            }
8731        }
8732    }
8733    if let Some(summary) = &table.host_groups {
8734        out.extend_from_slice(HOST_GROUPS);
8735        put_u16(
8736            &mut out,
8737            u16::try_from(summary.column).map_err(|_| invalid("host column overflows"))?,
8738        );
8739        put_u64(&mut out, summary.omitted_max);
8740        put_u16(
8741            &mut out,
8742            u16::try_from(summary.entries.len()).map_err(|_| invalid("too many host groups"))?,
8743        );
8744        for entry in &summary.entries {
8745            put_u32(
8746                &mut out,
8747                u32::try_from(entry.host.len()).map_err(|_| invalid("host name is too long"))?,
8748            );
8749            out.extend_from_slice(entry.host.as_bytes());
8750            put_u64(&mut out, entry.count);
8751            out.extend_from_slice(&entry.bytes_sum.to_le_bytes());
8752            put_u32(
8753                &mut out,
8754                u32::try_from(entry.minimum.len())
8755                    .map_err(|_| invalid("host minimum is too long"))?,
8756            );
8757            out.extend_from_slice(entry.minimum.as_bytes());
8758        }
8759    }
8760    // Written only when there is a declaration, so that the common file is the same bytes it was
8761    // and the section is not a byte of zero on every table in the world that never asked for one.
8762    if let Some(clustering) = &table.clustering {
8763        out.extend_from_slice(CLUSTERING);
8764        out.push(clustering.width().tag());
8765        put_u16(
8766            &mut out,
8767            u16::try_from(clustering.columns().len())
8768                .map_err(|_| invalid("too many clustering columns"))?,
8769        );
8770        for &column in clustering.columns() {
8771            put_u16(
8772                &mut out,
8773                u16::try_from(column).map_err(|_| invalid("clustering column index overflow"))?,
8774            );
8775        }
8776    }
8777    let demoted = (0..table.fields.len())
8778        .filter(|&column| table.demoted.get(column).copied().unwrap_or(false))
8779        .collect::<Vec<_>>();
8780    if !demoted.is_empty() {
8781        out.extend_from_slice(DEMOTED);
8782        put_u16(
8783            &mut out,
8784            u16::try_from(demoted.len()).map_err(|_| invalid("too many demoted columns"))?,
8785        );
8786        for column in demoted {
8787            put_u16(
8788                &mut out,
8789                u16::try_from(column).map_err(|_| invalid("demoted column index overflow"))?,
8790            );
8791        }
8792    }
8793    if !table.constraints.is_empty() {
8794        out.extend_from_slice(KEYS);
8795        put_count(&mut out, table.constraints.keys.len())?;
8796        for (columns, primary) in &table.constraints.keys {
8797            out.push(u8::from(*primary));
8798            put_columns(&mut out, columns)?;
8799        }
8800        put_count(&mut out, table.constraints.foreign.len())?;
8801        for foreign in &table.constraints.foreign {
8802            put_columns(&mut out, &foreign.columns)?;
8803            put_columns(&mut out, &foreign.referenced)?;
8804            put_u32(
8805                &mut out,
8806                u32::try_from(foreign.table.len())
8807                    .map_err(|_| invalid("table name is too long"))?,
8808            );
8809            out.extend_from_slice(foreign.table.as_bytes());
8810        }
8811    }
8812    // The section table, last, behind its own magic, for the same reason the frequency block is
8813    // behind its own: a reader that stops before it gets a table with no sections, and a table with
8814    // no sections is a correct table. The one difference from the blocks before it is that this one
8815    // is written even when it is empty, so that a file written by this build always says which
8816    // sections it has rather than leaving a reader to infer it from where the bytes ran out.
8817    out.extend_from_slice(SECTIONS);
8818    put_u64(&mut out, table.generation);
8819    put_u16(
8820        &mut out,
8821        u16::try_from(table.sections.len()).map_err(|_| invalid("too many sections"))?,
8822    );
8823    for held in &table.sections {
8824        held.encode(&mut out)?;
8825    }
8826    if table.dictionary_payloads.iter().any(|&bytes| bytes != 0) {
8827        out.extend_from_slice(DICTIONARY_PAYLOADS);
8828        put_u16(
8829            &mut out,
8830            u16::try_from(table.fields.len()).map_err(|_| invalid("too many columns"))?,
8831        );
8832        for at in 0..table.fields.len() {
8833            put_u64(&mut out, table.dictionary_payloads.get(at).copied().unwrap_or(0));
8834        }
8835    }
8836    Ok(out)
8837}
8838
8839/// The small level of the directory, naming every table in the file.
8840///
8841/// This is what a footer slot points at. Each entry carries its own checksum over its table
8842/// directory, so a table whose directory is torn is found when that table is first touched rather
8843/// than being trusted because the catalog around it checksummed.
8844///
8845/// The views go after the tables and are whole here, since a view is text and a column list and has
8846/// no pages for a second level to point at.
8847fn signed_integer(ty: &LogicalType) -> bool {
8848    matches!(
8849        ty,
8850        LogicalType::TinyInt | LogicalType::SmallInt | LogicalType::Integer | LogicalType::BigInt
8851    )
8852}
8853
8854fn integer_or_date(ty: &LogicalType) -> bool {
8855    matches!(
8856        ty,
8857        LogicalType::TinyInt
8858            | LogicalType::SmallInt
8859            | LogicalType::Integer
8860            | LogicalType::BigInt
8861            | LogicalType::UTinyInt
8862            | LogicalType::USmallInt
8863            | LogicalType::UInteger
8864            | LogicalType::UBigInt
8865            | LogicalType::Date
8866    )
8867}
8868
8869fn table_integer_extremes(table: &Table) -> Vec<StoredIntegerExtremes> {
8870    table
8871        .fields
8872        .iter()
8873        .enumerate()
8874        .map(|(column, field)| {
8875            if !integer_or_date(&field.ty) {
8876                return None;
8877            }
8878            let mut low: Option<i128> = None;
8879            let mut high: Option<i128> = None;
8880            for stripe in &table.stripes {
8881                let range = stripe.zone.column(column)?;
8882                if !range.exact {
8883                    return None;
8884                }
8885                match (range.low.as_ref(), range.high.as_ref()) {
8886                    (Some(Bound::Int(small)), Some(Bound::Int(large))) => {
8887                        low = Some(low.map_or(*small, |held| held.min(*small)));
8888                        high = Some(high.map_or(*large, |held| held.max(*large)));
8889                    }
8890                    (None, None) if stripe.rows == range.nulls => {}
8891                    _ => return None,
8892                }
8893            }
8894            Some(low.zip(high))
8895        })
8896        .collect()
8897}
8898
8899fn reader_integer_extremes(reader: &Reader) -> Result<Vec<StoredIntegerExtremes>> {
8900    reader
8901        .table
8902        .fields
8903        .iter()
8904        .enumerate()
8905        .map(|(column, field)| {
8906            if !integer_or_date(&field.ty) {
8907                return Ok(None);
8908            }
8909            match reader.exact_extremes(column)? {
8910                Some((Bound::Int(low), Bound::Int(high))) => Ok(Some(Some((low, high)))),
8911                None if reader.null_count(column)? == reader.table.rows as u64 => Ok(Some(None)),
8912                _ => Ok(None),
8913            }
8914        })
8915        .collect()
8916}
8917
8918fn table_complete_numeric_frequencies(table: &Table) -> Vec<StoredNumericFrequencies> {
8919    table
8920        .fields
8921        .iter()
8922        .enumerate()
8923        .map(|(column, field)| {
8924            if !integer_or_date(&field.ty) {
8925                return None;
8926            }
8927            let Some(Frequencies::Held(summary)) = table.frequencies.get(column)?.as_ref() else {
8928                return None;
8929            };
8930            if summary.omitted_max != 0 || summary.entries.len() > MAX_CATALOG_FREQUENCIES {
8931                return None;
8932            }
8933            let entries = summary
8934                .entries
8935                .iter()
8936                .map(|entry| {
8937                    let value = match entry.value {
8938                        FrequencyValue::Null => None,
8939                        FrequencyValue::Integer(value) => Some(value),
8940                        FrequencyValue::Code(_) => return None,
8941                    };
8942                    Some((value, entry.count))
8943                })
8944                .collect::<Option<Vec<_>>>()?;
8945            let rows = entries.iter().try_fold(0_u64, |sum, (_, count)| sum.checked_add(*count))?;
8946            (rows == table.rows as u64).then_some(entries)
8947        })
8948        .collect()
8949}
8950
8951/// The sixty four bits the close keys a numeric column's frequencies by, for a value the writer's
8952/// tally held.
8953///
8954/// The same bits [`Writer::visit_numeric`] hands over: a signed value sign extended to `i64`, and an
8955/// unsigned one as it is.
8956/// The value a column's sixty four bits stand for, read as signed or unsigned the way the column is.
8957fn integer_value(bits: u64, signed: bool) -> FrequencyValue {
8958    if signed {
8959        FrequencyValue::Integer(i128::from(bits as i64))
8960    } else {
8961        FrequencyValue::Integer(i128::from(bits))
8962    }
8963}
8964
8965fn frequency_bits(value: &Value) -> Option<u64> {
8966    Some(match value {
8967        Value::TinyInt(value) => i64::from(*value) as u64,
8968        Value::SmallInt(value) => i64::from(*value) as u64,
8969        Value::Integer(value) | Value::Date(value) => i64::from(*value) as u64,
8970        Value::BigInt(value) | Value::Timestamp(value) => *value as u64,
8971        Value::UTinyInt(value) => u64::from(*value),
8972        Value::USmallInt(value) => u64::from(*value),
8973        Value::UInteger(value) => u64::from(*value),
8974        Value::UBigInt(value) => *value,
8975        _ => return None,
8976    })
8977}
8978
8979fn numeric_frequency_value(value: &Value) -> Option<Option<i128>> {
8980    Some(match value {
8981        Value::Null => None,
8982        Value::TinyInt(value) => Some(i128::from(*value)),
8983        Value::SmallInt(value) => Some(i128::from(*value)),
8984        Value::Integer(value) | Value::Date(value) => Some(i128::from(*value)),
8985        Value::BigInt(value) => Some(i128::from(*value)),
8986        Value::UTinyInt(value) => Some(i128::from(*value)),
8987        Value::USmallInt(value) => Some(i128::from(*value)),
8988        Value::UInteger(value) => Some(i128::from(*value)),
8989        Value::UBigInt(value) => Some(i128::from(*value)),
8990        _ => return None,
8991    })
8992}
8993
8994fn reader_complete_numeric_frequencies(reader: &Reader) -> Result<Vec<StoredNumericFrequencies>> {
8995    reader
8996        .table
8997        .fields
8998        .iter()
8999        .enumerate()
9000        .map(|(column, field)| {
9001            if !integer_or_date(&field.ty) {
9002                return Ok(None);
9003            }
9004            let Some((entries, omitted_max)) = reader.frequency_head(column)? else {
9005                return Ok(None);
9006            };
9007            if omitted_max != 0 || entries.len() > MAX_CATALOG_FREQUENCIES {
9008                return Ok(None);
9009            }
9010            let entries = reader.decode_frequencies(column, &field.ty, &entries)?;
9011            let Some(entries) = entries
9012                .iter()
9013                .map(|(value, count)| Some((numeric_frequency_value(value)?, *count)))
9014                .collect::<Option<Vec<_>>>()
9015            else {
9016                return Ok(None);
9017            };
9018            let rows = entries.iter().try_fold(0_u64, |sum, (_, count)| sum.checked_add(*count));
9019            Ok((rows == Some(reader.table.rows as u64)).then_some(entries))
9020        })
9021        .collect()
9022}
9023
9024fn table_exact_sum(table: &Table, column: usize) -> Option<(i128, u64)> {
9025    table.stripes.iter().try_fold((0_i128, 0_u64), |(sum, count), stripe| {
9026        let range = stripe.zone.column(column)?;
9027        let sum = sum.checked_add(range.sum?)?;
9028        let nonnull = (stripe.rows as u64).checked_sub(range.nulls as u64)?;
9029        Some((sum, count.checked_add(nonnull)?))
9030    })
9031}
9032
9033fn table_aggregate_sums(table: &Table) -> Vec<Option<(i128, u64)>> {
9034    table
9035        .fields
9036        .iter()
9037        .enumerate()
9038        .map(|(column, field)| {
9039            signed_integer(&field.ty).then(|| table_exact_sum(table, column)).flatten()
9040        })
9041        .collect()
9042}
9043
9044fn reader_aggregate_sums(reader: &Reader) -> Result<Vec<Option<(i128, u64)>>> {
9045    reader
9046        .table
9047        .fields
9048        .iter()
9049        .enumerate()
9050        .map(
9051            |(column, field)| {
9052                if signed_integer(&field.ty) { reader.exact_sum(column) } else { Ok(None) }
9053            },
9054        )
9055        .collect()
9056}
9057
9058fn encode_catalog(
9059    entries: &[Entry],
9060    views: &[ViewEntry],
9061    card: Option<&KeptCard>,
9062    anchor: Option<&LogAnchor>,
9063) -> Result<Vec<u8>> {
9064    let mut out = CATALOG.to_vec();
9065    put_u32(&mut out, u32::try_from(entries.len()).map_err(|_| invalid("too many tables"))?);
9066    for entry in entries {
9067        let name = entry.name.as_bytes();
9068        put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("table name too long"))?);
9069        out.extend_from_slice(name);
9070        put_u64(&mut out, u64::try_from(entry.rows).map_err(|_| invalid("row count overflow"))?);
9071        put_u16(
9072            &mut out,
9073            u16::try_from(entry.fields.len()).map_err(|_| invalid("too many columns"))?,
9074        );
9075        for field in &entry.fields {
9076            let name = field.name.as_bytes();
9077            put_u16(
9078                &mut out,
9079                u16::try_from(name.len()).map_err(|_| invalid("column name too long"))?,
9080            );
9081            out.extend_from_slice(name);
9082            put_type(&mut out, &field.ty)?;
9083            out.push(u8::from(field.not_null));
9084        }
9085        put_u64(&mut out, entry.directory.offset);
9086        put_u32(&mut out, entry.directory.length);
9087        put_u64(&mut out, entry.directory.hash);
9088    }
9089    put_u32(&mut out, u32::try_from(views.len()).map_err(|_| invalid("too many views"))?);
9090    for view in views {
9091        let name = view.name.as_bytes();
9092        put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("view name too long"))?);
9093        out.extend_from_slice(name);
9094        put_long_text(&mut out, &view.sql, "view body")?;
9095        put_long_text(&mut out, &view.statement, "view statement")?;
9096        put_u16(
9097            &mut out,
9098            u16::try_from(view.aliases.len()).map_err(|_| invalid("too many aliases"))?,
9099        );
9100        for alias in &view.aliases {
9101            let alias = alias.as_bytes();
9102            put_u16(
9103                &mut out,
9104                u16::try_from(alias.len()).map_err(|_| invalid("alias name too long"))?,
9105            );
9106            out.extend_from_slice(alias);
9107        }
9108        put_u16(
9109            &mut out,
9110            u16::try_from(view.columns.len()).map_err(|_| invalid("too many columns"))?,
9111        );
9112        for field in &view.columns {
9113            let name = field.name.as_bytes();
9114            put_u16(
9115                &mut out,
9116                u16::try_from(name.len()).map_err(|_| invalid("column name too long"))?,
9117            );
9118            out.extend_from_slice(name);
9119            put_type(&mut out, &field.ty)?;
9120            out.push(u8::from(field.not_null));
9121        }
9122    }
9123    out.extend_from_slice(NONZERO_COUNTS);
9124    for entry in entries {
9125        if entry.nonzero.len() != entry.fields.len() {
9126            return Err(invalid("nonzero count width differs from schema"));
9127        }
9128        for count in &entry.nonzero {
9129            match count {
9130                None => out.push(0),
9131                Some(count) => {
9132                    out.push(1);
9133                    put_u64(&mut out, *count);
9134                }
9135            }
9136        }
9137    }
9138    out.extend_from_slice(AGGREGATE_SUMS);
9139    for entry in entries {
9140        if entry.aggregates.len() != entry.fields.len() {
9141            return Err(invalid("aggregate sum width differs from schema"));
9142        }
9143        for summary in &entry.aggregates {
9144            match summary {
9145                None => out.push(0),
9146                Some((sum, count)) => {
9147                    out.push(1);
9148                    out.extend_from_slice(&sum.to_le_bytes());
9149                    put_u64(&mut out, *count);
9150                }
9151            }
9152        }
9153    }
9154    out.extend_from_slice(DISTINCT_COUNTS);
9155    for entry in entries {
9156        if entry.distincts.len() != entry.fields.len() {
9157            return Err(invalid("distinct count width differs from schema"));
9158        }
9159        for count in &entry.distincts {
9160            match count {
9161                None => out.push(0),
9162                Some(count) => {
9163                    if *count > entry.rows as u64 {
9164                        return Err(invalid("distinct count exceeds table rows"));
9165                    }
9166                    out.push(1);
9167                    put_u64(&mut out, *count);
9168                }
9169            }
9170        }
9171    }
9172    out.extend_from_slice(INTEGER_EXTREMES);
9173    for entry in entries {
9174        if entry.extremes.len() != entry.fields.len() {
9175            return Err(invalid("integer extremes width differs from schema"));
9176        }
9177        for (field, extremes) in entry.fields.iter().zip(&entry.extremes) {
9178            match extremes {
9179                None => out.push(0),
9180                Some(None) if integer_or_date(&field.ty) => out.push(1),
9181                Some(Some((low, high))) if integer_or_date(&field.ty) && low <= high => {
9182                    out.push(2);
9183                    out.extend_from_slice(&low.to_le_bytes());
9184                    out.extend_from_slice(&high.to_le_bytes());
9185                }
9186                _ => return Err(invalid("integer extremes type or range differs")),
9187            }
9188        }
9189    }
9190    out.extend_from_slice(COMPLETE_FREQUENCIES);
9191    for entry in entries {
9192        if entry.frequencies.len() != entry.fields.len() {
9193            return Err(invalid("numeric frequency width differs from schema"));
9194        }
9195        for (field, frequencies) in entry.fields.iter().zip(&entry.frequencies) {
9196            match frequencies {
9197                None => out.push(0),
9198                Some(entries)
9199                    if integer_or_date(&field.ty) && entries.len() <= MAX_CATALOG_FREQUENCIES =>
9200                {
9201                    let mut total = 0_u64;
9202                    for (at, (value, count)) in entries.iter().enumerate() {
9203                        if entries[..at].iter().any(|(held, _)| held == value) {
9204                            return Err(invalid("numeric frequency value repeats"));
9205                        }
9206                        total = total
9207                            .checked_add(*count)
9208                            .ok_or_else(|| invalid("numeric frequency count overflows"))?;
9209                    }
9210                    if total != entry.rows as u64 {
9211                        return Err(invalid("numeric frequencies do not cover table rows"));
9212                    }
9213                    out.push(1);
9214                    out.push(entries.len() as u8);
9215                    for (value, count) in entries {
9216                        match value {
9217                            None => out.push(0),
9218                            Some(value) => {
9219                                out.push(1);
9220                                out.extend_from_slice(&value.to_le_bytes());
9221                            }
9222                        }
9223                        put_u64(&mut out, *count);
9224                    }
9225                }
9226                _ => return Err(invalid("numeric frequency type or width differs")),
9227            }
9228        }
9229    }
9230    if let Some(card) = card {
9231        out.extend_from_slice(DEVICE_CARD);
9232        let device = card.device.as_bytes();
9233        put_u16(&mut out, u16::try_from(device.len()).map_err(|_| invalid("device id too long"))?);
9234        out.extend_from_slice(device);
9235        put_u32(&mut out, u32::try_from(card.bytes.len()).map_err(|_| invalid("card too long"))?);
9236        out.extend_from_slice(&card.bytes);
9237    }
9238    if let Some(anchor) = anchor {
9239        anchor.encode(&mut out)?;
9240    }
9241    Ok(out)
9242}
9243
9244/// A length and that many bytes, for text that is allowed to be longer than a name.
9245fn put_long_text(out: &mut Vec<u8>, text: &str, what: &str) -> Result<()> {
9246    let bytes = text.as_bytes();
9247    put_u32(out, u32::try_from(bytes.len()).map_err(|_| invalid(&format!("{what} too long")))?);
9248    out.extend_from_slice(bytes);
9249    Ok(())
9250}
9251
9252/// Reads the catalog directory back, checking every span against the file before anything is
9253/// allocated for it.
9254fn decode_catalog(bytes: &[u8], size: u64) -> Result<Decoded> {
9255    let mut cur = Cursor::new(bytes);
9256    if cur.take(8)? != CATALOG {
9257        return Err(invalid("catalog magic differs"));
9258    }
9259    let count = cur.u32()? as usize;
9260    let mut entries: Vec<Entry> = Vec::with_capacity(count.min(1024));
9261    for _ in 0..count {
9262        let name = cur.text()?;
9263        let rows = usize::try_from(cur.u64()?).map_err(|_| invalid("row count does not fit"))?;
9264        let width = cur.u16()? as usize;
9265        let mut fields = Vec::with_capacity(width);
9266        for _ in 0..width {
9267            let name = cur.text()?;
9268            let ty = read_type(&mut cur)?;
9269            let not_null = match cur.u8()? {
9270                0 => false,
9271                1 => true,
9272                _ => return Err(invalid("nullability flag differs")),
9273            };
9274            fields.push(Field { name, ty, not_null });
9275        }
9276        let directory = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
9277        let end = directory
9278            .offset
9279            .checked_add(u64::from(directory.length))
9280            .ok_or_else(|| invalid("table directory offset overflow"))?;
9281        if directory.offset < HEADER
9282            || end > size
9283            || directory.length as usize > MAX_DIRECTORY
9284            || directory.length == 0
9285        {
9286            return Err(invalid("table directory range is outside the file"));
9287        }
9288        if entries.iter().any(|held| held.name == name) {
9289            return Err(invalid("two tables in the catalog have the same name"));
9290        }
9291        let nonzero = vec![None; fields.len()];
9292        let aggregates = vec![None; fields.len()];
9293        let distincts = vec![None; fields.len()];
9294        let extremes = vec![None; fields.len()];
9295        let frequencies = vec![None; fields.len()];
9296        entries.push(Entry {
9297            name,
9298            fields,
9299            rows,
9300            directory,
9301            nonzero,
9302            aggregates,
9303            distincts,
9304            extremes,
9305            frequencies,
9306        });
9307    }
9308    // A catalog that ends where the tables end is a catalog with no views in it, which is every
9309    // file written before format 25. That is why the count is allowed to be missing rather than
9310    // read as a zero that has to be there: an older file has nothing after the last table entry at
9311    // all, and [`READABLE`] says those files still open.
9312    let count = if cur.done() { 0 } else { cur.u32()? as usize };
9313    let mut views: Vec<ViewEntry> = Vec::with_capacity(count.min(1024));
9314    for _ in 0..count {
9315        let name = cur.text()?;
9316        let sql = cur.long_text()?;
9317        let statement = cur.long_text()?;
9318        let width = cur.u16()? as usize;
9319        let mut aliases = Vec::with_capacity(width);
9320        for _ in 0..width {
9321            aliases.push(cur.text()?);
9322        }
9323        let width = cur.u16()? as usize;
9324        let mut columns = Vec::with_capacity(width);
9325        for _ in 0..width {
9326            let name = cur.text()?;
9327            let ty = read_type(&mut cur)?;
9328            let not_null = match cur.u8()? {
9329                0 => false,
9330                1 => true,
9331                _ => return Err(invalid("nullability flag differs")),
9332            };
9333            columns.push(Field { name, ty, not_null });
9334        }
9335        // The same rule the tables above get, and for the same reason. Two entries under one name
9336        // is a catalog nothing can answer a lookup from, and finding that out here is better than
9337        // finding it out from whichever of the two a search happened to reach first.
9338        if views.iter().any(|held| held.name == name) {
9339            return Err(invalid("two views in the catalog have the same name"));
9340        }
9341        if entries.iter().any(|held| held.name == name) {
9342            return Err(invalid("a table and a view in the catalog have the same name"));
9343        }
9344        views.push(ViewEntry { name, sql, statement, aliases, columns });
9345    }
9346    if !cur.done() {
9347        if cur.take(8)? != NONZERO_COUNTS {
9348            return Err(invalid("catalog extension magic differs"));
9349        }
9350        for entry in &mut entries {
9351            for (field, count) in entry.fields.iter().zip(&mut entry.nonzero) {
9352                *count = match cur.u8()? {
9353                    0 => None,
9354                    1 if matches!(
9355                        field.ty,
9356                        LogicalType::TinyInt
9357                            | LogicalType::SmallInt
9358                            | LogicalType::Integer
9359                            | LogicalType::BigInt
9360                            | LogicalType::UTinyInt
9361                            | LogicalType::USmallInt
9362                            | LogicalType::UInteger
9363                            | LogicalType::UBigInt
9364                    ) =>
9365                    {
9366                        let value = cur.u64()?;
9367                        if value > entry.rows as u64 {
9368                            return Err(invalid("nonzero count exceeds rows"));
9369                        }
9370                        Some(value)
9371                    }
9372                    _ => return Err(invalid("nonzero count tag or column type differs")),
9373                };
9374            }
9375        }
9376    }
9377    if !cur.done() {
9378        if cur.take(8)? != AGGREGATE_SUMS {
9379            return Err(invalid("aggregate catalog extension magic differs"));
9380        }
9381        for entry in &mut entries {
9382            for (field, summary) in entry.fields.iter().zip(&mut entry.aggregates) {
9383                *summary = match cur.u8()? {
9384                    0 => None,
9385                    1 if signed_integer(&field.ty) => {
9386                        let sum = i128::from_le_bytes(
9387                            cur.take(16)?
9388                                .try_into()
9389                                .map_err(|_| invalid("aggregate sum is truncated"))?,
9390                        );
9391                        let count = cur.u64()?;
9392                        if count > entry.rows as u64 {
9393                            return Err(invalid("aggregate count exceeds table rows"));
9394                        }
9395                        Some((sum, count))
9396                    }
9397                    _ => return Err(invalid("aggregate sum tag or column type differs")),
9398                };
9399            }
9400        }
9401    }
9402    if !cur.done() {
9403        if cur.take(8)? != DISTINCT_COUNTS {
9404            return Err(invalid("distinct catalog extension magic differs"));
9405        }
9406        for entry in &mut entries {
9407            for count in &mut entry.distincts {
9408                *count = match cur.u8()? {
9409                    0 => None,
9410                    1 => {
9411                        let value = cur.u64()?;
9412                        if value > entry.rows as u64 {
9413                            return Err(invalid("distinct count exceeds table rows"));
9414                        }
9415                        Some(value)
9416                    }
9417                    _ => return Err(invalid("distinct count tag differs")),
9418                };
9419            }
9420        }
9421    }
9422    if !cur.done() {
9423        if cur.take(8)? != INTEGER_EXTREMES {
9424            return Err(invalid("integer extremes catalog extension magic differs"));
9425        }
9426        for entry in &mut entries {
9427            for (field, extremes) in entry.fields.iter().zip(&mut entry.extremes) {
9428                *extremes = match cur.u8()? {
9429                    0 => None,
9430                    1 if integer_or_date(&field.ty) => Some(None),
9431                    2 if integer_or_date(&field.ty) => {
9432                        let low = i128::from_le_bytes(
9433                            cur.take(16)?
9434                                .try_into()
9435                                .map_err(|_| invalid("minimum is truncated"))?,
9436                        );
9437                        let high = i128::from_le_bytes(
9438                            cur.take(16)?
9439                                .try_into()
9440                                .map_err(|_| invalid("maximum is truncated"))?,
9441                        );
9442                        if low > high {
9443                            return Err(invalid("integer extremes are reversed"));
9444                        }
9445                        Some(Some((low, high)))
9446                    }
9447                    _ => return Err(invalid("integer extremes tag or type differs")),
9448                };
9449            }
9450        }
9451    }
9452    if !cur.done() {
9453        if cur.take(8)? != COMPLETE_FREQUENCIES {
9454            return Err(invalid("numeric frequency catalog extension magic differs"));
9455        }
9456        for entry in &mut entries {
9457            for (field, frequencies) in entry.fields.iter().zip(&mut entry.frequencies) {
9458                *frequencies = match cur.u8()? {
9459                    0 => None,
9460                    1 if integer_or_date(&field.ty) => {
9461                        let len = cur.u8()? as usize;
9462                        if len > MAX_CATALOG_FREQUENCIES {
9463                            return Err(invalid("too many catalog numeric frequencies"));
9464                        }
9465                        let mut values = Vec::with_capacity(len);
9466                        let mut total = 0_u64;
9467                        for _ in 0..len {
9468                            let value = match cur.u8()? {
9469                                0 => None,
9470                                1 => Some(i128::from_le_bytes(cur.take(16)?.try_into().map_err(
9471                                    |_| invalid("numeric frequency value is truncated"),
9472                                )?)),
9473                                _ => return Err(invalid("numeric frequency value tag differs")),
9474                            };
9475                            if values.iter().any(|(held, _)| *held == value) {
9476                                return Err(invalid("numeric frequency value repeats"));
9477                            }
9478                            let count = cur.u64()?;
9479                            total = total
9480                                .checked_add(count)
9481                                .ok_or_else(|| invalid("numeric frequency count overflows"))?;
9482                            values.push((value, count));
9483                        }
9484                        if total != entry.rows as u64 {
9485                            return Err(invalid("numeric frequencies do not cover table rows"));
9486                        }
9487                        Some(values)
9488                    }
9489                    _ => return Err(invalid("numeric frequency tag or type differs")),
9490                };
9491            }
9492        }
9493    }
9494    let mut card = None;
9495    let mut anchor = None;
9496    // The extensions in the order they are written, each at most once. An older build stops at the
9497    // first magic it does not know, which is how a file it cannot read whole says so.
9498    while !cur.done() {
9499        let tag = cur.take(8)?;
9500        if tag == DEVICE_CARD && card.is_none() && anchor.is_none() {
9501            let device = cur.text()?;
9502            let len = cur.u32()? as usize;
9503            if len > MAX_CARD {
9504                return Err(invalid("device card is longer than any card"));
9505            }
9506            card = Some(KeptCard { device, bytes: cur.take(len)?.to_vec() });
9507        } else if tag == anchor::LOG_ANCHOR && anchor.is_none() {
9508            anchor = Some(LogAnchor::decode(&mut cur)?);
9509        } else {
9510            return Err(invalid("catalog extension magic differs or repeats"));
9511        }
9512    }
9513    Ok((entries, views, card, anchor))
9514}
9515
9516/// What [`decode_catalog`] reads: the tables, the views, the device card and the log anchor.
9517type Decoded = (Vec<Entry>, Vec<ViewEntry>, Option<KeptCard>, Option<LogAnchor>);
9518
9519/// The most a kept device card can take, which is many times what one holds.
9520const MAX_CARD: usize = 64 << 10;
9521
9522/// The device card a file keeps, as `rudb_io::device` encodes it, and the device it was measured
9523/// on.
9524///
9525/// `16-measurement.md` section 16.3 keeps the card in the file so that a process opening the file
9526/// does not measure the device again. It is only good on that device, so it carries the device id
9527/// and a file copied somewhere else keeps its card but nobody takes it.
9528#[derive(Debug, Clone, PartialEq, Eq)]
9529struct KeptCard {
9530    device: String,
9531    bytes: Vec<u8>,
9532}
9533
9534/// The directory a database file is in, which is the one its device card is about.
9535fn directory_of(path: &Path) -> &Path {
9536    path.parent().filter(|dir| !dir.as_os_str().is_empty()).unwrap_or(Path::new("."))
9537}
9538
9539/// The card the next commit of the file at `path` writes down.
9540///
9541/// The one this process has for the device the file is on when there is one, since it was either
9542/// measured here or read out of a file on the same device, and otherwise whatever the file already
9543/// kept. A file never makes a process measure: the card is measured when something asks for it,
9544/// and this only writes down what is already known.
9545fn card_for(path: &Path, held: Option<KeptCard>) -> Option<KeptCard> {
9546    let Ok(device) = rudb_io::device::device_key(directory_of(path)) else {
9547        return held;
9548    };
9549    match rudb_io::device::kept(&device) {
9550        Some(card) => Some(KeptCard { device, bytes: card.encode() }),
9551        None => held,
9552    }
9553}
9554
9555/// Hands the card a file kept to this process, when the file is still on the device it describes.
9556fn remember_card(path: &Path, card: Option<&KeptCard>) {
9557    let Some(card) = card else { return };
9558    let dir = directory_of(path);
9559    let Ok(device) = rudb_io::device::device_key(dir) else { return };
9560    if device != card.device {
9561        return;
9562    }
9563    if let Ok(decoded) = rudb_io::device::Card::decode(&card.bytes, dir) {
9564        rudb_io::device::remember(&device, decoded);
9565    }
9566}
9567
9568/// Reads the fields of a directory or a catalog in order, off bytes in memory or out of the file.
9569///
9570/// A catalog is small and is read whole. A table directory is not: at ten million rows of `hits` it
9571/// is nearly a megabyte, and holding that buffer while the table it describes is built out of it
9572/// put both at the peak of every query. Out of the file, the cursor holds one window of
9573/// [`DIRECTORY_WINDOW`] bytes and moves it forward as the fields are read, so what a directory
9574/// costs at open is what it decodes into and not that plus its own bytes.
9575struct Cursor<'a> {
9576    bytes: &'a [u8],
9577    at: usize,
9578    window: Option<Window<'a>>,
9579}
9580
9581/// The part of a directory in the file that a [`Cursor`] has read in.
9582struct Window<'a> {
9583    file: &'a File,
9584    offset: u64,
9585    length: usize,
9586    /// Where `held` starts, counted from the start of the directory.
9587    start: usize,
9588    held: Vec<u8>,
9589    /// How much to read at once, which is [`DIRECTORY_WINDOW`] outside the tests.
9590    size: usize,
9591}
9592
9593/// How much of a directory a cursor reading one out of the file holds at once.
9594const DIRECTORY_WINDOW: usize = 64 << 10;
9595
9596impl<'a> Cursor<'a> {
9597    fn new(bytes: &'a [u8]) -> Self {
9598        Self { bytes, at: 0, window: None }
9599    }
9600
9601    /// A cursor over `length` bytes of `file` from `offset`, which it reads a window at a time.
9602    fn over(file: &'a File, offset: u64, length: usize) -> Self {
9603        let window =
9604            Window { file, offset, length, start: 0, held: Vec::new(), size: DIRECTORY_WINDOW };
9605        Self { bytes: &[], at: 0, window: Some(window) }
9606    }
9607
9608    /// How many bytes the cursor walks in all.
9609    fn len(&self) -> usize {
9610        self.window.as_ref().map_or(self.bytes.len(), |window| window.length)
9611    }
9612
9613    /// Makes sure the next `len` bytes are in memory.
9614    fn ensure(&mut self, len: usize) -> Result<()> {
9615        let end = self.at.checked_add(len).ok_or_else(|| invalid("directory offset overflow"))?;
9616        if end > self.len() {
9617            return Err(invalid("directory is truncated"));
9618        }
9619        let Some(window) = &mut self.window else { return Ok(()) };
9620        if self.at < window.start || end > window.start + window.held.len() {
9621            let want = len.max(window.size).min(window.length - self.at);
9622            window.start = self.at;
9623            window.held.resize(want, 0);
9624            read_at(window.file, window.offset + self.at as u64, &mut window.held)?;
9625        }
9626        Ok(())
9627    }
9628
9629    /// `len` bytes from `at`, which [`Self::ensure`] has already brought in.
9630    fn held(&self, at: usize, len: usize) -> &[u8] {
9631        match &self.window {
9632            Some(window) => &window.held[at - window.start..at - window.start + len],
9633            None => &self.bytes[at..at + len],
9634        }
9635    }
9636
9637    /// The next `len` bytes, without moving past them.
9638    #[inline]
9639    fn peek(&mut self, len: usize) -> Result<&[u8]> {
9640        if self.window.is_none() {
9641            let bytes = self.bytes;
9642            return Ok(&bytes[self.at..self.end(len)?]);
9643        }
9644        self.ensure(len)?;
9645        Ok(self.held(self.at, len))
9646    }
9647
9648    /// The next `len` bytes, moving past them.
9649    ///
9650    /// Every data page is decoded through this, a byte or a word at a time, so a cursor over bytes
9651    /// already in memory takes them here and never reaches [`Self::ensure`]. With the window check
9652    /// on every call, q06 on TPC-H spent a seventh of its instructions in it.
9653    #[inline]
9654    fn take(&mut self, len: usize) -> Result<&[u8]> {
9655        if self.window.is_none() {
9656            let bytes = self.bytes;
9657            let (at, end) = (self.at, self.end(len)?);
9658            self.at = end;
9659            return Ok(&bytes[at..end]);
9660        }
9661        self.take_windowed(len)
9662    }
9663
9664    /// Moves over a checked field without reading its payload from a windowed directory.
9665    fn skip(&mut self, len: usize) -> Result<()> {
9666        let end = self.at.checked_add(len).ok_or_else(|| invalid("directory offset overflow"))?;
9667        if end > self.len() {
9668            return Err(invalid("directory is truncated"));
9669        }
9670        self.at = end;
9671        Ok(())
9672    }
9673
9674    fn skip_bound(&mut self) -> Result<()> {
9675        match self.u8()? {
9676            0 => Ok(()),
9677            1 => self.skip(16),
9678            2 => self.skip(8),
9679            3 => {
9680                let length = self.u32()? as usize;
9681                self.skip(length)
9682            }
9683            4 => self.skip(17),
9684            _ => Err(invalid("a stored bound has an unknown tag")),
9685        }
9686    }
9687
9688    /// Where `len` bytes from here end, when they end inside the bytes.
9689    #[inline]
9690    fn end(&self, len: usize) -> Result<usize> {
9691        let end = self.at.checked_add(len).ok_or_else(|| invalid("directory offset overflow"))?;
9692        if end > self.bytes.len() {
9693            return Err(invalid("directory is truncated"));
9694        }
9695        Ok(end)
9696    }
9697
9698    /// [`Self::take`] out of the file, a window at a time.
9699    #[inline(never)]
9700    fn take_windowed(&mut self, len: usize) -> Result<&[u8]> {
9701        self.ensure(len)?;
9702        self.at += len;
9703        Ok(self.held(self.at - len, len))
9704    }
9705    #[inline]
9706    fn u8(&mut self) -> Result<u8> {
9707        Ok(self.take(1)?[0])
9708    }
9709    #[inline]
9710    fn u16(&mut self) -> Result<u16> {
9711        Ok(u16::from_le_bytes(self.take(2)?.try_into().expect("two bytes")))
9712    }
9713    #[inline]
9714    fn u32(&mut self) -> Result<u32> {
9715        Ok(u32::from_le_bytes(self.take(4)?.try_into().expect("four bytes")))
9716    }
9717    #[inline]
9718    fn u64(&mut self) -> Result<u64> {
9719        Ok(u64::from_le_bytes(self.take(8)?.try_into().expect("eight bytes")))
9720    }
9721    fn var_u64(&mut self) -> Result<u64> {
9722        let mut value = 0_u64;
9723        for shift in (0..=63).step_by(7) {
9724            let byte = self.u8()?;
9725            let part = u64::from(byte & 0x7f);
9726            if shift == 63 && part > 1 {
9727                return Err(invalid("frequency ordinal varint overflows"));
9728            }
9729            value |= part << shift;
9730            if byte & 0x80 == 0 {
9731                return Ok(value);
9732            }
9733        }
9734        Err(invalid("frequency ordinal varint is too long"))
9735    }
9736    /// A zone map's end, in the layout `rudb_common::bounds` defines.
9737    ///
9738    /// The bytes are the ones this directory has written since format 10 and the codec moved to
9739    /// rank zero rather than being copied, because a column summary now writes the same two ends
9740    /// and two encodings of one type is how the two quietly stop agreeing.
9741    ///
9742    /// A bound's length is in the bound, so out of the file the cursor offers the codec a few bytes
9743    /// and offers it twice as many whenever it runs out before the directory does.
9744    fn bound(&mut self) -> Result<Option<Bound>> {
9745        let rest = self.len().saturating_sub(self.at);
9746        let mut want = 32;
9747        loop {
9748            let offered = self.peek(want.min(rest))?;
9749            let mut used = 0;
9750            match bounds::get(offered, &mut used) {
9751                Ok(bound) => {
9752                    self.at += used;
9753                    return Ok(bound);
9754                }
9755                Err(_) if want < rest => want *= 2,
9756                Err(error) => return Err(error),
9757            }
9758        }
9759    }
9760    fn text(&mut self) -> Result<String> {
9761        let len = self.u16()? as usize;
9762        String::from_utf8(self.take(len)?.to_vec()).map_err(|_| invalid("name is not UTF-8"))
9763    }
9764    /// Whether everything has been read, which is how a section that an older file does not have at
9765    /// all is told from one that is there and empty.
9766    fn done(&self) -> bool {
9767        self.at >= self.len()
9768    }
9769    /// The same, for text that is a query rather than a name.
9770    ///
9771    /// A name fits in sixteen bits of length and a view body does not have to. Nobody writes a 64
9772    /// kilobyte identifier by accident and people do write generated queries that long, and a view
9773    /// that could not be written down because its body was too big would be a limit invented here
9774    /// rather than one anything else in the engine has.
9775    fn long_text(&mut self) -> Result<String> {
9776        let len = self.u32()? as usize;
9777        String::from_utf8(self.take(len)?.to_vec()).map_err(|_| invalid("text is not UTF-8"))
9778    }
9779}
9780
9781/// One column's frequency synopsis, or `None` for a column that has none, checked against the column.
9782fn decode_summary(
9783    cur: &mut Cursor<'_>,
9784    field: &Field,
9785    rows: usize,
9786    values: bool,
9787) -> Result<Option<FrequencySummary>> {
9788    let Some((entries, omitted_max)) = decode_summary_head(cur, field, rows)? else {
9789        return Ok(None);
9790    };
9791    let ordinals = {
9792        let ordinal_count = cur.u32()? as usize;
9793        if ordinal_count > FREQUENCY_ORDINALS || ordinal_count > rows {
9794            return Err(invalid("frequency ordinal count exceeds its bound"));
9795        }
9796        let mut ordinals = Vec::with_capacity(ordinal_count);
9797        let mut previous = 0_u64;
9798        for at in 0..ordinal_count {
9799            let delta = cur.var_u64()?;
9800            if at != 0 && delta == 0 {
9801                return Err(invalid("frequency ordinals are not increasing"));
9802            }
9803            let ordinal = if at == 0 {
9804                delta
9805            } else {
9806                previous.checked_add(delta).ok_or_else(|| invalid("frequency ordinal overflows"))?
9807            };
9808            if ordinal >= rows as u64 {
9809                return Err(invalid("frequency ordinal is outside the table"));
9810            }
9811            ordinals.push(ordinal);
9812            previous = ordinal;
9813        }
9814        ordinals
9815    };
9816    let ordinal_entries = if values {
9817        let mut ordinal_entries = Vec::with_capacity(ordinals.len());
9818        for _ in 0..ordinals.len() {
9819            let entry = cur.u16()?;
9820            if entry as usize >= entries.len() {
9821                return Err(invalid("frequency ordinal value is outside its entries"));
9822            }
9823            ordinal_entries.push(entry);
9824        }
9825        ordinal_entries
9826    } else {
9827        Vec::new()
9828    };
9829    Ok(Some(FrequencySummary { entries, omitted_max, ordinals, ordinal_entries, ordinal_bound: 0 }))
9830}
9831
9832/// The entries of one column's frequency synopsis and the bound on what they leave out, without
9833/// the row ordinals that follow them, which only a pair count reads.
9834fn decode_summary_head(
9835    cur: &mut Cursor<'_>,
9836    field: &Field,
9837    rows: usize,
9838) -> Result<Option<(Vec<FrequencyEntry>, u64)>> {
9839    Ok(match cur.u8()? {
9840        0 => None,
9841        1 => {
9842            let omitted_max = cur.u64()?;
9843            let count = cur.u32()? as usize;
9844            if count > FREQUENCY_ENTRIES {
9845                return Err(invalid("frequency entry count exceeds its bound"));
9846            }
9847            let mut entries = Vec::with_capacity(count);
9848            // row at a time: directory decoding validates each persisted bounded frequency entry.
9849            for _ in 0..count {
9850                let value = match cur.u8()? {
9851                    0 => FrequencyValue::Null,
9852                    1 => FrequencyValue::Integer(i128::from_le_bytes(
9853                        cur.take(16)?.try_into().expect("sixteen bytes"),
9854                    )),
9855                    2 => FrequencyValue::Code(cur.u32()?),
9856                    _ => return Err(invalid("frequency value tag differs")),
9857                };
9858                let valid = matches!(
9859                    (&field.ty, value),
9860                    (_, FrequencyValue::Null)
9861                        | (LogicalType::Varchar | LogicalType::Blob, FrequencyValue::Code(_))
9862                        | (
9863                            LogicalType::TinyInt
9864                                | LogicalType::SmallInt
9865                                | LogicalType::Integer
9866                                | LogicalType::BigInt
9867                                | LogicalType::UTinyInt
9868                                | LogicalType::USmallInt
9869                                | LogicalType::UInteger
9870                                | LogicalType::UBigInt
9871                                | LogicalType::Date
9872                                | LogicalType::Timestamp,
9873                            FrequencyValue::Integer(_),
9874                        )
9875                );
9876                if !valid {
9877                    return Err(invalid("frequency value does not match its column"));
9878                }
9879                let count = cur.u64()?;
9880                if count == 0 || count > rows as u64 {
9881                    return Err(invalid("frequency count is outside the table"));
9882                }
9883                entries.push(FrequencyEntry { value, count });
9884            }
9885            if entries.windows(2).any(|pair| pair[0].count < pair[1].count) {
9886                return Err(invalid("frequency entries are not descending"));
9887            }
9888            Some((entries, omitted_max))
9889        }
9890        _ => return Err(invalid("frequency summary tag differs")),
9891    })
9892}
9893
9894/// Reads the fixed envelope of a frequency synopsis in a span-based directory.
9895fn summary_span(cur: &mut Cursor<'_>) -> Result<Option<(usize, usize)>> {
9896    let length = cur.u32()? as usize;
9897    let entries = cur.u32()? as usize;
9898    if entries > FREQUENCY_ENTRIES {
9899        return Err(invalid("frequency entry count exceeds its bound"));
9900    }
9901    if length == 0 {
9902        if entries != 0 {
9903            return Err(invalid("missing frequency synopsis has entries"));
9904        }
9905        return Ok(None);
9906    }
9907    if length > MAX_DIRECTORY || length > cur.len().saturating_sub(cur.at) {
9908        return Err(invalid("frequency synopsis span is outside the directory"));
9909    }
9910    Ok(Some((length, entries)))
9911}
9912
9913/// Skips a synopsis whose column the caller does not need. The directory checksum was checked
9914/// before this walk, and the fields still need their lengths and tags checked to find the next one.
9915fn skip_summary(cur: &mut Cursor<'_>, values: bool, rows: usize) -> Result<()> {
9916    match cur.u8()? {
9917        0 => Ok(()),
9918        1 => {
9919            cur.skip(8)?;
9920            let entries = cur.u32()? as usize;
9921            if entries > FREQUENCY_ENTRIES {
9922                return Err(invalid("frequency entry count exceeds its bound"));
9923            }
9924            for _ in 0..entries {
9925                match cur.u8()? {
9926                    0 => {}
9927                    1 => cur.skip(16)?,
9928                    2 => cur.skip(4)?,
9929                    _ => return Err(invalid("frequency value tag differs")),
9930                }
9931                cur.skip(8)?;
9932            }
9933            let ordinals = cur.u32()? as usize;
9934            if ordinals > FREQUENCY_ORDINALS || ordinals > rows {
9935                return Err(invalid("frequency ordinal count exceeds its bound"));
9936            }
9937            for _ in 0..ordinals {
9938                cur.var_u64()?;
9939            }
9940            if values {
9941                cur.skip(ordinals * 2)?;
9942            }
9943            Ok(())
9944        }
9945        _ => Err(invalid("frequency summary tag differs")),
9946    }
9947}
9948
9949/// Reads only the catalog, stripe null counts, and one frequency synopsis. This is the cold path
9950/// for a summary-backed count; constructing every page descriptor and zone map would make it cost
9951/// the size of the table directory even when no row is read.
9952fn quick_nonzero(
9953    mut cur: Cursor<'_>,
9954    name: &str,
9955    fields: &[Field],
9956    rows: usize,
9957    wanted: usize,
9958) -> Result<Option<u64>> {
9959    if cur.take(8)? != DIRECTORY || cur.text()? != name {
9960        return Err(invalid("table directory differs from the catalog"));
9961    }
9962    let width = cur.u16()? as usize;
9963    if width != fields.len() {
9964        return Err(invalid("table directory width differs from the catalog"));
9965    }
9966    for field in fields {
9967        let stored =
9968            Field { name: cur.text()?, ty: read_type(&mut cur)?, not_null: cur.u8()? != 0 };
9969        if &stored != field {
9970            return Err(invalid("table directory schema differs from the catalog"));
9971        }
9972    }
9973    let mut dictionaries = Vec::with_capacity(width);
9974    for field in fields {
9975        let held = match cur.u8()? {
9976            0 => false,
9977            tag if coded_type(&field.ty) && tag == dictionary_tag(&field.ty) => {
9978                cur.skip(20)?;
9979                true
9980            }
9981            _ => return Err(invalid("dictionary page tag differs")),
9982        };
9983        dictionaries.push(held);
9984    }
9985    for _ in 0..width {
9986        match cur.u8()? {
9987            0 => {}
9988            1 => cur.skip(8)?,
9989            _ => return Err(invalid("distinct count tag differs")),
9990        }
9991    }
9992    if cur.u64()? != rows as u64 {
9993        return Err(invalid("table row count differs from the catalog"));
9994    }
9995    let stripes = cur.u32()? as usize;
9996    let mut total = 0_usize;
9997    let mut nulls = 0_u64;
9998    for _ in 0..stripes {
9999        let parts = cur.u32()? as usize;
10000        if parts == 0 || parts > STRIPE_PARTS {
10001            return Err(invalid("stripe part count is outside its bound"));
10002        }
10003        let mut stripe_rows = 0_usize;
10004        for _ in 0..parts {
10005            stripe_rows = stripe_rows
10006                .checked_add(cur.u32()? as usize)
10007                .ok_or_else(|| invalid("stripe row count overflow"))?;
10008        }
10009        total =
10010            total.checked_add(stripe_rows).ok_or_else(|| invalid("stripe row count overflow"))?;
10011        cur.skip(12 + width * 12)?;
10012        for (field, held) in fields.iter().zip(&dictionaries) {
10013            if coded_type(&field.ty) && *held {
10014                cur.skip(20)?;
10015            }
10016        }
10017        for _ in 0..width * 2 {
10018            match cur.u8()? {
10019                0 => {}
10020                1 => cur.skip(20)?,
10021                _ => return Err(invalid("stripe page tag differs")),
10022            }
10023        }
10024        for column in 0..width {
10025            cur.skip_bound()?;
10026            cur.skip_bound()?;
10027            let count = cur.u32()? as u64;
10028            if count > stripe_rows as u64 {
10029                return Err(invalid("null count exceeds stripe rows"));
10030            }
10031            if column == wanted {
10032                nulls = nulls.checked_add(count).ok_or_else(|| invalid("null count overflow"))?;
10033            }
10034            cur.skip(1)?;
10035            match cur.u8()? {
10036                0 => {}
10037                1 => cur.skip(16)?,
10038                _ => return Err(invalid("a stripe sum has an unknown tag")),
10039            }
10040        }
10041    }
10042    if total != rows {
10043        return Err(invalid("table row count differs from stripes"));
10044    }
10045    if cur.done() {
10046        return Ok(None);
10047    }
10048    let magic = cur.take(8)?;
10049    let spanned = magic == FREQUENCIES_SPANS;
10050    let values = magic == FREQUENCIES || spanned;
10051    if !values && magic != FREQUENCIES_V2 {
10052        return Err(invalid("directory extension magic differs"));
10053    }
10054    if cur.u16()? as usize != width {
10055        return Err(invalid("frequency column count differs"));
10056    }
10057    for _ in 0..wanted {
10058        if spanned {
10059            if let Some((length, _)) = summary_span(&mut cur)? {
10060                cur.skip(length)?;
10061            }
10062        } else {
10063            skip_summary(&mut cur, values, rows)?;
10064        }
10065    }
10066    let summary = if spanned {
10067        let Some((length, entries)) = summary_span(&mut cur)? else {
10068            return Ok(None);
10069        };
10070        let start = cur.at;
10071        let summary = decode_summary(&mut cur, &fields[wanted], rows, values)?
10072            .ok_or_else(|| invalid("a stored synopsis is missing"))?;
10073        if cur.at - start != length || summary.entries.len() != entries {
10074            return Err(invalid("a stored synopsis differs from its directory span"));
10075        }
10076        summary
10077    } else {
10078        let Some(summary) = decode_summary(&mut cur, &fields[wanted], rows, values)? else {
10079            return Ok(None);
10080        };
10081        summary
10082    };
10083    let zero = summary
10084        .entries
10085        .iter()
10086        .find(|entry| entry.value == FrequencyValue::Integer(0))
10087        .map(|entry| entry.count)
10088        .or_else(|| (summary.omitted_max == 0).then_some(0));
10089    Ok(zero.and_then(|zero| (rows as u64).checked_sub(nulls)?.checked_sub(zero)))
10090}
10091
10092/// Walks the row-oriented directory while retaining only one column's index and page spans.
10093/// The catalog supplies the schema and the caller checks the complete directory checksum first.
10094fn quick_integer_fold(
10095    file: &File,
10096    mut cur: Cursor<'_>,
10097    entry: &Entry,
10098    size: u64,
10099    wanted: usize,
10100    emit: &mut impl FnMut(i64, u64) -> Result<()>,
10101) -> Result<()> {
10102    let name = &entry.name;
10103    let fields = &entry.fields;
10104    let rows = entry.rows;
10105    if cur.take(8)? != DIRECTORY || cur.text()? != name.as_str() {
10106        return Err(invalid("table directory differs from the catalog"));
10107    }
10108    let width = cur.u16()? as usize;
10109    if width != fields.len() {
10110        return Err(invalid("table directory width differs from the catalog"));
10111    }
10112    for field in fields {
10113        let stored =
10114            Field { name: cur.text()?, ty: read_type(&mut cur)?, not_null: cur.u8()? != 0 };
10115        if &stored != field {
10116            return Err(invalid("table directory schema differs from the catalog"));
10117        }
10118    }
10119    let mut dictionaries = Vec::with_capacity(width);
10120    for field in fields {
10121        dictionaries.push(match cur.u8()? {
10122            0 => false,
10123            tag if coded_type(&field.ty) && tag == dictionary_tag(&field.ty) => {
10124                cur.skip(20)?;
10125                true
10126            }
10127            _ => return Err(invalid("dictionary page tag differs")),
10128        });
10129    }
10130    for _ in 0..width {
10131        match cur.u8()? {
10132            0 => {}
10133            1 => cur.skip(8)?,
10134            _ => return Err(invalid("distinct count tag differs")),
10135        }
10136    }
10137    if cur.u64()? != rows as u64 {
10138        return Err(invalid("table row count differs from the catalog"));
10139    }
10140    let stripes = cur.u32()? as usize;
10141    let mut total = 0_usize;
10142    let mut bytes = Vec::new();
10143    for _ in 0..stripes {
10144        let parts = cur.u32()? as usize;
10145        if parts == 0 || parts > STRIPE_PARTS {
10146            return Err(invalid("stripe part count is outside its bound"));
10147        }
10148        let mut part_rows = Vec::with_capacity(parts);
10149        for _ in 0..parts {
10150            let count = cur.u32()? as usize;
10151            if count == 0 {
10152                return Err(invalid("empty part"));
10153            }
10154            total = total.checked_add(count).ok_or_else(|| invalid("stripe row count overflow"))?;
10155            part_rows.push(count);
10156        }
10157        let index = Span { offset: cur.u64()?, length: cur.u32()? };
10158        let section = index_section(parts)?;
10159        let index_length =
10160            section.checked_mul(width).ok_or_else(|| invalid("index page length overflow"))?;
10161        if index.offset < HEADER
10162            || index.offset.checked_add(u64::from(index.length)).is_none_or(|end| end > size)
10163            || index.length as usize != index_length
10164        {
10165            return Err(invalid("index page range is outside the file"));
10166        }
10167        cur.skip(wanted * 12)?;
10168        let page = Span { offset: cur.u64()?, length: cur.u32()? };
10169        if page.offset < HEADER
10170            || page.offset.checked_add(u64::from(page.length)).is_none_or(|end| end > size)
10171            || page.length as usize > MAX_PAGE
10172        {
10173            return Err(invalid("column page range is outside the file"));
10174        }
10175        cur.skip((width - wanted - 1) * 12)?;
10176        for (field, held) in fields.iter().zip(&dictionaries) {
10177            if coded_type(&field.ty) && *held {
10178                cur.skip(20)?;
10179            }
10180        }
10181        for _ in 0..width * 2 {
10182            match cur.u8()? {
10183                0 => {}
10184                1 => cur.skip(20)?,
10185                _ => return Err(invalid("stripe page tag differs")),
10186            }
10187        }
10188        for _ in 0..width {
10189            cur.skip_bound()?;
10190            cur.skip_bound()?;
10191            cur.skip(5)?;
10192            match cur.u8()? {
10193                0 => {}
10194                1 => cur.skip(16)?,
10195                _ => return Err(invalid("a stripe sum has an unknown tag")),
10196            }
10197        }
10198        let spans = read_index_span(file, index, page, parts, wanted)?;
10199        for (span, expected_rows) in spans.into_iter().zip(part_rows) {
10200            bytes.resize(span.length, 0);
10201            let at = page
10202                .offset
10203                .checked_add(span.start as u64)
10204                .ok_or_else(|| invalid("part range overflow"))?;
10205            read_at(file, at, &mut bytes)?;
10206            if checksum(&bytes) != span.hash {
10207                return Err(invalid("integer part checksum differs"));
10208            }
10209            if bytes.first() == Some(&5) && bytes.get(1) == Some(&0) {
10210                let decoded_rows = integer::fold(&bytes[2..], |value, count| {
10211                    check_integer_tally_value(value, &fields[wanted].ty)?;
10212                    emit(value, count)
10213                })?;
10214                if decoded_rows != expected_rows {
10215                    return Err(invalid("encoded integer part holds the wrong number of rows"));
10216                }
10217            } else {
10218                let column = decode(&fields[wanted].ty, expected_rows, &bytes, None)?;
10219                if let Some(packed) = column.packed_parts() {
10220                    let validity = column.validity();
10221                    let all_valid = column.none_null();
10222                    let base = packed.base();
10223                    let mut codes = [0_u64; 64];
10224                    for from in (0..expected_rows).step_by(codes.len()) {
10225                        let count = (expected_rows - from).min(codes.len());
10226                        packed.unpack(from, &mut codes[..count]);
10227                        for (offset, &code) in codes[..count].iter().enumerate() {
10228                            if all_valid || validity.is_valid(from + offset) {
10229                                // Vector::packed checked that this entire range fits the type.
10230                                emit((base + i128::from(code)) as i64, 1)?;
10231                            }
10232                        }
10233                    }
10234                    continue;
10235                }
10236                let column = column.into_flat()?;
10237                let validity = column.validity();
10238                macro_rules! count_decoded {
10239                    ($values:expr) => {
10240                        for (row, &value) in $values.as_slice().iter().enumerate() {
10241                            if validity.is_valid(row) {
10242                                emit(i64::from(value), 1)?;
10243                            }
10244                        }
10245                    };
10246                }
10247                match column.data() {
10248                    Some(Data::Int8(values)) => count_decoded!(values),
10249                    Some(Data::Int16(values)) => count_decoded!(values),
10250                    Some(Data::Int32(values)) => count_decoded!(values),
10251                    Some(Data::Int64(values)) => count_decoded!(values),
10252                    _ => return Err(invalid("decoded integer part has the wrong type")),
10253                }
10254            }
10255        }
10256    }
10257    if total != rows {
10258        return Err(invalid("table row count differs from stripes"));
10259    }
10260    Ok(())
10261}
10262
10263fn check_integer_tally_value(value: i64, ty: &LogicalType) -> Result<()> {
10264    let fits = match ty {
10265        LogicalType::TinyInt => i8::try_from(value).is_ok(),
10266        LogicalType::SmallInt => i16::try_from(value).is_ok(),
10267        LogicalType::Integer => i32::try_from(value).is_ok(),
10268        LogicalType::BigInt => true,
10269        _ => false,
10270    };
10271    if fits { Ok(()) } else { Err(invalid("encoded integer value is outside its column type")) }
10272}
10273
10274fn decode_directory(bytes: &[u8], size: u64) -> Result<Table> {
10275    read_directory(Cursor::new(bytes), size, None)
10276}
10277
10278/// A directory out of `cur`, which is a whole one in memory or one being read out of the file.
10279///
10280/// `stored_at` is where the directory starts in the file when it is being read out of it, and then
10281/// every frequency synopsis is checked and left there, as [`Frequencies::Stored`].
10282fn read_directory(mut cur: Cursor<'_>, size: u64, stored_at: Option<u64>) -> Result<Table> {
10283    if cur.take(8)? != DIRECTORY {
10284        return Err(invalid("directory magic differs"));
10285    }
10286    let name = cur.text()?;
10287    let width = cur.u16()? as usize;
10288    let mut fields = Vec::with_capacity(width);
10289    for _ in 0..width {
10290        let name = cur.text()?;
10291        let ty = read_type(&mut cur)?;
10292        let not_null = match cur.u8()? {
10293            0 => false,
10294            1 => true,
10295            _ => return Err(invalid("nullability flag differs")),
10296        };
10297        fields.push(Field { name, ty, not_null });
10298    }
10299    let mut dictionaries = Vec::with_capacity(width);
10300    for field in &fields {
10301        dictionaries.push(match cur.u8()? {
10302            0 => None,
10303            tag if tag == dictionary_tag(&field.ty) => {
10304                let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
10305                let end = page
10306                    .offset
10307                    .checked_add(u64::from(page.length))
10308                    .ok_or_else(|| invalid("dictionary page offset overflow"))?;
10309                // A global dictionary covers a whole column, not one bounded stripe. Its lazy
10310                // payload is intentionally allowed to grow past `MAX_PAGE`; only ordinary column
10311                // pages are capped there. `Writer::finish` has already bounded this length by the
10312                // on-disk `u32`, and the range check below keeps it inside the file.
10313                if page.offset < HEADER || end > size {
10314                    return Err(invalid("dictionary page range is outside the file"));
10315                }
10316                Some(page)
10317            }
10318            _ => return Err(invalid("dictionary page tag differs")),
10319        });
10320    }
10321    let mut distincts = Vec::with_capacity(width);
10322    for _ in 0..width {
10323        distincts.push(match cur.u8()? {
10324            0 => None,
10325            1 => Some(cur.u64()?),
10326            _ => return Err(invalid("distinct count tag differs")),
10327        });
10328    }
10329    let rows = usize::try_from(cur.u64()?).map_err(|_| invalid("row count does not fit"))?;
10330    let count = cur.u32()? as usize;
10331    let mut stripes = Vec::with_capacity(count);
10332    let mut total = 0_usize;
10333    for _ in 0..count {
10334        let count = cur.u32()? as usize;
10335        if count == 0 || count > STRIPE_PARTS {
10336            return Err(invalid("stripe part count is outside its bound"));
10337        }
10338        let mut parts = Vec::with_capacity(count);
10339        let mut stripe_rows = 0_usize;
10340        for _ in 0..count {
10341            let rows = cur.u32()?;
10342            if rows == 0 {
10343                return Err(invalid("empty part"));
10344            }
10345            parts.push(rows);
10346            stripe_rows = stripe_rows
10347                .checked_add(rows as usize)
10348                .ok_or_else(|| invalid("stripe row count overflow"))?;
10349        }
10350        total =
10351            total.checked_add(stripe_rows).ok_or_else(|| invalid("stripe row count overflow"))?;
10352        let index = Span { offset: cur.u64()?, length: cur.u32()? };
10353        let section = index_section(count)?;
10354        let wanted = section
10355            .checked_mul(width)
10356            .and_then(|bytes| u32::try_from(bytes).ok())
10357            .ok_or_else(|| invalid("index page length overflow"))?;
10358        let end = index
10359            .offset
10360            .checked_add(u64::from(index.length))
10361            .ok_or_else(|| invalid("index page offset overflow"))?;
10362        if index.offset < HEADER || end > size || index.length != wanted {
10363            return Err(invalid("index page range is outside the file"));
10364        }
10365        let mut pages = Vec::with_capacity(width);
10366        for _ in 0..width {
10367            let offset = cur.u64()?;
10368            let length = cur.u32()?;
10369            let end = offset
10370                .checked_add(u64::from(length))
10371                .ok_or_else(|| invalid("page offset overflow"))?;
10372            if offset < HEADER || end > size || length as usize > MAX_PAGE {
10373                return Err(invalid("page range is outside the file"));
10374            }
10375            pages.push(Span { offset, length });
10376        }
10377        let mut memberships = vec![None; width];
10378        for (column, field) in fields.iter().enumerate() {
10379            if !coded_type(&field.ty) || dictionaries[column].is_none() {
10380                continue;
10381            }
10382            let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
10383            let end = page
10384                .offset
10385                .checked_add(u64::from(page.length))
10386                .ok_or_else(|| invalid("membership page offset overflow"))?;
10387            if page.offset < HEADER || end > size || page.length as usize > MAX_PAGE {
10388                return Err(invalid("membership page range is outside the file"));
10389            }
10390            // No bytes is a stripe written after the column's dictionary was demoted, see
10391            // [`DEMOTED`], which is checked once the block that says so has been read.
10392            if page.length != 0 {
10393                memberships[column] = Some(page);
10394            }
10395        }
10396        let mut sieves = vec![None; width];
10397        for sieve in sieves.iter_mut().take(width) {
10398            match cur.u8()? {
10399                0 => continue,
10400                1 => {}
10401                _ => return Err(invalid("a sieve page has an unknown tag")),
10402            }
10403            let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
10404            let end = page
10405                .offset
10406                .checked_add(u64::from(page.length))
10407                .ok_or_else(|| invalid("sieve page offset overflow"))?;
10408            if page.offset < HEADER || end > size || page.length as usize > MAX_PAGE {
10409                return Err(invalid("sieve page range is outside the file"));
10410            }
10411            *sieve = Some(page);
10412        }
10413        let mut part_ranges = vec![None; width];
10414        for held in part_ranges.iter_mut().take(width) {
10415            match cur.u8()? {
10416                0 => continue,
10417                1 => {}
10418                _ => return Err(invalid("a part range page has an unknown tag")),
10419            }
10420            let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
10421            let end = page
10422                .offset
10423                .checked_add(u64::from(page.length))
10424                .ok_or_else(|| invalid("part range page offset overflow"))?;
10425            if page.offset < HEADER || end > size || page.length as usize > MAX_PAGE {
10426                return Err(invalid("part range page range is outside the file"));
10427            }
10428            *held = Some(page);
10429        }
10430        let mut ranges = Vec::with_capacity(width);
10431        for column in 0..width {
10432            let low = cur.bound()?;
10433            let high = cur.bound()?;
10434            let nulls = cur.u32()? as usize;
10435            if nulls > stripe_rows {
10436                return Err(invalid("null count exceeds stripe rows"));
10437            }
10438            let exact = cur.u8()? != 0;
10439            let sum = match cur.u8()? {
10440                0 => None,
10441                1 => Some(i128::from_le_bytes(
10442                    cur.take(16)?.try_into().map_err(|_| invalid("a stripe sum is truncated"))?,
10443                )),
10444                _ => return Err(invalid("a stripe sum has an unknown tag")),
10445            };
10446            // Files written before the ends of a decimal or a timestamp column carried their power
10447            // of ten hold a bare integer here, and that integer is the one the column holds, which
10448            // is what the power is over. So the type puts it back on the way in and an old file
10449            // prunes as well as a new one. A file that already wrote the power keeps it, because
10450            // this leaves anything that is not an integer alone.
10451            let ty = &fields.get(column).ok_or_else(|| invalid("a stripe range has no column"))?.ty;
10452            let low = low.map(|bound| scaled_as(bound, ty));
10453            let high = high.map(|bound| scaled_as(bound, ty));
10454            ranges.push(Range { low, high, nulls, exact, sum });
10455        }
10456        stripes.push(Stripe {
10457            rows: stripe_rows,
10458            parts,
10459            index,
10460            pages,
10461            memberships: Pages::from_slots(memberships)?,
10462            sieves: Pages::from_slots(sieves)?,
10463            part_ranges: Pages::from_slots(part_ranges)?,
10464            zone: Zone::from_ranges(ranges),
10465        });
10466    }
10467    if total != rows {
10468        return Err(invalid("table row count differs from stripes"));
10469    }
10470    // How many entries each column's synopsis lists, which is all a pair summary is checked against,
10471    // kept apart because the synopses themselves may be left in the file.
10472    let mut entry_counts = vec![0; width];
10473    let frequencies = if cur.done() {
10474        vec![None; width]
10475    } else {
10476        let frequency_magic = cur.take(8)?;
10477        let spanned = frequency_magic == FREQUENCIES_SPANS;
10478        let frequency_values = frequency_magic == FREQUENCIES || spanned;
10479        if !frequency_values && frequency_magic != FREQUENCIES_V2 {
10480            return Err(invalid("directory extension magic differs"));
10481        }
10482        if cur.u16()? as usize != width {
10483            return Err(invalid("frequency column count differs"));
10484        }
10485        let mut frequencies = Vec::with_capacity(width);
10486        for (field, entry_count) in fields.iter().zip(&mut entry_counts) {
10487            if spanned {
10488                let Some((length, entries)) = summary_span(&mut cur)? else {
10489                    frequencies.push(None);
10490                    continue;
10491                };
10492                *entry_count = entries;
10493                let start = cur.at;
10494                if let Some(offset) = stored_at {
10495                    cur.skip(length)?;
10496                    frequencies.push(Some(Frequencies::Stored {
10497                        span: Span {
10498                            offset: offset
10499                                .checked_add(start as u64)
10500                                .ok_or_else(|| invalid("frequency synopsis offset overflow"))?,
10501                            length: u32::try_from(length)
10502                                .map_err(|_| invalid("a frequency synopsis is too long"))?,
10503                        },
10504                        values: true,
10505                        entries,
10506                    }));
10507                } else {
10508                    let summary = decode_summary(&mut cur, field, rows, true)?
10509                        .ok_or_else(|| invalid("a stored synopsis is missing"))?;
10510                    if cur.at - start != length || summary.entries.len() != entries {
10511                        return Err(invalid("a stored synopsis differs from its directory span"));
10512                    }
10513                    frequencies.push(Some(Frequencies::Held(summary)));
10514                }
10515                continue;
10516            }
10517            let start = cur.at;
10518            let summary = decode_summary(&mut cur, field, rows, frequency_values)?;
10519            *entry_count = summary.as_ref().map_or(0, |summary| summary.entries.len());
10520            frequencies.push(match (summary, stored_at) {
10521                (None, _) => None,
10522                (Some(summary), None) => Some(Frequencies::Held(summary)),
10523                (Some(summary), Some(offset)) => Some(Frequencies::Stored {
10524                    span: Span {
10525                        offset: offset + start as u64,
10526                        length: u32::try_from(cur.at - start)
10527                            .map_err(|_| invalid("a frequency synopsis is too long"))?,
10528                    },
10529                    values: frequency_values,
10530                    entries: summary.entries.len(),
10531                }),
10532            });
10533        }
10534        frequencies
10535    };
10536    // There are two optional trailing blocks now rather than one, so the reader dispatches on the
10537    // magic it finds rather than on where the bytes ran out. That is what lets the two arrive
10538    // independently: a format 22 directory ends here and has neither, a directory written before
10539    // the section table has only the clustering declaration, and each one still opens without a
10540    // rewrite. It is the G1 exit criterion, which is that a reader that knows about sections opens
10541    // a file that predates them and answers every query, only without the graph path.
10542    //
10543    // A repeated block is refused rather than allowed to win, because two clustering declarations
10544    // in one directory is a torn directory and the only question is which of them is the lie.
10545    let mut clustering = None;
10546    let mut sections = Vec::new();
10547    let mut pair_frequencies = Vec::new();
10548    let mut seen_pair_frequencies = false;
10549    let mut ordinal_bounds = Vec::new();
10550    let mut seen_ordinal_bounds = false;
10551    let mut frequency_texts = vec![Vec::new(); width];
10552    let mut seen_frequency_texts = false;
10553    let mut host_groups = None;
10554    let mut demoted = Vec::new();
10555    let mut seen_sections = false;
10556    let mut dictionary_payloads = Vec::new();
10557    let mut seen_payloads = false;
10558    let mut constraints = Constraints::default();
10559    // Zero until a section table says otherwise, which is what a format 22 table gets and what
10560    // makes every section stamp fail to match on one, because real generations start at one.
10561    let mut generation = 0;
10562    while !cur.done() {
10563        let mut tag = [0u8; 8];
10564        tag.copy_from_slice(cur.take(8)?);
10565        if &tag == PAIR_FREQUENCIES {
10566            if seen_pair_frequencies {
10567                return Err(invalid("directory names two pair frequency blocks"));
10568            }
10569            seen_pair_frequencies = true;
10570            let count = cur.u16()? as usize;
10571            if count > MAX_PAIR_FREQUENCIES {
10572                return Err(invalid("pair frequency count exceeds its bound"));
10573            }
10574            pair_frequencies = Vec::with_capacity(count);
10575            for _ in 0..count {
10576                let first = cur.u16()?;
10577                let second = cur.u16()?;
10578                let first_at = first as usize;
10579                let second_at = second as usize;
10580                if frequencies.get(first_at).and_then(Option::as_ref).is_none() {
10581                    return Err(invalid("pair frequency first column has no synopsis"));
10582                }
10583                let first_entries = entry_counts[first_at];
10584                if !matches!(fields.get(second_at), Some(field) if field.ty == LogicalType::Varchar)
10585                    || dictionaries.get(second_at).copied().flatten().is_none()
10586                {
10587                    return Err(invalid("pair frequency second column has no stable dictionary"));
10588                }
10589                if pair_frequencies
10590                    .iter()
10591                    .any(|held: &PairFrequencySummary| held.first == first && held.second == second)
10592                {
10593                    return Err(invalid("directory repeats a pair frequency summary"));
10594                }
10595                let omitted_max = cur.u64()?;
10596                if omitted_max > rows as u64 {
10597                    return Err(invalid("pair frequency omitted count exceeds the table"));
10598                }
10599                let entries_count = cur.u16()? as usize;
10600                if entries_count > FREQUENCY_ENTRIES {
10601                    return Err(invalid("pair frequency entry count exceeds its bound"));
10602                }
10603                let mut entries = Vec::with_capacity(entries_count);
10604                for _ in 0..entries_count {
10605                    let first_entry = cur.u16()?;
10606                    if first_entry as usize >= first_entries {
10607                        return Err(invalid("pair frequency anchor is outside its synopsis"));
10608                    }
10609                    let second = match cur.u8()? {
10610                        0 => None,
10611                        1 => Some(cur.u32()?),
10612                        _ => return Err(invalid("pair frequency string tag differs")),
10613                    };
10614                    let count = cur.u64()?;
10615                    if count == 0 || count > rows as u64 {
10616                        return Err(invalid("pair frequency count is outside the table"));
10617                    }
10618                    entries.push(PairFrequencyEntry { first_entry, second, count });
10619                }
10620                if entries.windows(2).any(|pair| pair[0].count < pair[1].count) {
10621                    return Err(invalid("pair frequency entries are not descending"));
10622                }
10623                pair_frequencies.push(PairFrequencySummary { first, second, entries, omitted_max });
10624            }
10625        } else if &tag == ORDINAL_BOUNDS {
10626            if seen_ordinal_bounds {
10627                return Err(invalid("directory names two ordinal bound blocks"));
10628            }
10629            seen_ordinal_bounds = true;
10630            ordinal_bounds = vec![0; width];
10631            let count = cur.u16()? as usize;
10632            if count > width {
10633                return Err(invalid("ordinal bound count exceeds the columns"));
10634            }
10635            for _ in 0..count {
10636                let column = cur.u16()? as usize;
10637                let bound = cur.u64()?;
10638                if column >= width || frequencies.get(column).and_then(Option::as_ref).is_none() {
10639                    return Err(invalid("ordinal bound names a column with no synopsis"));
10640                }
10641                if bound == 0 || bound > rows as u64 || ordinal_bounds[column] != 0 {
10642                    return Err(invalid("ordinal bound is outside the table or repeated"));
10643                }
10644                ordinal_bounds[column] = bound;
10645            }
10646        } else if &tag == FREQUENCY_TEXTS {
10647            if seen_frequency_texts {
10648                return Err(invalid("directory names two frequency text blocks"));
10649            }
10650            seen_frequency_texts = true;
10651            let columns = cur.u16()? as usize;
10652            if columns > width {
10653                return Err(invalid("frequency text column count exceeds the schema"));
10654            }
10655            for _ in 0..columns {
10656                let column = cur.u16()? as usize;
10657                if !frequency_texts.get(column).is_some_and(Vec::is_empty) {
10658                    return Err(invalid("frequency text column is repeated or out of range"));
10659                }
10660                if !matches!(fields.get(column), Some(field) if coded_type(&field.ty))
10661                    || dictionaries.get(column).copied().flatten().is_none()
10662                    || frequencies.get(column).and_then(Option::as_ref).is_none()
10663                {
10664                    return Err(invalid("frequency texts belong to a non-string synopsis"));
10665                }
10666                let count = cur.u16()? as usize;
10667                if count == 0 || count != entry_counts[column] {
10668                    return Err(invalid("frequency text count differs from its synopsis"));
10669                }
10670                let mut texts = Vec::with_capacity(count);
10671                for _ in 0..count {
10672                    texts.push(match cur.u8()? {
10673                        0 => None,
10674                        1 => {
10675                            let length = cur.u32()? as usize;
10676                            let bytes = cur.take(length)?.to_vec();
10677                            if fields[column].ty == LogicalType::Varchar {
10678                                std::str::from_utf8(&bytes)
10679                                    .map_err(|_| invalid("frequency text is not UTF-8"))?;
10680                            }
10681                            Some(bytes)
10682                        }
10683                        _ => return Err(invalid("frequency text tag differs")),
10684                    });
10685                }
10686                frequency_texts[column] = texts;
10687            }
10688        } else if &tag == HOST_GROUPS {
10689            if host_groups.is_some() {
10690                return Err(invalid("directory names two host group blocks"));
10691            }
10692            let column = cur.u16()? as usize;
10693            if !matches!(fields.get(column), Some(field) if field.ty == LogicalType::Varchar)
10694                || dictionaries.get(column).copied().flatten().is_none()
10695            {
10696                return Err(invalid("host groups belong to a non-string dictionary"));
10697            }
10698            let omitted_max = cur.u64()?;
10699            if omitted_max > rows as u64 {
10700                return Err(invalid("host group bound exceeds the table"));
10701            }
10702            let count = cur.u16()? as usize;
10703            if count > host::CAPACITY {
10704                return Err(invalid("host group count exceeds its bound"));
10705            }
10706            let mut entries = Vec::with_capacity(count);
10707            let mut bytes = 0_usize;
10708            for _ in 0..count {
10709                let host_len = cur.u32()? as usize;
10710                bytes =
10711                    bytes.checked_add(host_len).ok_or_else(|| invalid("host bytes overflow"))?;
10712                if bytes > host::BYTE_BUDGET {
10713                    return Err(invalid("host groups exceed their byte budget"));
10714                }
10715                let host = std::str::from_utf8(cur.take(host_len)?)
10716                    .map_err(|_| invalid("host is not UTF-8"))?
10717                    .to_owned();
10718                let count = cur.u64()?;
10719                if count == 0 || count > rows as u64 {
10720                    return Err(invalid("host group count exceeds the table"));
10721                }
10722                let bytes_sum = i128::from_le_bytes(
10723                    cur.take(16)?
10724                        .try_into()
10725                        .map_err(|_| invalid("host length sum is truncated"))?,
10726                );
10727                if bytes_sum < 0 {
10728                    return Err(invalid("host length sum is negative"));
10729                }
10730                let minimum_len = cur.u32()? as usize;
10731                bytes =
10732                    bytes.checked_add(minimum_len).ok_or_else(|| invalid("host bytes overflow"))?;
10733                if bytes > host::BYTE_BUDGET {
10734                    return Err(invalid("host groups exceed their byte budget"));
10735                }
10736                let minimum = std::str::from_utf8(cur.take(minimum_len)?)
10737                    .map_err(|_| invalid("host minimum is not UTF-8"))?
10738                    .to_owned();
10739                entries.push(host::HostEntry { host, count, bytes_sum, minimum });
10740            }
10741            if entries.windows(2).any(|pair| pair[0].count < pair[1].count)
10742                || entries.iter().any(|entry| entry.host.is_empty() || entry.minimum.is_empty())
10743            {
10744                return Err(invalid("host groups are not in certified order"));
10745            }
10746            host_groups = Some(host::HostSummary { column, omitted_max, entries });
10747        } else if &tag == CLUSTERING {
10748            if clustering.is_some() {
10749                return Err(invalid("directory names two clustering declarations"));
10750            }
10751            let bucket = Width::from_tag(cur.u8()?)
10752                .ok_or_else(|| invalid("clustering width tag differs"))?;
10753            let count = cur.u16()? as usize;
10754            let mut columns = Vec::with_capacity(count.min(fields.len()));
10755            for _ in 0..count {
10756                columns.push(u32::from(cur.u16()?));
10757            }
10758            // Through the constructor and not built by hand, so that a file claiming a column the
10759            // table does not have is caught at open rather than at the first scan that trusted it.
10760            clustering = Some(Clustering::new(columns, bucket, &fields).map_err(|_| {
10761                invalid("stored clustering declaration does not match the table it is on")
10762            })?);
10763        } else if &tag == DEMOTED {
10764            if !demoted.is_empty() {
10765                return Err(invalid("directory names two demoted column blocks"));
10766            }
10767            let count = cur.u16()? as usize;
10768            if count == 0 || count > width {
10769                return Err(invalid("demoted column count is outside the schema"));
10770            }
10771            demoted = vec![false; width];
10772            for _ in 0..count {
10773                let column = cur.u16()? as usize;
10774                if dictionaries.get(column).copied().flatten().is_none() || demoted[column] {
10775                    return Err(invalid("a demoted column is repeated or has no dictionary"));
10776                }
10777                demoted[column] = true;
10778            }
10779        } else if &tag == SECTIONS {
10780            if seen_sections {
10781                return Err(invalid("directory names two section tables"));
10782            }
10783            seen_sections = true;
10784            generation = cur.u64()?;
10785            let count = cur.u16()? as usize;
10786            if count > MAX_SECTIONS {
10787                return Err(invalid("section count exceeds its bound"));
10788            }
10789            sections = Vec::with_capacity(count);
10790            // entry at a time: a malformed section entry is refused rather than turned into an
10791            // offset.
10792            for _ in 0..count {
10793                sections.push(Section::decode(cur.take(section::ENTRY_BYTES)?)?);
10794            }
10795            for held in &sections {
10796                let Some(end) = held.extent_page.checked_add(u64::from(held.extent_bytes)) else {
10797                    return Err(invalid("a section's extent table overflows the file"));
10798                };
10799                // The bound check is here and not in `section`, because only the caller knows how
10800                // big the file is. A section pointing past the end is a torn directory, and reading
10801                // the payload it names would be reading whatever else is at that offset.
10802                if held.extent_bytes != 0 && (held.extent_page < HEADER || end > size) {
10803                    return Err(invalid("a section's extent table is outside the file"));
10804                }
10805                if held.extents == 0 && held.extent_bytes != 0 {
10806                    return Err(invalid("a section with no extents names an extent table"));
10807                }
10808            }
10809        } else if &tag == DICTIONARY_PAYLOADS {
10810            if seen_payloads {
10811                return Err(invalid("directory names two dictionary payload blocks"));
10812            }
10813            seen_payloads = true;
10814            let count = cur.u16()? as usize;
10815            if count != fields.len() {
10816                return Err(invalid("dictionary payload block does not match the table's columns"));
10817            }
10818            dictionary_payloads = Vec::with_capacity(count);
10819            for _ in 0..count {
10820                let bytes = cur.u64()?;
10821                if bytes > size {
10822                    return Err(invalid("a dictionary payload is larger than the file"));
10823                }
10824                dictionary_payloads.push(bytes);
10825            }
10826        } else if &tag == KEYS {
10827            if !constraints.is_empty() {
10828                return Err(invalid("directory names two key blocks"));
10829            }
10830            let fits = |columns: &[u16]| {
10831                !columns.is_empty() && columns.iter().all(|&column| usize::from(column) < width)
10832            };
10833            let count = cur.u16()? as usize;
10834            for _ in 0..count {
10835                let primary = cur.u8()? != 0;
10836                let columns = columns_of(&mut cur)?;
10837                if !fits(&columns) {
10838                    return Err(invalid("a stored key names a column the table does not have"));
10839                }
10840                constraints.keys.push((columns, primary));
10841            }
10842            let count = cur.u16()? as usize;
10843            for _ in 0..count {
10844                let columns = columns_of(&mut cur)?;
10845                let referenced = columns_of(&mut cur)?;
10846                let len = cur.u32()? as usize;
10847                let table = std::str::from_utf8(cur.take(len)?)
10848                    .map_err(|_| invalid("a foreign key's table name is not UTF-8"))?
10849                    .to_owned();
10850                if !fits(&columns) || referenced.len() != columns.len() || table.is_empty() {
10851                    return Err(invalid("a stored foreign key does not match its table"));
10852                }
10853                constraints.foreign.push(StoredForeign { columns, table, referenced });
10854            }
10855            if constraints.is_empty() {
10856                return Err(invalid("a key block holds no key"));
10857            }
10858        } else {
10859            return Err(invalid("directory extension magic differs"));
10860        }
10861    }
10862    if !cur.done() {
10863        return Err(invalid("directory has trailing bytes"));
10864    }
10865    for stripe in &stripes {
10866        for (column, field) in fields.iter().enumerate() {
10867            if coded_type(&field.ty)
10868                && dictionaries[column].is_some()
10869                && stripe.memberships.get(column).is_none()
10870                && !demoted.get(column).copied().unwrap_or(false)
10871            {
10872                return Err(invalid("string page has no code membership index"));
10873            }
10874        }
10875    }
10876    Ok(Table {
10877        name,
10878        fields,
10879        stripes,
10880        rows,
10881        dictionaries,
10882        dictionary_payloads,
10883        demoted,
10884        distincts,
10885        frequencies,
10886        ordinal_bounds,
10887        pair_frequencies,
10888        frequency_texts,
10889        host_groups,
10890        clustering,
10891        generation,
10892        sections,
10893        constraints,
10894    })
10895}
10896
10897/// How many keys or columns follow, in the key block.
10898fn put_count(out: &mut Vec<u8>, count: usize) -> Result<()> {
10899    put_u16(out, u16::try_from(count).map_err(|_| invalid("too many constraints"))?);
10900    Ok(())
10901}
10902
10903/// A count and then that many column places, the layout the key block uses for every list.
10904fn put_columns(out: &mut Vec<u8>, columns: &[u16]) -> Result<()> {
10905    put_count(out, columns.len())?;
10906    for &column in columns {
10907        put_u16(out, column);
10908    }
10909    Ok(())
10910}
10911
10912/// What [`put_columns`] wrote, for a list of columns.
10913fn columns_of(cur: &mut Cursor<'_>) -> Result<Vec<u16>> {
10914    let count = cur.u16()? as usize;
10915    (0..count).map(|_| cur.u16()).collect()
10916}
10917
10918/// A zone map's end, in the layout `rudb_common::bounds` defines. See [`Cursor::bound`].
10919fn put_bound(out: &mut Vec<u8>, bound: Option<&Bound>) -> Result<()> {
10920    bounds::put(out, bound)
10921}
10922
10923/// Which cascades are worth trying on a run of dictionary codes.
10924///
10925/// The exhaustive chooser encodes every candidate at every level of a cascade three deep and keeps
10926/// the smallest, which on a part of 1024 codes is around a hundred full encodes to decide something
10927/// three candidates were always going to win. It is the right default for a crate that does not
10928/// know what it is looking at. Here we do know. Codes are counted from zero in the order the values
10929/// were first seen, so a part of them is one value, or a narrow band, or a few long runs, and those
10930/// are constant, frame of reference and run length. Nothing else has ever come first on this data.
10931///
10932/// A dictionary of dictionary codes is the one candidate that can never pay, because the codes are
10933/// already the dictionary, and it is also the most expensive one to try. Below the top level the
10934/// streams are an RLE's run values and run lengths, which are integers in their own right with no
10935/// runs left in them, so only the two flat candidates go down there.
10936///
10937/// This is size given up for time on purpose, and the ablation is this chooser against
10938/// [`chooser::EXHAUSTIVE`] on the same file.
10939#[derive(Debug)]
10940struct Codes;
10941
10942impl chooser::Chooser for Codes {
10943    fn name(&self) -> &'static str {
10944        "codes"
10945    }
10946
10947    fn narrow_strings(
10948        &self,
10949        _values: &[&[u8]],
10950        offered: &[string::Kind],
10951        _depth: u8,
10952    ) -> Vec<string::Kind> {
10953        // Never reached, because nothing here encodes strings through the cascade. The trait asks
10954        // for it and the honest answer to a question we have no opinion on is the whole list.
10955        offered.to_vec()
10956    }
10957
10958    fn narrow_integers(
10959        &self,
10960        _values: &[i64],
10961        offered: &[integer::Kind],
10962        depth: u8,
10963    ) -> Vec<integer::Kind> {
10964        // The contract is a non empty subset, and a chunk that offers none of the three is a chunk
10965        // this has no opinion about rather than one that cannot be written.
10966        narrowed_to(Codes::keep(depth), offered)
10967    }
10968
10969    fn considers_integer(&self, kind: integer::Kind, depth: u8) -> bool {
10970        Codes::keep(depth).contains(&kind)
10971    }
10972}
10973
10974impl Codes {
10975    fn keep(depth: u8) -> &'static [integer::Kind] {
10976        if depth == 0 {
10977            &[integer::Kind::Constant, integer::Kind::Packed, integer::Kind::Rle]
10978        } else {
10979            &[integer::Kind::Constant, integer::Kind::Packed]
10980        }
10981    }
10982}
10983
10984/// The kinds of `offered` that are in `keep`, or all of `offered` when none of them are.
10985///
10986/// `Packed` applies to every chunk and both choosers keep it, so the fallback is never taken on a
10987/// chunk the cascade offers. It is there because the contract is a non empty subset and a chooser
10988/// that returned nothing would be a chunk that cannot be written. It is also why saying no to a kind
10989/// in `considers_integer` is safe: a kind that is never offered could only have been kept through
10990/// this fallback, and the fallback is never reached.
10991fn narrowed_to(keep: &[integer::Kind], offered: &[integer::Kind]) -> Vec<integer::Kind> {
10992    let narrowed: Vec<integer::Kind> =
10993        offered.iter().copied().filter(|kind| keep.contains(kind)).collect();
10994    if narrowed.is_empty() { offered.to_vec() } else { narrowed }
10995}
10996
10997/// Which cascades are worth trying on a part of plain integers.
10998///
10999/// Wider than [`Codes`] because the values are not codes and carry whatever shape the column has.
11000/// A timestamp column climbs, so delta is the one that matters and is the reason this exists at
11001/// all: three timestamp columns in ClickBench were coming out at exactly eight bytes a row with
11002/// nothing asked of them. The same three columns are why the stride is here, since a timestamp
11003/// loaded from a source that recorded whole seconds is microseconds with twenty zero bits under
11004/// every value. A column that is one value with a handful of exceptions is sparse. What is still
11005/// left out is the dictionary, for the same reason as in [`Codes`]: it is the most
11006/// expensive candidate to try and this file already puts the columns that want one through a
11007/// dictionary of their own before they ever reach here.
11008#[derive(Debug)]
11009struct Fixed;
11010
11011impl chooser::Chooser for Fixed {
11012    fn name(&self) -> &'static str {
11013        "fixed"
11014    }
11015
11016    fn narrow_strings(
11017        &self,
11018        _values: &[&[u8]],
11019        offered: &[string::Kind],
11020        _depth: u8,
11021    ) -> Vec<string::Kind> {
11022        offered.to_vec()
11023    }
11024
11025    fn narrow_integers(
11026        &self,
11027        _values: &[i64],
11028        offered: &[integer::Kind],
11029        depth: u8,
11030    ) -> Vec<integer::Kind> {
11031        narrowed_to(Fixed::keep(depth), offered)
11032    }
11033
11034    fn considers_integer(&self, kind: integer::Kind, depth: u8) -> bool {
11035        Fixed::keep(depth).contains(&kind)
11036    }
11037}
11038
11039impl Fixed {
11040    fn keep(depth: u8) -> &'static [integer::Kind] {
11041        if depth == 0 {
11042            &[
11043                integer::Kind::Constant,
11044                integer::Kind::Packed,
11045                integer::Kind::Delta,
11046                integer::Kind::Rle,
11047                integer::Kind::Sparse,
11048                integer::Kind::Strided,
11049            ]
11050        } else {
11051            &[integer::Kind::Constant, integer::Kind::Packed, integer::Kind::Delta]
11052        }
11053    }
11054}
11055
11056/// Every value of an integer part as an `i64`, or `None` for a part this cannot widen without
11057/// losing one.
11058///
11059/// `UBIGINT` is the only integer type left out, because half its range does not fit and a page that
11060/// silently wrapped would be worse than a page that stays plain. Booleans and strings are not
11061/// integers and have their own ways of being small.
11062fn widened(data: &Data) -> Option<Vec<i64>> {
11063    match data {
11064        Data::Int8(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
11065        Data::UInt8(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
11066        Data::Int16(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
11067        Data::UInt16(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
11068        Data::Int32(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
11069        Data::UInt32(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
11070        Data::Int64(values) => Some(values.to_vec()),
11071        _ => None,
11072    }
11073}
11074
11075/// A cascaded page decoded straight into the width the column is declared at.
11076///
11077/// A value that does not fit is a page that disagrees with the directory about what the column is,
11078/// which is a damaged file rather than a caller error, so it is refused rather than truncated. The
11079/// decoder does that check a block at a time where it can, see [`integer::decode_as`].
11080fn cascade(ty: &LogicalType, bytes: &[u8], rows: usize) -> Result<Data> {
11081    fn wanted<T: integer::Lane>(bytes: &[u8], rows: usize) -> Result<Vec<T>> {
11082        let values = integer::decode_as::<T>(bytes)
11083            .map_err(|error| invalid(&format!("page value is not of its type: {error}")))?;
11084        if values.len() != rows {
11085            return Err(invalid("cascade page holds the wrong number of rows"));
11086        }
11087        Ok(values)
11088    }
11089    Ok(match ty {
11090        LogicalType::TinyInt => Data::Int8(wanted::<i8>(bytes, rows)?.into()),
11091        LogicalType::UTinyInt => Data::UInt8(wanted::<u8>(bytes, rows)?.into()),
11092        LogicalType::SmallInt => Data::Int16(wanted::<i16>(bytes, rows)?.into()),
11093        LogicalType::USmallInt => Data::UInt16(wanted::<u16>(bytes, rows)?.into()),
11094        LogicalType::Integer | LogicalType::Date => Data::Int32(wanted::<i32>(bytes, rows)?.into()),
11095        LogicalType::UInteger => Data::UInt32(wanted::<u32>(bytes, rows)?.into()),
11096        LogicalType::BigInt
11097        | LogicalType::Timestamp
11098        | LogicalType::Time
11099        | LogicalType::TimeTz
11100        | LogicalType::TimestampTz
11101        | LogicalType::TimestampS
11102        | LogicalType::TimestampMs
11103        | LogicalType::TimestampNs => Data::Int64(wanted::<i64>(bytes, rows)?.into()),
11104        // A decimal is an integer of unscaled units, so the cascade reads back into whichever
11105        // integer the declared width says the column is stored as.
11106        LogicalType::Decimal { .. } => match ty.physical() {
11107            PhysicalType::Int16 => Data::Int16(wanted::<i16>(bytes, rows)?.into()),
11108            PhysicalType::Int32 => Data::Int32(wanted::<i32>(bytes, rows)?.into()),
11109            PhysicalType::Int64 => Data::Int64(wanted::<i64>(bytes, rows)?.into()),
11110            _ => return Err(invalid("cascade codec belongs to a decimal that is not an integer")),
11111        },
11112        _ => return Err(invalid("cascade codec belongs to a page that is not integers")),
11113    })
11114}
11115
11116/// How many bytes a part of this type costs written out plainly, which is what the cascade has to
11117/// beat before it is worth the decode.
11118fn plain_width(ty: &LogicalType) -> Option<usize> {
11119    Some(match ty {
11120        LogicalType::TinyInt | LogicalType::UTinyInt => 1,
11121        LogicalType::SmallInt | LogicalType::USmallInt => 2,
11122        LogicalType::Integer | LogicalType::UInteger | LogicalType::Date => 4,
11123        LogicalType::BigInt
11124        | LogicalType::Timestamp
11125        | LogicalType::Time
11126        | LogicalType::TimeTz
11127        | LogicalType::TimestampTz
11128        | LogicalType::TimestampS
11129        | LogicalType::TimestampMs
11130        | LogicalType::TimestampNs => 8,
11131        LogicalType::Decimal { .. } => match ty.physical() {
11132            PhysicalType::Int16 => 2,
11133            PhysicalType::Int32 => 4,
11134            PhysicalType::Int64 => 8,
11135            // The widest decimals are stored as `i128`, which the cascade does not widen into, so
11136            // they take the plain path and there is nothing here to compare against.
11137            _ => return None,
11138        },
11139        _ => return None,
11140    })
11141}
11142
11143/// A part's plain integers through the cascade, or `None` when nothing it offers is worth it.
11144///
11145/// What it has to beat is whatever the page would otherwise have cost, which is the bit packed form
11146/// where there is one and the plain width where there is not. Both are cheaper to decode than a
11147/// cascade, so a tie goes to them.
11148fn cascaded(
11149    flat: &Vector,
11150    ty: &LogicalType,
11151    packed: Option<&Packed<'_>>,
11152    settling: &mut Settling,
11153) -> Result<Option<Vec<u8>>> {
11154    let (Some(width), Some(data)) = (plain_width(ty), flat.data()) else { return Ok(None) };
11155    let Some(values) = widened(data) else { return Ok(None) };
11156    let plain = values.len().saturating_mul(width);
11157    let best = match packed {
11158        // The tag, the base, the word count and the words, which is what the codec 2 branch writes.
11159        Some(packed) => plain.min(21 + size_of_val(packed.words())),
11160        None => plain,
11161    };
11162    let out = settling.encode(&values)?;
11163    Ok((out.len() < best).then_some(out))
11164}
11165
11166/// How often the parts of one column in one stripe search the cascade again, in parts.
11167///
11168/// A stripe is 64 parts, so this is four searches a stripe where there were 64. The search is
11169/// what the cascade costs: on ClickBench `hits` the integer cascade was about a tenth of the load's
11170/// CPU and nearly all of it under `encode_pages`, trying six trees on every part to keep the one
11171/// the part before had kept.
11172const SEARCH_EVERY: usize = 16;
11173
11174/// What the parts of one column in one stripe have settled on in the integer cascade, and the
11175/// symbol table its text pages compress against.
11176///
11177/// One of these per column per stripe, used in part order, so what a part comes out as depends on
11178/// the stripe and not on which thread wrote it or on how many there were.
11179#[derive(Debug, Default)]
11180struct Settling {
11181    /// The shape of the last part that was searched, with what its top level offered, its length
11182    /// and its row count, which is the size a replay is held to.
11183    shape: Option<Shape>,
11184    /// Parts replayed since that search.
11185    since: usize,
11186    /// The FSST table of the last text page that trained one. See [`Settling::text`].
11187    symbols: Option<Symbols>,
11188}
11189
11190/// A table trained on one text page, with what that page came to and how many pages have used it
11191/// since.
11192#[derive(Debug)]
11193struct Symbols {
11194    shape: chooser::Settled,
11195    /// The trained page compressed and plain, in bytes, which is the ratio a later page is held to.
11196    /// Zero compressed when the table came out empty.
11197    len: usize,
11198    payload: usize,
11199    since: usize,
11200}
11201
11202impl Settling {
11203    /// A text page as one FSST chunk, against the table an earlier page of the stripe trained where
11204    /// there is one.
11205    ///
11206    /// The same rule as [`Self::encode`]: the table is used for [`SEARCH_EVERY`] pages and is kept
11207    /// while a page comes out no more than a quarter bigger a byte than the page it was trained on.
11208    /// Past that the page trains a table of its own and the pages after it use that one. Every page
11209    /// still carries the table it was compressed with, so nothing a reader does changes.
11210    fn text(&mut self, values: &[&[u8]], payload: usize) -> Result<Option<Vec<u8>>> {
11211        if let Some(symbols) = self.symbols.as_mut().filter(|symbols| symbols.since < SEARCH_EVERY)
11212        {
11213            let out = string::encode_fsst(values, &symbols.shape)?;
11214            // An empty table stays empty for the pages after, which are the same kind of text.
11215            let held = match &out {
11216                None => symbols.len == 0,
11217                Some(out) => {
11218                    (out.len() as u128) * (symbols.payload as u128) * 4
11219                        <= (symbols.len as u128) * (payload as u128) * 5
11220                }
11221            };
11222            if held {
11223                symbols.since += 1;
11224                return Ok(out);
11225            }
11226        }
11227        let shape = string::fsst_shape(values);
11228        let out = string::encode_fsst(values, &shape)?;
11229        let len = out.as_ref().map_or(0, Vec::len);
11230        self.symbols = Some(Symbols { shape, len, payload: payload.max(1), since: 0 });
11231        Ok(out)
11232    }
11233
11234    /// A part's integers through the cascade, replaying the settled shape where there is one.
11235    ///
11236    /// The replay is kept when it held and came out no more than a quarter bigger a row than the
11237    /// part the shape was searched on. Past that the column has changed under it and the part is
11238    /// searched. A replay that stopped fitting partway has already searched from where it stopped,
11239    /// so its shape is taken as the new one rather than searched a second time.
11240    fn encode(&mut self, values: &[i64]) -> Result<Vec<u8>> {
11241        if let Some(shape) = self.shape.as_ref().filter(|_| self.since < SEARCH_EVERY) {
11242            let replay = chooser::Replay::new(&shape.kinds, &Fixed).expecting(&shape.offered);
11243            let out = integer::encode_with(values, &replay)?;
11244            if !replay.held() {
11245                self.settle(&out, values.len(), replay.first_offered())?;
11246                return Ok(out);
11247            }
11248            let grown = (out.len() as u128) * (shape.rows as u128) * 4;
11249            if grown <= (shape.len as u128) * (values.len() as u128) * 5 {
11250                self.since += 1;
11251                return Ok(out);
11252            }
11253        }
11254        // A replay of nothing is the search, and says what the top level offered on the way.
11255        let search = chooser::Replay::new(&[], &Fixed);
11256        let out = integer::encode_with(values, &search)?;
11257        self.settle(&out, values.len(), search.first_offered())?;
11258        Ok(out)
11259    }
11260
11261    fn settle(&mut self, out: &[u8], rows: usize, offered: Vec<integer::Kind>) -> Result<()> {
11262        let kinds = integer::shape(out)?;
11263        self.shape = Some(Shape { kinds, offered, len: out.len().max(1), rows: rows.max(1) });
11264        self.since = 0;
11265        Ok(())
11266    }
11267}
11268
11269/// A searched part's cascade, what its top level was offered, and what it came to.
11270#[derive(Debug)]
11271struct Shape {
11272    kinds: Vec<integer::Kind>,
11273    offered: Vec<integer::Kind>,
11274    len: usize,
11275    rows: usize,
11276}
11277
11278/// A part's dictionary codes through the integer cascade, or `None` when the cascade did not pay.
11279///
11280/// Until now this stream was a `u32` a row with nothing asked of it, and on ClickBench that was
11281/// 400,185,326 bytes for every one of the 28 varchar columns, the same count for `URL` as for a
11282/// column holding the empty string in nearly every row. Codes are dense integers counted from zero
11283/// and a part holds 1024 of them, which is the shape frame of reference is best at, and a column
11284/// with one value everywhere comes back a constant costing nothing per row rather than four bytes.
11285///
11286/// The result is taken only when it is smaller than the plain form. A cascade is allowed to come
11287/// out larger on a part whose codes are genuinely wide, `URL` has about sixty million distinct
11288/// values, and there is no reason to pay for the decode when it does.
11289/// A varchar page as one FSST layer, or `None` when it did not pay.
11290///
11291/// Until now a varchar page that neither the global dictionary nor the per page dictionary claimed
11292/// was written out raw: four bytes of offset a row and then the bytes. That is the right answer for
11293/// a page of values with nothing in common and the wrong one for a page of English, and a column of
11294/// comments is the case this exists for.
11295///
11296/// One layer and not the full string cascade, which is what the payload blocks of a global
11297/// dictionary go through. The cascade is a search: it encodes the page under every candidate it has
11298/// and recurses into the integer cascade for the lengths of each one, and on TPC-H `orders` that
11299/// took the write from 6.9 s to 48.3 s. It reads back no faster than the dictionary it replaced
11300/// either, 1.807 G instructions against 1.810 G for `select o_comment from orders`, because
11301/// unpicking a nest of layers a value at a time costs what the dictionary's payload block decode
11302/// cost. Raw pages of the same column read in 0.686 G, which says the whole of the difference is
11303/// what the page has to be put back together from.
11304///
11305/// FSST alone keeps most of what the cascade found and gives all of that back. Decoding it is one
11306/// pass over the payload into one buffer, the values are laid end to end in it the way the raw form
11307/// already lays them out, and what the reader hands a chunk is views over that buffer.
11308///
11309/// The page dictionary gets first refusal because it is cheaper still, and it wins on a page whose
11310/// values repeat. What is left for this is the page whose values mostly do not, which is exactly the
11311/// page that was being written raw.
11312///
11313/// Taken only when it comes out smaller than the raw form, so a page of incompressible values pays
11314/// nothing at read time for having been offered.
11315///
11316/// The symbol table is trained once for several pages of the stripe rather than once a page. See
11317/// [`Settling::text`].
11318fn text_compressed(flat: &Vector, settling: &mut Settling) -> Result<Option<Vec<u8>>> {
11319    let mut values: Vec<&[u8]> = Vec::with_capacity(flat.len());
11320    let mut payload = 0_usize;
11321    for row in 0..flat.len() {
11322        // bytes_at: the rows were checked for UTF-8 on the way in, and checking them again here
11323        // was most of what the loop cost.
11324        let text = flat.bytes_at(row).unwrap_or(b"");
11325        payload = payload.saturating_add(text.len());
11326        values.push(text);
11327    }
11328    // What codec 0 writes for a varchar page: an offset a row and one more, then the payload.
11329    let plain = (flat.len() + 1).saturating_mul(4).saturating_add(payload);
11330    let Some(out) = settling.text(&values, payload)? else {
11331        return Ok(None);
11332    };
11333    Ok((out.len() < plain).then_some(out))
11334}
11335
11336fn encoded_codes(codes: &[u32]) -> Result<Option<Vec<u8>>> {
11337    let wide: Vec<i64> = codes.iter().map(|code| i64::from(*code)).collect();
11338    let coded = integer::encode_with(&wide, &Codes)?;
11339    let plain = codes.len().saturating_mul(size_of::<u32>());
11340    Ok((coded.len() < plain).then_some(coded))
11341}
11342
11343/// The validity of a page, which is a flag and then, when some rows are null and some are not, a
11344/// bit a row with the valid ones set.
11345fn push_validity(out: &mut Vec<u8>, flat: &Vector) {
11346    let flag = match flat.validity() {
11347        Validity::AllValid => 0,
11348        Validity::AllInvalid => 1,
11349        Validity::Mask(_) => 2,
11350    };
11351    out.push(flag);
11352    if flag == 2 {
11353        for group in (0..flat.len()).step_by(8) {
11354            let mut bits = 0_u8;
11355            for bit in 0..8 {
11356                if group + bit < flat.len() && !flat.is_null_at(group + bit) {
11357                    bits |= 1 << bit;
11358                }
11359            }
11360            out.push(bits);
11361        }
11362    }
11363}
11364
11365/// One part of a column coded against its global dictionary as a page, from the codes and the
11366/// validity [`push_validity`] wrote for it.
11367///
11368/// The codes go through the integer cascade when that comes out smaller than four bytes a code,
11369/// which on a column that repeats itself it nearly always does, and are written as they are when it
11370/// does not.
11371fn coded_page(codes: &[u32], validity: &[u8]) -> Result<Vec<u8>> {
11372    let coded = encoded_codes(codes)?;
11373    let mut out = Vec::with_capacity(
11374        1 + validity.len() + coded.as_ref().map_or(size_of_val(codes), Vec::len),
11375    );
11376    out.push(if coded.is_some() { 4 } else { 3 });
11377    out.extend_from_slice(validity);
11378    match coded {
11379        Some(coded) => out.extend_from_slice(&coded),
11380        None => {
11381            for &code in codes {
11382                put_u32(&mut out, code);
11383            }
11384        }
11385    }
11386    Ok(out)
11387}
11388
11389/// One part of one column as a page, for every column that is not coded against a global
11390/// dictionary. Those are built by [`coded_page`] from codes [`prepare`] handed out.
11391fn encode(vector: &Vector, settling: &mut Settling) -> Result<Vec<u8>> {
11392    let ty = vector.logical_type();
11393    // flatten: the file writer needs a uniform scalar page and does it once per loaded chunk.
11394    let flat = vector.flatten()?;
11395    let mut out = Vec::new();
11396    let dictionary = if coded_type(ty) { string_dictionary(&flat)? } else { None };
11397    let compressed_text = if dictionary.is_none() && coded_type(ty) {
11398        text_compressed(&flat, settling)?
11399    } else {
11400        None
11401    };
11402    let packed_vector = if dictionary.is_none() { Some(flat.bit_packed()?) } else { None };
11403    let packed = packed_vector.as_ref().and_then(Vector::packed_parts);
11404    // Only where nothing else has claimed the page, which is the plain integer case. A packed part
11405    // is still on the table because the cascade has to beat it too: the bit pack takes a part only
11406    // when it halves it, so a column that shrinks by a third was coming out whole.
11407    let cascade =
11408        if dictionary.is_none() { cascaded(&flat, ty, packed.as_ref(), settling)? } else { None };
11409    out.push(if cascade.is_some() {
11410        5
11411    } else if dictionary.is_some() {
11412        1
11413    } else if compressed_text.is_some() {
11414        6
11415    } else if packed.is_some() {
11416        2
11417    } else {
11418        0
11419    });
11420    push_validity(&mut out, &flat);
11421    if let Some(cascade) = cascade {
11422        out.extend_from_slice(&cascade);
11423        return Ok(out);
11424    }
11425    if let Some(dictionary) = dictionary {
11426        out.extend_from_slice(&dictionary);
11427        return Ok(out);
11428    }
11429    if let Some(compressed_text) = compressed_text {
11430        out.extend_from_slice(&compressed_text);
11431        return Ok(out);
11432    }
11433    if let Some(packed) = packed {
11434        if packed.offset() != 0 {
11435            return Err(invalid("writer received a sliced packed vector"));
11436        }
11437        out.push(u8::try_from(packed.width()).map_err(|_| invalid("packed width overflow"))?);
11438        out.extend_from_slice(&packed.base().to_le_bytes());
11439        put_u32(
11440            &mut out,
11441            u32::try_from(packed.words().len()).map_err(|_| invalid("too many packed words"))?,
11442        );
11443        for word in packed.words() {
11444            put_u64(&mut out, *word);
11445        }
11446        return Ok(out);
11447    }
11448    let data = flat.data().ok_or_else(|| invalid("scalar column did not flatten"))?;
11449    match (ty, data) {
11450        (LogicalType::TinyInt, Data::Int8(values)) => {
11451            for value in &**values {
11452                out.extend_from_slice(&value.to_le_bytes());
11453            }
11454        }
11455        (LogicalType::UTinyInt, Data::UInt8(values)) => {
11456            for value in &**values {
11457                out.extend_from_slice(&value.to_le_bytes());
11458            }
11459        }
11460        (LogicalType::SmallInt, Data::Int16(values)) => {
11461            for value in &**values {
11462                out.extend_from_slice(&value.to_le_bytes());
11463            }
11464        }
11465        (LogicalType::USmallInt, Data::UInt16(values)) => {
11466            for value in &**values {
11467                out.extend_from_slice(&value.to_le_bytes());
11468            }
11469        }
11470        (LogicalType::UInteger, Data::UInt32(values)) => {
11471            for value in &**values {
11472                out.extend_from_slice(&value.to_le_bytes());
11473            }
11474        }
11475        (LogicalType::UBigInt, Data::UInt64(values)) => {
11476            for value in &**values {
11477                out.extend_from_slice(&value.to_le_bytes());
11478            }
11479        }
11480        (LogicalType::Integer | LogicalType::Date, Data::Int32(values)) => {
11481            for value in &**values {
11482                out.extend_from_slice(&value.to_le_bytes());
11483            }
11484        }
11485        (
11486            LogicalType::BigInt
11487            | LogicalType::Timestamp
11488            | LogicalType::Time
11489            | LogicalType::TimeTz
11490            | LogicalType::TimestampTz
11491            | LogicalType::TimestampS
11492            | LogicalType::TimestampMs
11493            | LogicalType::TimestampNs,
11494            Data::Int64(values),
11495        ) => {
11496            for value in &**values {
11497                out.extend_from_slice(&value.to_le_bytes());
11498            }
11499        }
11500        // A hugeint and a uuid are both the 128 bit lane, and a uuid's bits are the ones the rest of
11501        // the engine already carries it in, so nothing about the value changes on the way down.
11502        (LogicalType::HugeInt | LogicalType::Uuid, Data::Int128(values)) => {
11503            for value in &**values {
11504                out.extend_from_slice(&value.to_le_bytes());
11505            }
11506        }
11507        (LogicalType::UHugeInt, Data::UInt128(values)) => {
11508            for value in &**values {
11509                out.extend_from_slice(&value.to_le_bytes());
11510            }
11511        }
11512        // Plainly, in the IEEE bytes. The integer encodings do not apply to a float and none of the
11513        // float codecs is worth having before somebody has measured a corpus of them.
11514        (LogicalType::Float, Data::Float32(values)) => {
11515            for value in &**values {
11516                out.extend_from_slice(&value.to_le_bytes());
11517            }
11518        }
11519        (LogicalType::Double, Data::Float64(values)) => {
11520            for value in &**values {
11521                out.extend_from_slice(&value.to_le_bytes());
11522            }
11523        }
11524        // Three counts and not one number. Months, days and microseconds stay apart on disk because
11525        // they are apart in the value: a month is not a fixed number of days and a day is not a
11526        // fixed number of microseconds, which is the whole reason the type has three fields.
11527        (LogicalType::Interval, Data::Interval(values)) => {
11528            for (months, days, micros) in &**values {
11529                out.extend_from_slice(&months.to_le_bytes());
11530                out.extend_from_slice(&days.to_le_bytes());
11531                out.extend_from_slice(&micros.to_le_bytes());
11532            }
11533        }
11534        (LogicalType::Boolean, Data::Bool(values)) => {
11535            for value in &**values {
11536                out.push(u8::from(*value));
11537            }
11538        }
11539        // The unscaled integer and nothing else. Scale is a property of the column and it is in the
11540        // directory already, so writing it a value at a time would be paying for it twice.
11541        (LogicalType::Decimal { .. }, Data::Int16(values)) => {
11542            for value in &**values {
11543                out.extend_from_slice(&value.to_le_bytes());
11544            }
11545        }
11546        (LogicalType::Decimal { .. }, Data::Int32(values)) => {
11547            for value in &**values {
11548                out.extend_from_slice(&value.to_le_bytes());
11549            }
11550        }
11551        (LogicalType::Decimal { .. }, Data::Int64(values)) => {
11552            for value in &**values {
11553                out.extend_from_slice(&value.to_le_bytes());
11554            }
11555        }
11556        (LogicalType::Decimal { .. }, Data::Int128(values)) => {
11557            for value in &**values {
11558                out.extend_from_slice(&value.to_le_bytes());
11559            }
11560        }
11561        // A blob and a bit string go down the way a varchar does, because the layout is the same
11562        // one: an offset a value and then the bytes. What is not the same is that nothing here may
11563        // read the payload as text, which is why this arm asks the column for bytes rather than for
11564        // a string, and why the codecs above that do read text are all asked of a varchar by name.
11565        (LogicalType::Varchar | LogicalType::Blob | LogicalType::Bit, Data::Varlen(values)) => {
11566            let mut bytes = Vec::new();
11567            put_u32(&mut out, 0);
11568            for row in 0..vector.len() {
11569                let value = values.bytes(row).ok_or_else(|| invalid("string view is invalid"))?;
11570                bytes.extend_from_slice(value);
11571                put_u32(
11572                    &mut out,
11573                    u32::try_from(bytes.len())
11574                        .map_err(|_| invalid("string payload exceeds 4GiB"))?,
11575                );
11576            }
11577            out.extend_from_slice(&bytes);
11578        }
11579        _ => return Err(Error::not_implemented(format!("native page for {ty}"))),
11580    }
11581    Ok(out)
11582}
11583
11584fn put_varint(out: &mut Vec<u8>, mut value: u32) {
11585    while value >= 0x80 {
11586        out.push((value as u8 & 0x7f) | 0x80);
11587        value >>= 7;
11588    }
11589    out.push(value as u8);
11590}
11591
11592/// The distinct codes of one part, which is what a stripe's membership index is merged from.
11593fn unique_codes(codes: &[u32]) -> Vec<u32> {
11594    let mut unique = codes.to_vec();
11595    unique.sort_unstable();
11596    unique.dedup();
11597    unique
11598}
11599
11600/// The union of the sorted distinct codes of every part in a stripe.
11601///
11602/// Pairwise up a tree rather than one long list concatenated and sorted. Both are the same order of
11603/// work on paper and the tree is the one that does not sort what is already in order: sixty four
11604/// sorted lists become one in six passes over the values.
11605fn merged_codes(lists: Vec<Vec<u32>>) -> Vec<u32> {
11606    let mut lists = lists;
11607    while lists.len() > 1 {
11608        let mut next = Vec::with_capacity(lists.len().div_ceil(2));
11609        for pair in lists.chunks(2) {
11610            match pair {
11611                [left, right] => next.push(merged_pair(left, right)),
11612                [only] => next.push(only.clone()),
11613                _ => {}
11614            }
11615        }
11616        lists = next;
11617    }
11618    lists.pop().unwrap_or_default()
11619}
11620
11621fn merged_pair(left: &[u32], right: &[u32]) -> Vec<u32> {
11622    let mut out = Vec::with_capacity(left.len().saturating_add(right.len()));
11623    let mut at = 0;
11624    let mut to = 0;
11625    while at < left.len() && to < right.len() {
11626        match left[at].cmp(&right[to]) {
11627            Ordering::Less => {
11628                out.push(left[at]);
11629                at += 1;
11630            }
11631            Ordering::Greater => {
11632                out.push(right[to]);
11633                to += 1;
11634            }
11635            Ordering::Equal => {
11636                out.push(left[at]);
11637                at += 1;
11638                to += 1;
11639            }
11640        }
11641    }
11642    out.extend_from_slice(&left[at..]);
11643    out.extend_from_slice(&right[to..]);
11644    out
11645}
11646
11647/// The widest bounds and the total null count of a stripe, from the bounds of its parts.
11648///
11649/// A bound that is missing from any part is missing from the stripe, because a missing bound means
11650/// nothing is known and a stripe that holds an unknown cannot claim one.
11651fn merged_range(ranges: impl Iterator<Item = Range>) -> Range {
11652    let mut merged = Range::default();
11653    let mut first = true;
11654    for range in ranges {
11655        merged.nulls = merged.nulls.saturating_add(range.nulls);
11656        // Both of these have to survive every part, so one part that could not say anything makes
11657        // the stripe unable to say it either. A sum is dropped on overflow rather than wrapped,
11658        // which leaves the stripe with exact ends and no total, which is a true thing to say.
11659        merged.sum = match (merged.sum.take(), range.sum) {
11660            (Some(held), Some(next)) if !first => held.checked_add(next),
11661            (_, next) if first => next,
11662            _ => None,
11663        };
11664        merged.exact = if first { range.exact } else { merged.exact && range.exact };
11665        if first {
11666            merged.low = range.low;
11667            merged.high = range.high;
11668            first = false;
11669            continue;
11670        }
11671        merged.low = match (merged.low.take(), range.low) {
11672            (Some(held), Some(next)) => Some(held.smaller(next)),
11673            _ => None,
11674        };
11675        merged.high = match (merged.high.take(), range.high) {
11676            (Some(held), Some(next)) => Some(held.larger(next)),
11677            _ => None,
11678        };
11679    }
11680    merged
11681}
11682
11683/// One stripe's sieves for one column: the part count, a length for each part, then their bytes.
11684///
11685/// One page for the whole stripe rather than one per part, because a part's sieve is a few hundred
11686/// bytes and sixty four of those are sixty four directory entries and sixty four reads for something
11687/// a scan walks straight through. A part with no sieve writes a length of zero and costs four bytes.
11688/// `bound` cut down to [`PART_BOUND_BYTES`], still a bound of the side it was.
11689///
11690/// A prefix of a string sorts at or before the string, so cutting one down leaves a low end that is
11691/// still a low end. A high end has to go the other way, so the cut prefix is stepped up at the last
11692/// byte that can carry it, and a prefix of nothing but `0xFF` has no such byte and gives up the
11693/// bound rather than claiming one that is too small. Anything that is not a string is already a
11694/// fixed width and is left alone.
11695fn shortened(bound: Option<Bound>, high: bool) -> Option<Bound> {
11696    match bound {
11697        Some(Bound::Bytes(mut value)) if value.len() > PART_BOUND_BYTES => {
11698            value.truncate(PART_BOUND_BYTES);
11699            if !high {
11700                return Some(Bound::Bytes(value));
11701            }
11702            while let Some(last) = value.pop() {
11703                if last < u8::MAX {
11704                    value.push(last + 1);
11705                    return Some(Bound::Bytes(value));
11706                }
11707            }
11708            None
11709        }
11710        other => other,
11711    }
11712}
11713
11714/// The ranges of one column's parts of one stripe, as a page.
11715///
11716/// The two ends and the null count, and not `exact` or the total. Those two answer a `MIN` or a
11717/// `SUM` out of the directory, and the directory already answers those per stripe, where the same
11718/// number costs sixty times less to keep. What a part range is for is skipping the part, and
11719/// skipping needs the ends. So a range read back from here says it is not exact, which is true of a
11720/// string end that was cut down anyway.
11721fn encode_part_ranges(ranges: &[Range]) -> Result<Vec<u8>> {
11722    let mut out = Vec::new();
11723    put_u32(
11724        &mut out,
11725        u32::try_from(ranges.len()).map_err(|_| invalid("too many parts in a stripe"))?,
11726    );
11727    for range in ranges {
11728        put_bound(&mut out, shortened(range.low.clone(), false).as_ref())?;
11729        put_bound(&mut out, shortened(range.high.clone(), true).as_ref())?;
11730        put_u32(&mut out, u32::try_from(range.nulls).map_err(|_| invalid("null count overflow"))?);
11731    }
11732    Ok(out)
11733}
11734
11735/// The ranges one encoded page holds, one entry per part of the stripe.
11736fn decode_part_ranges(bytes: &[u8]) -> Result<Vec<Range>> {
11737    let mut cur = Cursor::new(bytes);
11738    let parts = cur.u32()? as usize;
11739    let mut out = Vec::new();
11740    for _ in 0..parts {
11741        let low = cur.bound()?;
11742        let high = cur.bound()?;
11743        let nulls = cur.u32()? as usize;
11744        out.push(Range { low, high, nulls, exact: false, sum: None });
11745    }
11746    Ok(out)
11747}
11748
11749fn encode_sieves<'a>(sieves: impl Iterator<Item = &'a Option<Sieve>>) -> Result<Vec<u8>> {
11750    let held: Vec<&Option<Sieve>> = sieves.collect();
11751    let mut out = Vec::new();
11752    put_u32(
11753        &mut out,
11754        u32::try_from(held.len()).map_err(|_| invalid("too many parts in a stripe"))?,
11755    );
11756    for sieve in &held {
11757        let length = sieve.as_ref().map_or(0, Sieve::len);
11758        put_u32(&mut out, u32::try_from(length).map_err(|_| invalid("sieve length overflow"))?);
11759    }
11760    // flatten: a part with no sieve wrote a length of zero above and contributes no bytes here.
11761    for sieve in held.into_iter().flatten() {
11762        out.extend_from_slice(&sieve.to_bytes());
11763    }
11764    Ok(out)
11765}
11766
11767/// The sieves one encoded page holds, one entry per part of the stripe.
11768///
11769/// A part whose bytes are not a sieve this version understands comes back as `None`, which is a part
11770/// that gets read. That is how a file written by a later version of the sieve stays readable rather
11771/// than being a corrupt page.
11772fn decode_sieves(bytes: &[u8]) -> Result<Vec<Option<Sieve>>> {
11773    let parts = u32::from_le_bytes(
11774        bytes
11775            .get(..4)
11776            .ok_or_else(|| invalid("sieve page is truncated"))?
11777            .try_into()
11778            .map_err(|_| invalid("sieve page is truncated"))?,
11779    ) as usize;
11780    let mut lengths = Vec::with_capacity(parts);
11781    for part in 0..parts {
11782        let at = 4 + part * 4;
11783        let field = bytes.get(at..at + 4).ok_or_else(|| invalid("sieve page is truncated"))?;
11784        lengths.push(u32::from_le_bytes(
11785            field.try_into().map_err(|_| invalid("sieve page is truncated"))?,
11786        ) as usize);
11787    }
11788    let mut at = 4 + parts * 4;
11789    let mut out = Vec::with_capacity(parts);
11790    for length in lengths {
11791        if length == 0 {
11792            out.push(None);
11793            continue;
11794        }
11795        let end = at.checked_add(length).ok_or_else(|| invalid("sieve page is truncated"))?;
11796        let field = bytes.get(at..end).ok_or_else(|| invalid("sieve page is truncated"))?;
11797        out.push(Sieve::from_bytes(field));
11798        at = end;
11799    }
11800    if at != bytes.len() {
11801        return Err(invalid("sieve page has trailing bytes"));
11802    }
11803    Ok(out)
11804}
11805
11806/// One stripe's membership index: the code count and then the codes as ascending deltas.
11807///
11808/// The codes have to be sorted and distinct already, which is what [`unique_codes`] and
11809/// [`merged_codes`] hand over. Anything else decodes as different codes, so neither of those two is
11810/// a step a caller can skip.
11811fn encode_membership(unique: &[u32]) -> Vec<u8> {
11812    let mut out = Vec::with_capacity(unique.len().saturating_mul(2).saturating_add(5));
11813    put_varint(&mut out, u32::try_from(unique.len()).unwrap_or(u32::MAX));
11814    let mut previous = 0;
11815    for (at, &code) in unique.iter().enumerate() {
11816        put_varint(&mut out, if at == 0 { code } else { code - previous });
11817        previous = code;
11818    }
11819    out
11820}
11821
11822fn take_varint(bytes: &[u8], at: &mut usize) -> Result<u32> {
11823    let mut value = 0_u32;
11824    for shift in (0..35).step_by(7) {
11825        let byte = *bytes.get(*at).ok_or_else(|| invalid("membership varint is truncated"))?;
11826        *at += 1;
11827        let part = u32::from(byte & 0x7f);
11828        if shift == 28 && part > 0x0f {
11829            return Err(invalid("membership varint overflow"));
11830        }
11831        value = value
11832            .checked_add(
11833                part.checked_shl(shift).ok_or_else(|| invalid("membership varint overflow"))?,
11834            )
11835            .ok_or_else(|| invalid("membership varint overflow"))?;
11836        if byte & 0x80 == 0 {
11837            return Ok(value);
11838        }
11839    }
11840    Err(invalid("membership varint is too long"))
11841}
11842
11843fn decode_membership(bytes: &[u8]) -> Result<Vec<u32>> {
11844    let mut at = 0;
11845    let count = take_varint(bytes, &mut at)? as usize;
11846    let mut codes = Vec::with_capacity(count);
11847    let mut previous = 0_u32;
11848    for index in 0..count {
11849        let delta = take_varint(bytes, &mut at)?;
11850        let code = if index == 0 {
11851            delta
11852        } else {
11853            previous.checked_add(delta).ok_or_else(|| invalid("membership code overflow"))?
11854        };
11855        if index > 0 && code <= previous {
11856            return Err(invalid("membership codes are not increasing"));
11857        }
11858        codes.push(code);
11859        previous = code;
11860    }
11861    if at != bytes.len() {
11862        return Err(invalid("membership page has trailing bytes"));
11863    }
11864    Ok(codes)
11865}
11866
11867/// A varchar page as a dictionary of its distinct values and a code a row, or `None` when that does
11868/// not come out smaller than the raw form.
11869///
11870/// Every text page that no global dictionary claims asks this first, including the page of
11871/// comments that never has a repeat, so the map is hashed with [`Spread`] rather than SipHash and
11872/// sized for the page up front. With the default hasher and growth it was 4% of the instructions of
11873/// a `lineitem` load from CSV, all of it on `l_comment` pages this then refused.
11874fn string_dictionary(vector: &Vector) -> Result<Option<Vec<u8>>> {
11875    let mut by_text: HashMap<&[u8], u32, Spread> =
11876        HashMap::with_capacity_and_hasher(vector.len(), Spread);
11877    let mut values = Vec::new();
11878    let mut codes = Vec::with_capacity(vector.len());
11879    let mut plain_bytes = 0_usize;
11880    for row in 0..vector.len() {
11881        let text = vector.bytes_at(row).unwrap_or(b"");
11882        plain_bytes = plain_bytes.saturating_add(text.len());
11883        let code = match by_text.get(text) {
11884            Some(&code) => code,
11885            None => {
11886                let code = u32::try_from(values.len())
11887                    .map_err(|_| invalid("too many dictionary values"))?;
11888                by_text.insert(text, code);
11889                values.push(text);
11890                code
11891            }
11892        };
11893        codes.push(code);
11894    }
11895    let dictionary_bytes = values.iter().map(|value| value.len()).sum::<usize>();
11896    let encoded = 8_usize
11897        .saturating_add((values.len() + 1).saturating_mul(4))
11898        .saturating_add(dictionary_bytes)
11899        .saturating_add(codes.len().saturating_mul(4));
11900    let plain = (vector.len() + 1).saturating_mul(4).saturating_add(plain_bytes);
11901    if encoded >= plain {
11902        return Ok(None);
11903    }
11904    let mut out = Vec::with_capacity(encoded);
11905    put_u32(
11906        &mut out,
11907        u32::try_from(values.len()).map_err(|_| invalid("too many dictionary values"))?,
11908    );
11909    put_u32(
11910        &mut out,
11911        u32::try_from(dictionary_bytes).map_err(|_| invalid("dictionary payload exceeds 4GiB"))?,
11912    );
11913    let mut offset = 0_u32;
11914    put_u32(&mut out, offset);
11915    for value in &values {
11916        offset = offset
11917            .checked_add(
11918                u32::try_from(value.len()).map_err(|_| invalid("dictionary value is too long"))?,
11919            )
11920            .ok_or_else(|| invalid("dictionary payload exceeds 4GiB"))?;
11921        put_u32(&mut out, offset);
11922    }
11923    for value in values {
11924        out.extend_from_slice(value);
11925    }
11926    for code in codes {
11927        put_u32(&mut out, code);
11928    }
11929    Ok(Some(out))
11930}
11931
11932/// The room one closing column takes under [`CLOSE_BYTES`], given back when dropped.
11933struct Room<'a, T> {
11934    state: &'a Mutex<(T, usize)>,
11935    finished: &'a Condvar,
11936    bytes: usize,
11937}
11938
11939impl<T> Drop for Room<'_, T> {
11940    fn drop(&mut self) {
11941        let mut held = self.state.lock().unwrap_or_else(PoisonError::into_inner);
11942        held.1 -= self.bytes;
11943        drop(held);
11944        self.finished.notify_all();
11945    }
11946}
11947
11948/// One column's work at the end of a load, as [`Writer::close_columns`] schedules it.
11949enum Closing<'a> {
11950    /// A numeric column's frequencies, whether to count its distinct values exactly, and the
11951    /// range to count them in a flat array when it is short enough.
11952    Numeric {
11953        column: usize,
11954        counted: bool,
11955        dense: Option<(u64, usize)>,
11956    },
11957    Dictionary {
11958        index: usize,
11959        dictionary: &'a GlobalDictionary,
11960    },
11961}
11962
11963/// What one [`Closing`] came back with, by column.
11964enum Closed {
11965    Numeric(usize, (Option<FrequencySummary>, Option<u64>)),
11966    Dictionary(usize, ClosedDictionary),
11967}
11968
11969/// What [`Writer::close_dictionary`] builds for one column and [`Writer::close`] writes.
11970struct ClosedDictionary {
11971    /// `None` for a demoted dictionary, which holds only some of the column. See [`DEMOTED`].
11972    distinct: Option<u64>,
11973    frequencies: Option<FrequencySummary>,
11974    texts: Vec<Option<Vec<u8>>>,
11975    hosts: Option<host::HostSummary>,
11976    encoded: EncodedDictionary,
11977    /// The bytes of the column's payload blocks, which are already in the file.
11978    payload: u64,
11979}
11980
11981struct EncodedDictionary {
11982    index: Vec<u8>,
11983    ranks: Vec<u8>,
11984    grams: Vec<u8>,
11985}
11986
11987/// Sorts codes into the byte order of the values they name, eight bytes of depth at a time.
11988///
11989/// # What the shape of the data does to a comparison sort
11990///
11991/// Distinct values against distinct prefixes, on the eight million row `hits`:
11992///
11993/// ```text
11994///   distinct   first 8   first 16   first 32   column
11995///  2,266,417        50      8,892    232,630   URL
11996///  2,346,025        49      8,534    204,060   Referer
11997///  1,357,764    81,362    348,340    861,579   Title
11998/// ```
11999///
12000/// Two and a quarter million URLs have fifty distinct first eight bytes between them, because they
12001/// all begin `http://` and then a host and there are not many hosts. So a sort that leads with
12002/// those eight bytes settles almost nothing on `URL` and `Referer`, whatever the comment on it used
12003/// to say, and almost every pair falls through to a comparison of whole values that agree for most
12004/// of their length. `Title` is free text and separates at eight bytes, which is why the design
12005/// looked right when it was written.
12006///
12007/// # What is done about it
12008///
12009/// Sort on eight bytes of the value at the current depth, held beside the code, and then take each
12010/// run that those eight bytes leave tied and sort it again on the next eight. A value is fetched
12011/// from the payload once per eight bytes of depth rather than once per comparison, and the sort
12012/// itself runs over an array of integers that is in cache rather than over pointers into a payload
12013/// that is hundreds of megabytes.
12014///
12015/// That is the whole trick, and it matters because the payload touch is the expensive part. The
12016/// bytes themselves are nearly free once the line is in cache, so reading eight at a time and
12017/// throwing away the ones that were not needed beats going back for each one.
12018///
12019/// # Why the length has to be carried
12020///
12021/// The eight bytes are padded with zero when the value has fewer than eight left, and a zero byte
12022/// can appear in a value, so equal keys do not mean equal bytes. What is true is that a value which
12023/// ran out inside the window is a prefix of any other value with the same key, and a prefix sorts
12024/// first, so how many of the eight bytes were real is the tie break and nothing further is needed.
12025/// A run is only worth another pass when all eight were real, because otherwise the run is one
12026/// value: a dictionary holds a value once.
12027fn sort_by_value<'a>(codes: &mut [u32], values: impl Fn(u32) -> &'a [u8]) {
12028    let mut work = vec![(0, codes.len(), 0)];
12029    let mut keyed: Vec<(u64, u8, u32)> = Vec::new();
12030    while let Some((from, to, depth)) = work.pop() {
12031        let part = &mut codes[from..to];
12032        keyed.clear();
12033        keyed.extend(part.iter().map(|&code| {
12034            let value = values(code);
12035            let rest = value.get(depth..).unwrap_or_default();
12036            (head(rest), rest.len().min(8) as u8, code)
12037        }));
12038        keyed.sort_unstable();
12039        for (slot, entry) in part.iter_mut().zip(keyed.iter()) {
12040            *slot = entry.2;
12041        }
12042        let mut start = 0;
12043        while start < keyed.len() {
12044            let (key, taken, _) = keyed[start];
12045            let mut end = start + 1;
12046            while end < keyed.len() && keyed[end].0 == key && keyed[end].1 == taken {
12047                end += 1;
12048            }
12049            if taken == 8 && end - start > 1 {
12050                work.push((from + start, from + end, depth + 8));
12051            }
12052            start = end;
12053        }
12054    }
12055}
12056
12057/// How few codes are worth sorting on more than one thread.
12058const PARALLEL_SORT_MIN: usize = 1 << 16;
12059
12060/// How many buckets a thread gets in [`sort_by_value_across`], so that a thread that drew a slow
12061/// bucket is not what the others wait for.
12062const BUCKETS_PER_WORKER: usize = 4;
12063
12064/// How many sampled codes stand for each bucket when the splitters are picked.
12065const SAMPLES_PER_BUCKET: usize = 32;
12066
12067/// [`sort_by_value`] over `workers` threads, with the same answer.
12068///
12069/// A sample sort. A sample of the codes is sorted and cut into as many equal runs as there are
12070/// buckets, and the values at the cuts are the splitters. Every code goes to the bucket its value
12071/// falls in by a binary search of the splitters, the buckets are laid end to end in splitter order,
12072/// and each bucket is then sorted on its own by whichever thread takes it. Every value in a bucket
12073/// sorts after every value in the bucket before, so the buckets sorted one by one are the codes
12074/// sorted.
12075///
12076/// The answer is the one [`sort_by_value`] gives down to the order of equal values, not only the
12077/// order of different ones. A global dictionary holds each value once, so there are none, but the
12078/// sort does not rely on it: equal values land in the same bucket in code order, which is the order
12079/// [`sort_by_value`] leaves them in, since the code is the last thing it sorts on.
12080///
12081/// On the 10m ClickBench sample the close sorts five columns of one to three and a half million
12082/// distinct values, one column at a time, and until this each sort ran on one thread while the
12083/// other thirty one waited for it.
12084fn sort_by_value_across<'a>(
12085    codes: &mut [u32],
12086    values: impl Fn(u32) -> &'a [u8] + Sync,
12087    workers: usize,
12088) {
12089    if workers <= 1 || codes.len() < PARALLEL_SORT_MIN {
12090        sort_by_value(codes, values);
12091        return;
12092    }
12093    let buckets = workers * BUCKETS_PER_WORKER;
12094    let wanted = buckets * SAMPLES_PER_BUCKET;
12095    let mut sample = (0..wanted).map(|at| codes[at * codes.len() / wanted]).collect::<Vec<_>>();
12096    sort_by_value(&mut sample, &values);
12097    let splitters =
12098        (1..buckets).map(|cut| values(sample[cut * sample.len() / buckets])).collect::<Vec<_>>();
12099    let values = &values;
12100    let splitters = &splitters;
12101    let per = codes.len().div_ceil(workers);
12102    // Which bucket each code goes to, a run of the codes per thread.
12103    let places = std::thread::scope(|scope| {
12104        codes
12105            .chunks(per)
12106            .map(|run| {
12107                scope.spawn(move || {
12108                    run.iter()
12109                        .map(|&code| {
12110                            let value = values(code);
12111                            splitters.partition_point(|splitter| *splitter <= value) as u32
12112                        })
12113                        .collect::<Vec<_>>()
12114                })
12115            })
12116            .collect::<Vec<_>>()
12117            .into_iter()
12118            .flat_map(|handle| {
12119                handle.join().unwrap_or_else(|panic| std::panic::resume_unwind(panic))
12120            })
12121            .collect::<Vec<_>>()
12122    });
12123    let mut starts = vec![0_usize; buckets + 1];
12124    for &place in &places {
12125        starts[place as usize + 1] += 1;
12126    }
12127    for bucket in 0..buckets {
12128        starts[bucket + 1] += starts[bucket];
12129    }
12130    let mut laid = vec![0_u32; codes.len()];
12131    let mut next = starts.clone();
12132    for (&code, &place) in codes.iter().zip(&places) {
12133        laid[next[place as usize]] = code;
12134        next[place as usize] += 1;
12135    }
12136    drop(places);
12137    let mut runs = Vec::with_capacity(buckets);
12138    let mut rest = laid.as_mut_slice();
12139    for bucket in 0..buckets {
12140        let (run, after) = rest.split_at_mut(starts[bucket + 1] - starts[bucket]);
12141        runs.push(run);
12142        rest = after;
12143    }
12144    // The largest buckets first, since they are taken from the back.
12145    runs.sort_by_key(|run| run.len());
12146    let queue = Mutex::new(runs);
12147    std::thread::scope(|scope| {
12148        for _ in 0..workers {
12149            scope.spawn(|| {
12150                loop {
12151                    let taken = queue.lock().unwrap_or_else(PoisonError::into_inner).pop();
12152                    let Some(run) = taken else { break };
12153                    sort_by_value(run, values);
12154                }
12155            });
12156        }
12157    });
12158    codes.copy_from_slice(&laid);
12159}
12160
12161/// The first eight bytes of a value as an integer that sorts the way the bytes sort.
12162///
12163/// A value shorter than eight bytes is padded with zeros after it. The short case is two loads of
12164/// four that overlap rather than a copy of however many bytes there are, because a copy of a length
12165/// the compiler cannot see is a call to `memcpy`, and this runs once a value at every level of the
12166/// sort in [`sort_by_value`]. On a load of a million rows of `hits` that call was 2.9 percent of
12167/// the load's cycles, and the loads that replace it put 1.5 percent on the sort itself.
12168fn head(bytes: &[u8]) -> u64 {
12169    if let Some(word) = bytes.first_chunk::<8>() {
12170        return u64::from_be_bytes(*word);
12171    }
12172    let len = bytes.len();
12173    if len >= 4 {
12174        let front = u64::from(u32::from_be_bytes([bytes[0], bytes[1], bytes[2], bytes[3]]));
12175        let back = &bytes[len - 4..];
12176        let back = u64::from(u32::from_be_bytes([back[0], back[1], back[2], back[3]]));
12177        return (front << 32) | (back << (8 * (8 - len)));
12178    }
12179    bytes.iter().enumerate().fold(0, |word, (at, &byte)| word | (u64::from(byte) << (56 - 8 * at)))
12180}
12181
12182/// One column's dictionary page, which is its index and its sorted order.
12183///
12184/// The payload is not in it. Its blocks are in the file already, written as each was encoded, and
12185/// `places` says where, in block order. With `scattered` set the index records each block's start
12186/// and length, so a reader can find one wherever it went.
12187///
12188/// `scattered` false lays the blocks out the way a file written before format 26 has them, one
12189/// behind the next with only the ends recorded. Nothing in the writer asks for that any more. It is
12190/// kept because [`open_global_dictionary`] still reads those files and a reading path that nothing
12191/// can produce is a reading path nothing tests.
12192fn encode_global_dictionary(
12193    dictionary: &GlobalDictionary,
12194    order: &[(u64, u32)],
12195    places: &[Placed],
12196    scattered: bool,
12197) -> Result<EncodedDictionary> {
12198    let values = dictionary.values();
12199    if order.len() != values {
12200        return Err(invalid("global dictionary order does not cover its values"));
12201    }
12202    let blocks = values.div_ceil(TEXT_PAYLOAD_VALUES);
12203    if places.len() != blocks {
12204        return Err(invalid("global dictionary payload is not the blocks it says it is"));
12205    }
12206    if dictionary.grams.len() != blocks {
12207        return Err(invalid("global dictionary signatures do not cover its blocks"));
12208    }
12209    let (ranks, rank_ends) = encode_ranks(order, code_width(values))?;
12210    let rank_blocks = values.div_ceil(TEXT_RANK_BLOCK);
12211    let offset_bits = offset_width(&dictionary.ends);
12212    let payload_words = if scattered { 3 } else { 2 };
12213    let index_len = DICTIONARY_HEADER
12214        .checked_add(offset_bytes(values, offset_bits))
12215        .and_then(|len| len.checked_add(blocks.checked_mul(payload_words * 8)?))
12216        .and_then(|len| len.checked_add(rank_blocks.checked_mul(16)?))
12217        .and_then(|len| len.checked_add(8))
12218        .ok_or_else(|| invalid("global dictionary index length overflow"))?;
12219    let mut index = Vec::with_capacity(index_len);
12220    put_u32(
12221        &mut index,
12222        u32::try_from(values).map_err(|_| invalid("global dictionary has too many values"))?,
12223    );
12224    put_u32(&mut index, TEXT_PAYLOAD_VALUES as u32);
12225    put_u32(
12226        &mut index,
12227        u32::try_from(blocks).map_err(|_| invalid("global dictionary has too many blocks"))?,
12228    );
12229    let flag = (if scattered { DICTIONARY_SCATTERED } else { 0 })
12230        | DICTIONARY_GRAMS
12231        | DICTIONARY_WIDE_GRAMS;
12232    put_u32(&mut index, offset_bits as u32 | flag);
12233    encode_offsets(&dictionary.ends, offset_bits, &mut index)?;
12234    // Where each block is and how long it is, so a reader can find one. The stored blocks are
12235    // shorter than the decoded ones and by a different amount each, so their lengths are the one
12236    // thing the offsets above no longer say, and where they start is no longer arithmetic on the
12237    // block before once a block is written the moment it is encoded.
12238    let mut end = 0_u64;
12239    for place in places {
12240        if scattered {
12241            put_u64(&mut index, place.start);
12242            put_u64(&mut index, place.length);
12243        } else {
12244            end = end
12245                .checked_add(place.length)
12246                .ok_or_else(|| invalid("global dictionary payload overflow"))?;
12247            put_u64(&mut index, end);
12248        }
12249    }
12250    for place in places {
12251        put_u64(&mut index, place.hash);
12252    }
12253    // The same two lists for the sorted order. A rank block is packed at whatever width its own
12254    // heads need, so where one ends is no longer arithmetic on the block number.
12255    if rank_ends.len() != rank_blocks {
12256        return Err(invalid("global dictionary order is not the blocks it says it is"));
12257    }
12258    for end in &rank_ends {
12259        put_u64(&mut index, *end);
12260    }
12261    let mut at = 0_usize;
12262    for end in &rank_ends {
12263        let end = usize::try_from(*end).map_err(|_| invalid("global dictionary order overflow"))?;
12264        put_u64(&mut index, checksum(&ranks[at..end]));
12265        at = end;
12266    }
12267    let gram_len = blocks
12268        .checked_mul(TEXT_GRAM_BYTES)
12269        .ok_or_else(|| invalid("global dictionary signature count overflow"))?;
12270    let mut grams = Vec::with_capacity(gram_len);
12271    for block in &dictionary.grams {
12272        grams.extend_from_slice(block);
12273    }
12274    put_u64(&mut index, checksum(&grams));
12275    if index.len() != index_len {
12276        return Err(invalid("global dictionary index is not the length it was laid out for"));
12277    }
12278    Ok(EncodedDictionary { index, ranks, grams })
12279}
12280
12281/// How many blocks of the payload the shape is settled on.
12282///
12283/// Eight blocks is 8,192 values, which is the sample `chooser::Sampled` draws and is that size for
12284/// the same reason. They are spread across the dictionary rather than taken off the front, because
12285/// a dictionary is in the order values were first seen and the front of it is the first morsel of
12286/// the load.
12287const PAYLOAD_SAMPLE_BLOCKS: usize = 8;
12288
12289/// The shapes the payload encoder picks between.
12290///
12291/// Narrow on purpose. The exhaustive search encodes every candidate at every level and runs at two
12292/// to six megabytes a second on this data, which over the twelve gigabytes of dictionary `hits`
12293/// carries is about an hour of processor time, so it cannot be what a load does. Each of these
12294/// settles the outer level and the one below it, which is where almost all of that hour goes, and
12295/// leaves the levels under them to the exhaustive search where the chunks are small enough for it
12296/// to cost nothing.
12297///
12298/// Measured on the five ClickBench columns that have a dictionary worth the name, at 1,024 values a
12299/// block, against the exhaustive search over the same blocks:
12300///
12301/// | column | exhaustive | FRONT then LZ | LZ then FSST | LZ then PLAIN |
12302/// |---|---|---|---|---|
12303/// | 2 | 2.923 at 4.3 MB/s | 2.587 at 21.2 | 2.593 at 36.1 | 2.538 at 53.6 |
12304/// | 13 | 3.093 at 3.1 | 3.029 at 36.4 | 2.921 at 35.7 | 2.770 at 82.9 |
12305/// | 14 | 2.330 at 2.1 | 2.283 at 24.3 | 2.213 at 23.5 | 2.113 at 67.6 |
12306/// | 39 | 2.459 at 5.3 | 2.147 at 10.6 | 2.145 at 29.3 | 2.088 at 43.1 |
12307/// | 56 | 4.694 at 6.3 | 4.381 at 51.0 | 4.172 at 50.6 | 3.983 at 86.8 |
12308///
12309/// The best of the three per column is 98 percent of the exhaustive ratio for a tenth of the time.
12310/// `FSST` and `PLAIN` on their own are in the list as a floor rather than to win. `FSST` is the
12311/// right answer for text that does not share prefixes with its neighbours, and `PLAIN` is there so
12312/// that a column nothing compresses is found out in the sample and written at a gigabyte a second
12313/// rather than searched for an answer that does not exist.
12314fn payload_shapes() -> Vec<chooser::Settled> {
12315    let integers = vec![integer::Kind::Packed];
12316    [
12317        vec![string::Kind::Front, string::Kind::Lz],
12318        vec![string::Kind::Lz, string::Kind::Fsst],
12319        vec![string::Kind::Lz, string::Kind::Plain],
12320        vec![string::Kind::Fsst],
12321        vec![string::Kind::Plain],
12322    ]
12323    .into_iter()
12324    .map(|strings| chooser::Settled::new(strings, integers.clone()))
12325    .collect()
12326}
12327
12328/// Syncs the file, and counts the sync and how long it took as a publish wait when a load is being
12329/// profiled.
12330///
12331/// A wait rather than time, because the time is already in the publish span around it. What the
12332/// wait columns add is how much of publish was the device, which on the WSL2 disk of the gaming PC
12333/// is most of it: a sync there costs about two milliseconds (see `rudb_device_card`).
12334fn synced(file: &dyn rudb_io::File, profile: Option<&LoadProfile>) -> Result<()> {
12335    let started = profile.map(|_| std::time::Instant::now());
12336    file.sync()?;
12337    if let (Some(profile), Some(started)) = (profile, started) {
12338        profile.waited(
12339            Stage::Publish,
12340            u64::try_from(started.elapsed().as_nanos()).unwrap_or(u64::MAX),
12341        );
12342    }
12343    Ok(())
12344}
12345
12346/// One sealed dictionary block on its way to being encoded outside the writer's lock.
12347///
12348/// Handed out by the merge that sealed it and encoded with the pages of the same stripe. See
12349/// [`GlobalDictionary::hand_out`].
12350#[derive(Debug)]
12351pub(crate) struct Unencoded {
12352    column: usize,
12353    at: usize,
12354    ends: Vec<u32>,
12355    bytes: Vec<u8>,
12356    shape: chooser::Settled,
12357}
12358
12359impl Unencoded {
12360    /// The encoded block and its signature.
12361    pub(crate) fn encode(&self) -> Result<EncodedBlock> {
12362        let values = block_values(&self.ends, &self.bytes);
12363        Ok((string::encode_with(&values, &self.shape)?, block_grams(&values)))
12364    }
12365
12366    /// The column and the block number the encoded block goes back to.
12367    pub(crate) fn place(&self) -> (usize, usize) {
12368        (self.column, self.at)
12369    }
12370}
12371
12372/// One encoded dictionary block and the signature of the values in it.
12373///
12374/// Boxed because it is carried around in things that are otherwise small.
12375pub(crate) type EncodedBlock = (Vec<u8>, Box<[u8; TEXT_GRAM_BYTES]>);
12376
12377/// The conservative four-byte substring signature of one block's values.
12378fn block_grams(values: &[&[u8]]) -> Box<[u8; TEXT_GRAM_BYTES]> {
12379    let mut grams = Box::new([0_u8; TEXT_GRAM_BYTES]);
12380    for value in values {
12381        for gram in value.windows(4) {
12382            for bit in gram_bits(gram, TEXT_GRAM_BYTES) {
12383                grams[bit / 8] |= 1 << (bit % 8);
12384            }
12385        }
12386    }
12387    grams
12388}
12389
12390/// The values of one block, given where each of them ends relative to the block.
12391fn block_values<'a>(ends: &[u32], bytes: &'a [u8]) -> Vec<&'a [u8]> {
12392    let mut out = Vec::with_capacity(ends.len());
12393    let mut from = 0;
12394    for &to in ends {
12395        out.push(&bytes[from..to as usize]);
12396        from = to as usize;
12397    }
12398    out
12399}
12400
12401/// Encodes every block still raw at the end of a load: the part block each column ends on and,
12402/// for a column too small to have settled a shape, every block it has.
12403///
12404/// Across threads, the way [`encode_ready`] does it. This ran one column at a time on the thread
12405/// closing the table, and a column that never settled a shape encodes each block by trying every
12406/// candidate, so on a million rows of `hits` it was most of the load's CPU on one core.
12407fn finish_dictionaries(dictionaries: &mut [Option<GlobalDictionary>]) -> Result<()> {
12408    for dictionary in dictionaries.iter_mut().flatten() {
12409        if !dictionary.early.is_empty() {
12410            return Err(Error::internal("a dictionary block handed out never came back"));
12411        }
12412        dictionary.seal_rest();
12413        dictionary.settle_rest()?;
12414    }
12415    encode_waiting(dictionaries)?;
12416    // A block handed out and never given back leaves a gap nothing above would notice when it was
12417    // the last one, so the count is checked against the values as well.
12418    if dictionaries
12419        .iter()
12420        .flatten()
12421        .any(|dictionary| dictionary.encoded() != dictionary.values().div_ceil(TEXT_PAYLOAD_VALUES))
12422    {
12423        return Err(Error::internal("a dictionary block handed out never came back"));
12424    }
12425    Ok(())
12426}
12427
12428/// Encodes the waiting blocks of every dictionary across threads, and appends them to their columns
12429/// in order.
12430fn encode_waiting(dictionaries: &mut [Option<GlobalDictionary>]) -> Result<()> {
12431    let jobs = dictionaries
12432        .iter()
12433        .enumerate()
12434        .flat_map(|(column, held)| {
12435            (0..held.as_ref().map_or(0, |held| held.waiting.len())).map(move |at| (column, at))
12436        })
12437        .collect::<Vec<_>>();
12438    if jobs.is_empty() {
12439        return Ok(());
12440    }
12441    let one = |column: usize, at: usize| -> Result<(usize, usize, EncodedBlock)> {
12442        let held = dictionaries[column].as_ref().ok_or_else(|| Error::internal("no dictionary"))?;
12443        Ok((column, at, held.encode_waiting(at)?))
12444    };
12445    let workers = std::thread::available_parallelism()
12446        .map_or(1, usize::from)
12447        .min(MAX_FREQUENCY_WORKERS)
12448        .min(jobs.len());
12449    let made = if workers <= 1 {
12450        jobs.iter().map(|&(column, at)| one(column, at)).collect::<Result<Vec<_>>>()?
12451    } else {
12452        let next = AtomicUsize::new(0);
12453        let jobs = &jobs;
12454        let pieces = std::thread::scope(|scope| {
12455            (0..workers)
12456                .map(|_| {
12457                    scope.spawn(|| {
12458                        let mut mine = Vec::new();
12459                        loop {
12460                            let job = next.fetch_add(1, Atomic::Relaxed);
12461                            let Some(&(column, at)) = jobs.get(job) else { break };
12462                            mine.push(one(column, at)?);
12463                        }
12464                        Ok(mine)
12465                    })
12466                })
12467                .collect::<Vec<_>>()
12468                .into_iter()
12469                .map(|handle| {
12470                    handle
12471                        .join()
12472                        .map_err(|_| Error::internal("a dictionary encode worker panicked"))?
12473                })
12474                .collect::<Result<Vec<_>>>()
12475        })?;
12476        pieces.into_iter().flatten().collect()
12477    };
12478    let mut done: Vec<Vec<(usize, EncodedBlock)>> =
12479        (0..dictionaries.len()).map(|_| Vec::new()).collect();
12480    for (column, at, bytes) in made {
12481        done[column].push((at, bytes));
12482    }
12483    for (column, mut made) in done.into_iter().enumerate() {
12484        if made.is_empty() {
12485            continue;
12486        }
12487        let Some(held) = dictionaries[column].as_mut() else { continue };
12488        made.sort_by_key(|(at, _)| *at);
12489        let waiting = std::mem::take(&mut held.waiting);
12490        for ((at, _), (_, block)) in waiting.into_iter().zip(made) {
12491            if held.encoded() != at {
12492                return Err(Error::internal("a dictionary block was encoded out of order"));
12493            }
12494            held.push_block(block);
12495        }
12496    }
12497    Ok(())
12498}
12499
12500/// Which of [`payload_shapes`] comes out smallest over a sample of the blocks.
12501///
12502/// Every shape is encoded over the same sample and the smallest wins, which is the exhaustive
12503/// search moved up a level: over shapes of a column rather than over candidates of a chunk. The
12504/// sample is spread across the dictionary so that the first and last blocks are both in it, because
12505/// a dictionary written in first seen order has its common values at the front and its long tail at
12506/// the back, and those do not compress alike. Which blocks those are is
12507/// [`GlobalDictionary::seal`]'s to decide, because by the time this is called the rest of them have
12508/// been encoded and the raw bytes are gone.
12509fn settle_shape(sample: &[Vec<&[u8]>]) -> Result<chooser::Settled> {
12510    let mut best: Option<(chooser::Settled, usize)> = None;
12511    for shape in payload_shapes() {
12512        let mut size = 0;
12513        for block in sample {
12514            size += string::encode_with(block, &shape)?.len();
12515        }
12516        if best.as_ref().is_none_or(|(_, smallest)| size < *smallest) {
12517            best = Some((shape, size));
12518        }
12519    }
12520    best.map(|(shape, _)| shape)
12521        .ok_or_else(|| invalid("no shape applies to a global dictionary payload"))
12522}
12523
12524/// The sorted order laid out the way a reader reads it, in blocks of [`TEXT_RANK_BLOCK`] entries.
12525///
12526/// Each block holds its heads first and then its codes, rather than pairing them, because a search
12527/// asks for a head at every probe and for a code about once a search. Keeping the heads together
12528/// means a probe touches eight bytes of a block rather than twelve spread over it, and the last few
12529/// probes of a search, which are the ones that land in the same block, touch the same cache line.
12530fn encode_ranks(order: &[(u64, u32)], code_bits: usize) -> Result<(Vec<u8>, Vec<u64>)> {
12531    let mut out = Vec::with_capacity(order.len() * 4);
12532    let mut ends = Vec::with_capacity(order.len().div_ceil(TEXT_RANK_BLOCK));
12533    let mut heads = Vec::with_capacity(TEXT_RANK_BLOCK);
12534    let mut codes = Vec::with_capacity(TEXT_RANK_BLOCK);
12535    for block in order.chunks(TEXT_RANK_BLOCK) {
12536        // The order is sorted by value and a head is a prefix of a value, so the heads of a block
12537        // rise, the smallest is the first and the largest is the last.
12538        let base = block.first().map_or(0, |&(head, _)| head);
12539        let span = block.last().map_or(0, |&(head, _)| head.wrapping_sub(base));
12540        let width = (u64::BITS - span.leading_zeros()) as usize;
12541        heads.clear();
12542        codes.clear();
12543        for &(head, code) in block {
12544            heads.push(head.wrapping_sub(base));
12545            codes.push(u64::from(code));
12546        }
12547        put_u64(&mut out, base);
12548        out.push(width as u8);
12549        bitpack::pack_tail(&heads, width, &mut out)
12550            .map_err(|_| invalid("global dictionary heads do not pack"))?;
12551        bitpack::pack_tail(&codes, code_bits, &mut out)
12552            .map_err(|_| invalid("global dictionary codes do not pack"))?;
12553        ends.push(out.len() as u64);
12554    }
12555    Ok((out, ends))
12556}
12557
12558/// Opens a column's global dictionary, which reads its index and none of its payload.
12559///
12560/// `keep_budget` is how many decoded payload bytes this dictionary may hold on to, and every
12561/// caller bar the test of the ceiling passes [`TEXT_KEEP_BUDGET`]. It is a parameter rather than
12562/// the constant read where it is used because a test of a ceiling that cannot be moved has to build
12563/// a quarter of a gigabyte of dictionary to reach it.
12564fn open_global_dictionary(
12565    file: Arc<File>,
12566    page: Page,
12567    ty: &LogicalType,
12568    keep_budget: usize,
12569) -> Result<Vector> {
12570    if !coded_type(ty) {
12571        return Err(invalid("global dictionary belongs to a non-string column"));
12572    }
12573    let mut header = [0; DICTIONARY_HEADER];
12574    read_at(&file, page.offset, &mut header)?;
12575    let count = u32::from_le_bytes(header[0..4].try_into().expect("four bytes")) as usize;
12576    let per_block = u32::from_le_bytes(header[4..8].try_into().expect("four bytes")) as usize;
12577    let blocks = u32::from_le_bytes(header[8..12].try_into().expect("four bytes")) as usize;
12578    let width = u32::from_le_bytes(header[12..16].try_into().expect("four bytes"));
12579    let scattered = width & DICTIONARY_SCATTERED != 0;
12580    let has_grams = width & DICTIONARY_GRAMS != 0;
12581    let gram_width =
12582        if width & DICTIONARY_WIDE_GRAMS != 0 { TEXT_GRAM_BYTES } else { NARROW_GRAM_BYTES };
12583    let offset_bits = (width & !DICTIONARY_FLAGS) as usize;
12584    if per_block != TEXT_PAYLOAD_VALUES {
12585        return Err(invalid("global dictionary block width differs"));
12586    }
12587    if blocks != count.div_ceil(TEXT_PAYLOAD_VALUES) {
12588        return Err(invalid("global dictionary block count differs from its value count"));
12589    }
12590    if offset_bits > u32::BITS as usize {
12591        return Err(invalid("global dictionary packs offsets past a payload"));
12592    }
12593    let offset_len = offset_bytes(count, offset_bits);
12594    // The sorted order is kept out of the index on purpose. The index is read and checksummed in
12595    // full the moment the column is first touched, and the order is half again the size of the
12596    // offsets, so putting it there would make every query that reads a string column pay for a
12597    // search that most of them never make.
12598    let ranks = count;
12599    let rank_blocks = ranks.div_ceil(TEXT_RANK_BLOCK);
12600    // Three words a payload block, for where it starts, how long it is and what it hashes to, or
12601    // two of them on a file that has the blocks back to back and needs no start. Two a rank block
12602    // either way, since those are still one run.
12603    let payload_words = if scattered { 3 } else { 2 };
12604    let hash_len = blocks
12605        .checked_mul(payload_words * 8)
12606        .and_then(|len| len.checked_add(rank_blocks.checked_mul(16)?))
12607        .and_then(|len| len.checked_add(usize::from(has_grams) * 8))
12608        .ok_or_else(|| invalid("global dictionary block count overflow"))?;
12609    let gram_len = if has_grams {
12610        blocks
12611            .checked_mul(gram_width)
12612            .ok_or_else(|| invalid("global dictionary signature count overflow"))?
12613    } else {
12614        0
12615    };
12616    let index_len = DICTIONARY_HEADER
12617        .checked_add(offset_len)
12618        .and_then(|len| len.checked_add(hash_len))
12619        .ok_or_else(|| invalid("global dictionary header overflow"))?;
12620    if index_len > page.length as usize {
12621        return Err(invalid("global dictionary offset index exceeds its page"));
12622    }
12623    let mut index = vec![0; index_len];
12624    index[..DICTIONARY_HEADER].copy_from_slice(&header);
12625    read_at(&file, page.offset + DICTIONARY_HEADER as u64, &mut index[DICTIONARY_HEADER..])?;
12626    if checksum(&index) != page.hash {
12627        return Err(invalid("global dictionary index checksum differs"));
12628    }
12629    let word_end = index_len - usize::from(has_grams) * 8;
12630    let gram_hash = has_grams
12631        .then(|| u64::from_le_bytes(index[word_end..index_len].try_into().expect("eight bytes")));
12632    let mut words = index[DICTIONARY_HEADER + offset_len..word_end]
12633        .chunks_exact(8)
12634        .map(|part| u64::from_le_bytes(part.try_into().expect("eight bytes")))
12635        .collect::<Vec<_>>();
12636    let mut rest = words.split_off(blocks * payload_words);
12637    let rank_hashes = rest.split_off(rank_blocks);
12638    let rank_ends = rest;
12639    // A rank block packs its heads at whatever width its own values need, so its length is no longer
12640    // arithmetic on the block number and the reader has to be told where each one ends.
12641    if rank_ends.windows(2).any(|pair| pair[0] >= pair[1]) {
12642        return Err(invalid("global dictionary order blocks do not rise"));
12643    }
12644    let rank_len = usize::try_from(rank_ends.last().copied().unwrap_or_default())
12645        .map_err(|_| invalid("global dictionary rank overflow"))?;
12646    let body_len = index_len
12647        .checked_add(rank_len)
12648        .ok_or_else(|| invalid("global dictionary header overflow"))?;
12649    if body_len > page.length as usize {
12650        return Err(invalid("global dictionary order exceeds its page"));
12651    }
12652    let gram_end = body_len
12653        .checked_add(gram_len)
12654        .ok_or_else(|| invalid("global dictionary signature length overflow"))?;
12655    if gram_end > page.length as usize {
12656        return Err(invalid("global dictionary signatures exceed their page"));
12657    }
12658    let grams = gram_hash.map(|hash| NativeGrams {
12659        start: page.offset + body_len as u64,
12660        length: gram_len,
12661        width: gram_width,
12662        hash,
12663        verdicts: Mutex::new(Vec::new()),
12664    });
12665    // The offsets stay where they were read, behind the header, rather than being copied out. On a
12666    // dictionary of millions of values they are megabytes, and a copy is as many fresh pages to
12667    // fault in again on a query that may want a handful of strings.
12668    let mut offsets = index;
12669    offsets.truncate(DICTIONARY_HEADER + offset_len);
12670    let hashes = words.split_off(blocks * (payload_words - 1));
12671    let (starts, lengths) = if scattered {
12672        let mut starts = Vec::with_capacity(blocks);
12673        let mut lengths = Vec::with_capacity(blocks);
12674        for pair in words.chunks_exact(2) {
12675            starts.push(pair[0]);
12676            lengths.push(pair[1]);
12677        }
12678        (starts, lengths)
12679    } else {
12680        // A file written before the blocks said where they were has them behind one another at the
12681        // end of the page, so the base is where the sorted order stops and each end is the start of
12682        // the one after it. Turning them round here is what lets everything below take one shape.
12683        let base = page.offset + gram_end as u64;
12684        let mut starts = Vec::with_capacity(blocks);
12685        let mut lengths = Vec::with_capacity(blocks);
12686        let mut at = 0_u64;
12687        for &end in &words {
12688            let len = end
12689                .checked_sub(at)
12690                .ok_or_else(|| invalid("global dictionary block ends before it starts"))?;
12691            starts.push(base + at);
12692            lengths.push(len);
12693            at = end;
12694        }
12695        (starts, lengths)
12696    };
12697    // What the offsets bound is the decoded payload, and what the page length counts is the stored
12698    // one, so on a format 26 file the block lengths adding up to the rest of the page is the one
12699    // thing that ties the index to the page. From format 27 the blocks are written during the load
12700    // and the page is only the index and the order, so there the most that can be said is that
12701    // every block is somewhere in the file past its header.
12702    let stored_len = page.length as u64 - gram_end as u64;
12703    if scattered && stored_len == 0 {
12704        let size = file.metadata().map_err(io)?.len();
12705        let inside = starts.iter().zip(&lengths).all(|(&start, &len)| {
12706            start >= HEADER && start.checked_add(len).is_some_and(|end| end <= size)
12707        });
12708        if !inside {
12709            return Err(invalid("global dictionary block lies outside the file"));
12710        }
12711    } else if lengths.iter().try_fold(0_u64, |sum, len| sum.checked_add(*len)) != Some(stored_len) {
12712        return Err(invalid("global dictionary blocks do not bound the payload"));
12713    }
12714    Vector::external_text(
12715        ty.clone(),
12716        Arc::new(NativeText {
12717            file,
12718            values: count,
12719            offsets,
12720            offset_bits,
12721            value_ends: OnceLock::new(),
12722            value_lens: OnceLock::new(),
12723            ends_asked: AtomicUsize::new(0),
12724            ranks,
12725            rank_at: page.offset + index_len as u64,
12726            rank_ends,
12727            rank_hashes,
12728            rank_blocks: (0..rank_blocks).map(|_| OnceLock::new()).collect(),
12729            code_bits: code_width(count),
12730            code_ranks: OnceLock::new(),
12731            starts,
12732            lengths,
12733            hashes,
12734            grams,
12735            blocks: (0..blocks).map(|_| OnceLock::new()).collect(),
12736            char_lens: (0..blocks).map(|_| OnceLock::new()).collect(),
12737            keep_budget,
12738            payload_kept: AtomicUsize::new(0),
12739            swept: (0..blocks).map(|_| AtomicBool::new(false)).collect(),
12740            visit_dropped: AtomicUsize::new(0),
12741            searched: Mutex::new(HashMap::new()),
12742        }),
12743    )
12744}
12745
12746/// What a stored page is, without decoding a value out of it.
12747///
12748/// Two layers, and both of them belong in the answer. The codec byte at the front of every page is
12749/// the format's own choice, and it is what says whether the column came back as codes into a table
12750/// wide dictionary, as a bit packed page, as an encoding cascade or as the bytes themselves. Under
12751/// the cascade codecs there is a second choice the encoder made per chunk, and that is what
12752/// [`integer::describe`] and [`string::describe`] already write out as `DICT(PACKED, PACKED)`.
12753///
12754/// This mirrors the tags [`decode`] reads and has to be kept beside it. A page whose header this
12755/// cannot walk comes back as text rather than as an error, because a caller asking what a file
12756/// looks like is usually asking because something is wrong with it, and a report that stops at the
12757/// first bad page is a report that says nothing about the other nine hundred.
12758fn page_encoding(ty: &LogicalType, rows: usize, bytes: &[u8]) -> String {
12759    /// The page header is the codec, the validity tag and, for a page that stores a mask, the mask.
12760    fn cascade_at(rows: usize, bytes: &[u8]) -> Result<(u8, usize)> {
12761        let mut cur = Cursor::new(bytes);
12762        let codec = cur.u8()?;
12763        if cur.u8()? == 2 {
12764            cur.take(rows.div_ceil(8))?;
12765        }
12766        Ok((codec, cur.at))
12767    }
12768    let Ok((codec, at)) = cascade_at(rows, bytes) else {
12769        return "UNREADABLE".to_string();
12770    };
12771    let tail = &bytes[at..];
12772    let described = |described: Result<String>| described.unwrap_or_else(|_| "UNREADABLE".into());
12773    match codec {
12774        0 => match ty {
12775            LogicalType::Varchar | LogicalType::Blob => "PLAIN".to_string(),
12776            _ => "FIXED".to_string(),
12777        },
12778        1 => "DICT(PLAIN)".to_string(),
12779        2 => "FOR+BITPACK".to_string(),
12780        3 => "TABLE DICT".to_string(),
12781        4 => format!("TABLE DICT({})", described(integer::describe(tail))),
12782        5 => described(integer::describe(tail)),
12783        6 => described(string::describe(tail)),
12784        other => format!("CODEC {other}"),
12785    }
12786}
12787
12788/// Selected stable dictionary codes from one page.
12789///
12790/// Pair-frequency construction needs at most the bounded heavy-hitter rows. Reading those code
12791/// positions directly avoids materializing every code in each part that contains a candidate.
12792fn decode_selected_stable_codes(
12793    rows: usize,
12794    bytes: &[u8],
12795    positions: &[usize],
12796    out: &mut Vec<Option<u32>>,
12797) -> Result<bool> {
12798    if positions.windows(2).any(|pair| pair[0] >= pair[1])
12799        || positions.last().is_some_and(|&position| position >= rows)
12800    {
12801        return Err(invalid("selected code positions are not sorted and in range"));
12802    }
12803    let mut cur = Cursor::new(bytes);
12804    let codec = cur.u8()?;
12805    if codec != 3 && codec != 4 {
12806        return Ok(false);
12807    }
12808    let flag = cur.u8()?;
12809    let mask = match flag {
12810        0 | 1 => None,
12811        2 => {
12812            let at = cur.at;
12813            let len = rows.div_ceil(8);
12814            cur.take(len)?;
12815            Some((at, len))
12816        }
12817        _ => return Err(invalid("page validity tag differs")),
12818    };
12819    let valid = |row: usize| match flag {
12820        0 => true,
12821        1 => false,
12822        2 => mask.is_some_and(|(at, _)| bytes[at + row / 8] >> (row % 8) & 1 == 1),
12823        _ => unreachable!("the validity tag was checked"),
12824    };
12825    if codec == 4 {
12826        let wide = integer::decode_selected(&bytes[cur.at..], positions)?;
12827        for (&row, code) in positions.iter().zip(wide) {
12828            let code = u32::try_from(code).map_err(|_| invalid("code is not a code"))?;
12829            out.push(valid(row).then_some(code));
12830        }
12831        return Ok(true);
12832    }
12833    let codes_at = cur.at;
12834    let codes_len = rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?;
12835    cur.take(codes_len)?;
12836    if cur.at != bytes.len() {
12837        return Err(invalid("global code page has trailing bytes"));
12838    }
12839    let codes = &bytes[codes_at..codes_at + codes_len];
12840    for &row in positions {
12841        let at = row.checked_mul(4).ok_or_else(|| invalid("dictionary code offset overflow"))?;
12842        let code = u32::from_le_bytes(
12843            codes[at..at + 4].try_into().map_err(|_| invalid("dictionary code is truncated"))?,
12844        );
12845        out.push(valid(row).then_some(code));
12846    }
12847    Ok(true)
12848}
12849
12850/// [`decode`] of only the rows at `positions`, which rise.
12851///
12852/// A compressed text page decompresses only those rows, see [`string::decode_flat_at`], and checks
12853/// only those rows are text. Every other page is decoded whole and gathered, since its values are
12854/// fixed width or its strings are shared through a dictionary, and there picking comes after.
12855fn decode_at(
12856    ty: &LogicalType,
12857    rows: usize,
12858    bytes: &[u8],
12859    global: Option<Arc<Vector>>,
12860    positions: &[u32],
12861) -> Result<Vector> {
12862    if positions.last().is_some_and(|&last| last as usize >= rows) {
12863        return Err(invalid("a position is past the end of the part"));
12864    }
12865    // Past about one row in eight, unpacking the whole part and picking the rows out is the cheaper
12866    // of the two, since a unit unpacks at a fraction of what a row unpacked on its own costs.
12867    if bytes.first() == Some(&5)
12868        && positions.len().saturating_mul(8) <= rows
12869        // Past the codec, the validity flag and the mask a flag of 2 has.
12870        && bytes
12871            .get(2 + if bytes.get(1) == Some(&2) { rows.div_ceil(8) } else { 0 }..)
12872            .is_some_and(|body| integer::pointed(body) || integer::run_length(body))
12873    {
12874        return cascade_at(ty, rows, bytes, positions);
12875    }
12876    if bytes.first() != Some(&6) {
12877        return decode(ty, rows, bytes, global)?.gather(positions);
12878    }
12879    if !coded_type(ty) {
12880        return Err(invalid("compressed text codec belongs to a non-string page"));
12881    }
12882    let mut cur = Cursor::new(bytes);
12883    cur.u8()?;
12884    let validity = match cur.u8()? {
12885        0 => Validity::AllValid,
12886        1 => Validity::AllInvalid,
12887        2 => {
12888            let mask = cur.take(rows.div_ceil(8))?;
12889            Validity::from_iter(positions.len(), |at| {
12890                let row = positions[at] as usize;
12891                mask[row / 8] >> (row % 8) & 1 == 1
12892            })
12893        }
12894        _ => return Err(invalid("page validity tag differs")),
12895    };
12896    let (payload, ends) = string::decode_flat_at(&bytes[cur.at..], positions)?.into_parts();
12897    let mut values = StringColumn::over(Buffer::from_vec(payload).into_page());
12898    push_values(&mut values, ty, &ends)?;
12899    Ok(Vector::flat(ty.clone(), Data::Varlen(values))?.with_validity(validity))
12900}
12901
12902/// The rows `positions` names of an integer cascade page, unpacked at those rows alone.
12903///
12904/// A scan whose join keeps a few rows in a thousand reads its other columns only at those rows, and
12905/// decoding the whole part to pick them out afterwards was most of what it cost. In TPC-H q17 the
12906/// bitmap over the parts of one brand and container keeps about one `lineitem` row in a thousand.
12907fn cascade_at(ty: &LogicalType, rows: usize, bytes: &[u8], positions: &[u32]) -> Result<Vector> {
12908    fn wanted<T: integer::Lane>(values: &[i64]) -> Result<Vec<T>> {
12909        values
12910            .iter()
12911            .map(|&value| T::fit(value).ok_or_else(|| invalid("page value is not of its type")))
12912            .collect()
12913    }
12914    let mut cur = Cursor::new(bytes);
12915    cur.u8()?;
12916    let validity = match cur.u8()? {
12917        0 => Validity::AllValid,
12918        1 => Validity::AllInvalid,
12919        2 => {
12920            let mask = cur.take(rows.div_ceil(8))?;
12921            Validity::from_iter(positions.len(), |at| {
12922                let row = positions[at] as usize;
12923                mask[row / 8] >> (row % 8) & 1 == 1
12924            })
12925        }
12926        _ => return Err(invalid("page validity tag differs")),
12927    };
12928    let at: Vec<usize> = positions.iter().map(|&row| row as usize).collect();
12929    let values = integer::decode_selected(&bytes[cur.at..], &at)
12930        .map_err(|error| invalid(&format!("page value is not of its type: {error}")))?;
12931    if values.len() != positions.len() {
12932        return Err(invalid("cascade page holds the wrong number of rows"));
12933    }
12934    let data = match ty {
12935        LogicalType::TinyInt => Data::Int8(wanted::<i8>(&values)?.into()),
12936        LogicalType::UTinyInt => Data::UInt8(wanted::<u8>(&values)?.into()),
12937        LogicalType::SmallInt => Data::Int16(wanted::<i16>(&values)?.into()),
12938        LogicalType::USmallInt => Data::UInt16(wanted::<u16>(&values)?.into()),
12939        LogicalType::Integer | LogicalType::Date => Data::Int32(wanted::<i32>(&values)?.into()),
12940        LogicalType::UInteger => Data::UInt32(wanted::<u32>(&values)?.into()),
12941        LogicalType::BigInt
12942        | LogicalType::Timestamp
12943        | LogicalType::Time
12944        | LogicalType::TimeTz
12945        | LogicalType::TimestampTz
12946        | LogicalType::TimestampS
12947        | LogicalType::TimestampMs
12948        | LogicalType::TimestampNs => Data::Int64(values.into()),
12949        LogicalType::Decimal { .. } => match ty.physical() {
12950            PhysicalType::Int16 => Data::Int16(wanted::<i16>(&values)?.into()),
12951            PhysicalType::Int32 => Data::Int32(wanted::<i32>(&values)?.into()),
12952            PhysicalType::Int64 => Data::Int64(values.into()),
12953            _ => return Err(invalid("cascade codec belongs to a decimal that is not an integer")),
12954        },
12955        _ => return Err(invalid("cascade codec belongs to a page that is not integers")),
12956    };
12957    Ok(Vector::flat(ty.clone(), data)?.with_validity(validity))
12958}
12959
12960/// The values of a string or blob page, laid end to end in the page's payload from its start, each
12961/// ending where `ends` says. A varchar is checked for text on the way in, once over the whole run,
12962/// and a blob or a bit string is not, since neither ever claimed to hold any.
12963fn push_values(values: &mut StringColumn, ty: &LogicalType, ends: &[usize]) -> Result<()> {
12964    if ty == &LogicalType::Varchar {
12965        return values.push_run_in_place(0, ends);
12966    }
12967    let mut start = 0;
12968    for &end in ends {
12969        let len = end
12970            .checked_sub(start)
12971            .ok_or_else(|| invalid("a string value ends before it starts"))?;
12972        values.push_bytes_in_place(start, len)?;
12973        start = end;
12974    }
12975    Ok(())
12976}
12977
12978fn decode(
12979    ty: &LogicalType,
12980    rows: usize,
12981    bytes: &[u8],
12982    global: Option<Arc<Vector>>,
12983) -> Result<Vector> {
12984    let mut cur = Cursor::new(bytes);
12985    let codec = cur.u8()?;
12986    let flag = cur.u8()?;
12987    let validity = match flag {
12988        0 => Validity::AllValid,
12989        1 => Validity::AllInvalid,
12990        2 => {
12991            let mask = cur.take(rows.div_ceil(8))?;
12992            Validity::from_iter(rows, |row| mask[row / 8] >> (row % 8) & 1 == 1)
12993        }
12994        _ => return Err(invalid("page validity tag differs")),
12995    };
12996    if codec == 1 {
12997        if !coded_type(ty) {
12998            return Err(invalid("dictionary codec belongs to a non-string page"));
12999        }
13000        let count = cur.u32()? as usize;
13001        let payload_len = cur.u32()? as usize;
13002        let offset_bytes = cur.take(
13003            (count + 1)
13004                .checked_mul(4)
13005                .ok_or_else(|| invalid("dictionary offset count overflow"))?,
13006        )?;
13007        let offsets = offset_bytes
13008            .chunks_exact(4)
13009            .map(|part| u32::from_le_bytes(part.try_into().expect("four bytes")))
13010            .collect::<Vec<_>>();
13011        let payload = cur.take(payload_len)?.to_vec();
13012        if offsets.first() != Some(&0)
13013            || offsets.last().copied().map(|last| last as usize) != Some(payload.len())
13014            || offsets.windows(2).any(|pair| pair[0] > pair[1])
13015        {
13016            return Err(invalid("dictionary offsets do not bound the payload"));
13017        }
13018        // A page, because every chunk cut out of this dictionary points at the same payload and a
13019        // page is what lets a cut be the views and nothing else.
13020        let mut strings = StringColumn::over(Buffer::from_vec(payload).into_page());
13021        let ends: Vec<usize> = offsets[1..].iter().map(|&end| end as usize).collect();
13022        push_values(&mut strings, ty, &ends)?;
13023        let mut codes = Vec::with_capacity(rows);
13024        for _ in 0..rows {
13025            codes.push(cur.u32()?);
13026        }
13027        if codes.iter().any(|code| *code as usize >= count) {
13028            return Err(invalid("dictionary code is out of range"));
13029        }
13030        if cur.at != bytes.len() {
13031            return Err(invalid("dictionary page has trailing bytes"));
13032        }
13033        let dictionary = Vector::flat(ty.clone(), Data::Varlen(strings))?;
13034        return Ok(Vector::dictionary(codes, dictionary)?.with_validity(validity));
13035    }
13036    if codec == 3 || codec == 4 {
13037        let dictionary = global.ok_or_else(|| invalid("global code page has no dictionary"))?;
13038        let codes = if codec == 4 {
13039            // The cascade holds the whole tail of the page and says how long it is itself, so the
13040            // check that nothing is left over is the one the decoder already makes.
13041            // Straight into `u32`, which is also the check that every code is one: a code outside
13042            // it is a corrupt file and the decoder refuses it, a block at a time where it can.
13043            let codes = integer::decode_as::<u32>(&bytes[cur.at..])
13044                .map_err(|error| invalid(&format!("code is not a code: {error}")))?;
13045            if codes.len() != rows {
13046                return Err(invalid("encoded code page holds the wrong number of rows"));
13047            }
13048            codes
13049        } else {
13050            let mut codes = Vec::with_capacity(rows);
13051            for _ in 0..rows {
13052                codes.push(cur.u32()?);
13053            }
13054            if cur.at != bytes.len() {
13055                return Err(invalid("global code page has trailing bytes"));
13056            }
13057            codes
13058        };
13059        let highest = codes.iter().copied().max();
13060        return Ok(Vector::stable_dictionary_validated(codes, dictionary, highest)?
13061            .with_validity(validity));
13062    }
13063    if codec == 6 {
13064        if !coded_type(ty) {
13065            return Err(invalid("compressed text codec belongs to a non-string page"));
13066        }
13067        // As codec 5, the layer holds the whole tail of the page and says how long it is itself.
13068        // It comes back as one buffer with the values laid end to end and where each one ends, which
13069        // is the raw form's layout, so what is left to do here is what codec 0 does.
13070        let (payload, ends) = string::decode_flat(&bytes[cur.at..])?.into_parts();
13071        if ends.len() != rows {
13072            return Err(invalid("compressed text page holds the wrong number of rows"));
13073        }
13074        // A page, because this is read once and handed out a chunk at a time, and a cut of a paged
13075        // payload moves views rather than bytes.
13076        let mut values = StringColumn::over(Buffer::from_vec(payload).into_page());
13077        push_values(&mut values, ty, &ends)?;
13078        return Ok(Vector::flat(ty.clone(), Data::Varlen(values))?.with_validity(validity));
13079    }
13080    if codec == 5 {
13081        // The cascade holds the whole tail of the page and says how long it is itself.
13082        let data = cascade(ty, &bytes[cur.at..], rows)?;
13083        return Ok(Vector::flat(ty.clone(), data)?.with_validity(validity));
13084    }
13085    if codec == 2 {
13086        let width = u32::from(cur.u8()?);
13087        let base = i128::from_le_bytes(cur.take(16)?.try_into().expect("sixteen bytes"));
13088        let count = cur.u32()? as usize;
13089        let length = count.checked_mul(8).ok_or_else(|| invalid("packed page is too long"))?;
13090        let words: Vec<u64> = cur
13091            .take(length)?
13092            .chunks_exact(8)
13093            .map(|word| u64::from_le_bytes(word.try_into().expect("eight bytes")))
13094            .collect();
13095        if cur.at != bytes.len() {
13096            return Err(invalid("packed page has trailing bytes"));
13097        }
13098        return Ok(Vector::packed(ty.clone(), words, width, base, rows)?.with_validity(validity));
13099    }
13100    if codec != 0 {
13101        return Err(invalid("page codec is unknown"));
13102    }
13103    let data = match ty {
13104        LogicalType::TinyInt => {
13105            let values = cur.take(rows)?;
13106            Data::Int8(values.iter().map(|item| *item as i8).collect::<Vec<_>>().into())
13107        }
13108        LogicalType::UTinyInt => Data::UInt8(cur.take(rows)?.to_vec().into()),
13109        LogicalType::SmallInt => {
13110            let values =
13111                cur.take(rows.checked_mul(2).ok_or_else(|| invalid("page size overflow"))?)?;
13112            Data::Int16(
13113                values
13114                    .chunks_exact(2)
13115                    .map(|item| i16::from_le_bytes(item.try_into().expect("two bytes")))
13116                    .collect::<Vec<_>>()
13117                    .into(),
13118            )
13119        }
13120        LogicalType::USmallInt => {
13121            let values =
13122                cur.take(rows.checked_mul(2).ok_or_else(|| invalid("page size overflow"))?)?;
13123            Data::UInt16(
13124                values
13125                    .chunks_exact(2)
13126                    .map(|item| u16::from_le_bytes(item.try_into().expect("two bytes")))
13127                    .collect::<Vec<_>>()
13128                    .into(),
13129            )
13130        }
13131        LogicalType::UInteger => {
13132            let values =
13133                cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
13134            Data::UInt32(
13135                values
13136                    .chunks_exact(4)
13137                    .map(|item| u32::from_le_bytes(item.try_into().expect("four bytes")))
13138                    .collect::<Vec<_>>()
13139                    .into(),
13140            )
13141        }
13142        LogicalType::UBigInt => {
13143            let values =
13144                cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
13145            Data::UInt64(
13146                values
13147                    .chunks_exact(8)
13148                    .map(|item| u64::from_le_bytes(item.try_into().expect("eight bytes")))
13149                    .collect::<Vec<_>>()
13150                    .into(),
13151            )
13152        }
13153        LogicalType::Integer | LogicalType::Date => {
13154            let values =
13155                cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
13156            Data::Int32(
13157                values
13158                    .chunks_exact(4)
13159                    .map(|item| i32::from_le_bytes(item.try_into().expect("four bytes")))
13160                    .collect::<Vec<_>>()
13161                    .into(),
13162            )
13163        }
13164        LogicalType::BigInt
13165        | LogicalType::Timestamp
13166        | LogicalType::Time
13167        | LogicalType::TimeTz
13168        | LogicalType::TimestampTz
13169        | LogicalType::TimestampS
13170        | LogicalType::TimestampMs
13171        | LogicalType::TimestampNs => {
13172            let values =
13173                cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
13174            Data::Int64(
13175                values
13176                    .chunks_exact(8)
13177                    .map(|item| i64::from_le_bytes(item.try_into().expect("eight bytes")))
13178                    .collect::<Vec<_>>()
13179                    .into(),
13180            )
13181        }
13182        LogicalType::HugeInt | LogicalType::Uuid => {
13183            let values =
13184                cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
13185            Data::Int128(
13186                values
13187                    .chunks_exact(16)
13188                    .map(|item| i128::from_le_bytes(item.try_into().expect("sixteen bytes")))
13189                    .collect::<Vec<_>>()
13190                    .into(),
13191            )
13192        }
13193        LogicalType::UHugeInt => {
13194            let values =
13195                cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
13196            Data::UInt128(
13197                values
13198                    .chunks_exact(16)
13199                    .map(|item| u128::from_le_bytes(item.try_into().expect("sixteen bytes")))
13200                    .collect::<Vec<_>>()
13201                    .into(),
13202            )
13203        }
13204        LogicalType::Float => {
13205            let values =
13206                cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
13207            Data::Float32(
13208                values
13209                    .chunks_exact(4)
13210                    .map(|item| f32::from_le_bytes(item.try_into().expect("four bytes")))
13211                    .collect::<Vec<_>>()
13212                    .into(),
13213            )
13214        }
13215        LogicalType::Double => {
13216            let values =
13217                cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
13218            Data::Float64(
13219                values
13220                    .chunks_exact(8)
13221                    .map(|item| f64::from_le_bytes(item.try_into().expect("eight bytes")))
13222                    .collect::<Vec<_>>()
13223                    .into(),
13224            )
13225        }
13226        LogicalType::Interval => {
13227            let values =
13228                cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
13229            Data::Interval(
13230                values
13231                    .chunks_exact(16)
13232                    .map(|item| {
13233                        (
13234                            i32::from_le_bytes(item[..4].try_into().expect("four bytes")),
13235                            i32::from_le_bytes(item[4..8].try_into().expect("four bytes")),
13236                            i64::from_le_bytes(item[8..].try_into().expect("eight bytes")),
13237                        )
13238                    })
13239                    .collect::<Vec<_>>()
13240                    .into(),
13241            )
13242        }
13243        LogicalType::Boolean => {
13244            let values = cur.take(rows)?;
13245            if values.iter().any(|value| *value > 1) {
13246                return Err(invalid("boolean page has another value"));
13247            }
13248            Data::Bool(values.iter().map(|value| *value == 1).collect::<Vec<_>>().into())
13249        }
13250        // Whichever integer the declared width says, which is the mapping the rest of the engine
13251        // already uses for a decimal in memory.
13252        LogicalType::Decimal { .. } => match ty.physical() {
13253            PhysicalType::Int16 => {
13254                let values =
13255                    cur.take(rows.checked_mul(2).ok_or_else(|| invalid("page size overflow"))?)?;
13256                Data::Int16(
13257                    values
13258                        .chunks_exact(2)
13259                        .map(|item| i16::from_le_bytes(item.try_into().expect("two bytes")))
13260                        .collect::<Vec<_>>()
13261                        .into(),
13262                )
13263            }
13264            PhysicalType::Int32 => {
13265                let values =
13266                    cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
13267                Data::Int32(
13268                    values
13269                        .chunks_exact(4)
13270                        .map(|item| i32::from_le_bytes(item.try_into().expect("four bytes")))
13271                        .collect::<Vec<_>>()
13272                        .into(),
13273                )
13274            }
13275            PhysicalType::Int64 => {
13276                let values =
13277                    cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
13278                Data::Int64(
13279                    values
13280                        .chunks_exact(8)
13281                        .map(|item| i64::from_le_bytes(item.try_into().expect("eight bytes")))
13282                        .collect::<Vec<_>>()
13283                        .into(),
13284                )
13285            }
13286            _ => {
13287                let values =
13288                    cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
13289                Data::Int128(
13290                    values
13291                        .chunks_exact(16)
13292                        .map(|item| i128::from_le_bytes(item.try_into().expect("sixteen bytes")))
13293                        .collect::<Vec<_>>()
13294                        .into(),
13295                )
13296            }
13297        },
13298        LogicalType::Varchar | LogicalType::Blob | LogicalType::Bit => {
13299            let offset_bytes = cur
13300                .take((rows + 1).checked_mul(4).ok_or_else(|| invalid("offset count overflow"))?)?;
13301            let offsets = offset_bytes
13302                .chunks_exact(4)
13303                .map(|part| u32::from_le_bytes(part.try_into().expect("four bytes")))
13304                .collect::<Vec<_>>();
13305            let payload = cur.take(bytes.len() - cur.at)?.to_vec();
13306            if offsets.first() != Some(&0)
13307                || offsets.last().copied().map(|last| last as usize) != Some(payload.len())
13308                || offsets.windows(2).any(|pair| pair[0] > pair[1])
13309            {
13310                return Err(invalid("string offsets do not bound the payload"));
13311            }
13312            // A page for the reason the dictionary payload above is one: the page is read once and
13313            // handed out a chunk at a time, and a cut of a paged payload moves views rather than
13314            // bytes.
13315            //
13316            // A varchar is checked for text on the way in and a blob and a bit string are not,
13317            // because the second pair never claimed to hold any. Reading them through the checking
13318            // seam would refuse a column for holding exactly what it was told to hold.
13319            let mut values = StringColumn::over(Buffer::from_vec(payload).into_page());
13320            let ends: Vec<usize> = offsets[1..].iter().map(|&end| end as usize).collect();
13321            push_values(&mut values, ty, &ends)?;
13322            Data::Varlen(values)
13323        }
13324        _ => return Err(Error::not_implemented(format!("native page for {ty}"))),
13325    };
13326    if cur.at != bytes.len() {
13327        return Err(invalid("page has trailing bytes"));
13328    }
13329    Ok(Vector::flat(ty.clone(), data)?.with_validity(validity))
13330}
13331
13332#[cfg(test)]
13333mod tests {
13334    use std::fs::{self, OpenOptions};
13335    use std::io::{Seek, SeekFrom, Write};
13336    use std::path::PathBuf;
13337    use std::time::{SystemTime, UNIX_EPOCH};
13338
13339    use rudb_common::Stat;
13340    use rudb_common::Value;
13341    use rudb_common::bounds::{Frequencies, Op, Remainder, Zones};
13342    use rudb_common::stat::Provenance;
13343
13344    use super::*;
13345
13346    #[test]
13347    fn head_is_the_value_padded_to_eight_bytes() {
13348        let bytes: Vec<u8> = (1..=12).collect();
13349        for len in 0..=bytes.len() {
13350            let value = &bytes[..len];
13351            let mut word = [0; 8];
13352            let take = len.min(8);
13353            word[..take].copy_from_slice(&value[..take]);
13354            assert_eq!(head(value), u64::from_be_bytes(word), "{len} bytes");
13355        }
13356        assert!(head(b"ab") < head(b"ab\x01"));
13357        assert!(head(b"abcd") < head(b"abce"));
13358    }
13359
13360    #[test]
13361    fn spanned_frequency_header_rejects_missing_or_out_of_bounds_payloads() {
13362        for (length, entries) in [(0_u32, 1_u32), (9, 0), (1, FREQUENCY_ENTRIES as u32 + 1)] {
13363            let mut bytes = Vec::new();
13364            put_u32(&mut bytes, length);
13365            put_u32(&mut bytes, entries);
13366            bytes.push(1);
13367            assert!(summary_span(&mut Cursor::new(&bytes)).is_err());
13368        }
13369        let mut bytes = Vec::new();
13370        put_u32(&mut bytes, 1);
13371        put_u32(&mut bytes, 0);
13372        bytes.push(1);
13373        assert_eq!(summary_span(&mut Cursor::new(&bytes)).expect("one byte"), Some((1, 0)));
13374    }
13375
13376    #[test]
13377    fn a_name_taken_in_pieces_is_the_name_of_the_pieces_joined() {
13378        let bytes: Vec<u8> =
13379            (0..300_u32).map(|at| (at.wrapping_mul(2_654_435_761) >> 13) as u8).collect();
13380        for length in [0, 1, 7, 31, 32, 33, 63, 64, 65, 100, 300] {
13381            let whole = content_name(&bytes[..length]);
13382            for step in [1, 3, 8, 31, 32, 33, 64, 301] {
13383                let mut namer = ContentNamer::default();
13384                bytes[..length].chunks(step).for_each(|piece| namer.update(piece));
13385                assert_eq!(namer.finish(), whole, "{length} bytes in pieces of {step}");
13386            }
13387        }
13388    }
13389
13390    /// The chooser as it was before it could rule kinds out up front: the same narrowing, with every
13391    /// kind tested for. What it writes is what the file used to hold.
13392    #[derive(Debug)]
13393    struct TestsEverything<'a>(&'a dyn chooser::Chooser);
13394
13395    impl chooser::Chooser for TestsEverything<'_> {
13396        fn name(&self) -> &'static str {
13397            "tests everything"
13398        }
13399
13400        fn narrow_strings(
13401            &self,
13402            values: &[&[u8]],
13403            offered: &[string::Kind],
13404            depth: u8,
13405        ) -> Vec<string::Kind> {
13406            self.0.narrow_strings(values, offered, depth)
13407        }
13408
13409        fn narrow_integers(
13410            &self,
13411            values: &[i64],
13412            offered: &[integer::Kind],
13413            depth: u8,
13414        ) -> Vec<integer::Kind> {
13415            self.0.narrow_integers(values, offered, depth)
13416        }
13417    }
13418
13419    #[test]
13420    fn ruling_kinds_out_before_testing_for_them_writes_the_same_bytes() {
13421        let columns: Vec<Vec<i64>> = vec![
13422            vec![],
13423            vec![5; 1000],
13424            (0..1000).collect(),
13425            (0..1000).map(|row| 1_600_000_000_000_000 + row * 1_000_000).collect(),
13426            (0..1000).map(|row| row / 50).collect(),
13427            (0..1000).map(|row| if row % 97 == 0 { row } else { 0 }).collect(),
13428            (0..1000).map(|row| (row * 7919) % 13).collect(),
13429            (0..1000).map(|row| (row * 2_654_435_761) % 1_000_003).collect(),
13430            (0..1000).map(|row| [3, 3, 3, 9, 9, 1][row as usize % 6]).collect(),
13431            (0..1000).map(|row| i64::MIN + row % 3).collect(),
13432        ];
13433        let choosers: [&dyn chooser::Chooser; 2] = [&Fixed, &Codes];
13434        for column in &columns {
13435            for chooser in choosers {
13436                let quick = integer::encode_with(column, chooser).unwrap();
13437                let full = integer::encode_with(column, &TestsEverything(chooser)).unwrap();
13438                assert_eq!(
13439                    quick,
13440                    full,
13441                    "{} on {:?}",
13442                    chooser.name(),
13443                    &column[..column.len().min(8)]
13444                );
13445            }
13446        }
13447    }
13448
13449    /// Parts of a column that all look alike come out of a settled shape byte for byte as they
13450    /// come out of a search, because the search would have kept the same tree on every one.
13451    #[test]
13452    fn parts_that_look_alike_replay_to_the_bytes_a_search_writes() {
13453        let mut settling = Settling::default();
13454        for part in 0..STRIPE_PARTS as i64 {
13455            let values: Vec<i64> = (0..2048)
13456                .map(|row| 1_600_000_000_000_000 + (part * 2048 + row) * 1_000_000 + row % 7)
13457                .collect();
13458            let searched = integer::encode_with(&values, &Fixed).unwrap();
13459            assert_eq!(settling.encode(&values).unwrap(), searched, "part {part}");
13460        }
13461    }
13462
13463    /// Text pages compressed against a table an earlier page trained read back as they went in, and
13464    /// a page of different text trains a table of its own rather than coming out as big as the
13465    /// earlier table would make it.
13466    #[test]
13467    fn text_pages_share_a_table_until_the_text_changes() {
13468        let words = ["carefully", "final", "deposits", "sleep", "furiously", "quickly", "among"];
13469        let english: Vec<Vec<u8>> = (0..1024)
13470            .map(|row: usize| {
13471                let pick = |at: usize| words[(row * 7 + at * 3) % words.len()];
13472                format!("{} {} {} the {}", pick(0), pick(1), pick(2), pick(3)).into_bytes()
13473            })
13474            .collect();
13475        let digits: Vec<Vec<u8>> =
13476            (0..1024_u64).map(|row| format!("{:020}", row * 7_919_993).into_bytes()).collect();
13477        let mut settling = Settling::default();
13478        for page in 0..8 {
13479            let values: Vec<&[u8]> =
13480                if page < 4 { &english } else { &digits }.iter().map(Vec::as_slice).collect();
13481            let payload = values.iter().map(|value| value.len()).sum();
13482            let out = settling.text(&values, payload).unwrap().unwrap();
13483            assert_eq!(string::decode(&out).unwrap(), values, "page {page}");
13484            let alone = string::encode_only(string::Kind::Fsst, &values).unwrap().unwrap();
13485            assert!(
13486                out.len() * 4 <= alone.len() * 5,
13487                "page {page}: {} against {}",
13488                out.len(),
13489                alone.len()
13490            );
13491            let since = settling.symbols.as_ref().unwrap().since;
13492            assert_eq!(since, page % 4, "page {page}");
13493        }
13494    }
13495
13496    /// A column that changes shape partway through a stripe still reads back, and no part comes
13497    /// out much bigger than a search would have made it, because a replay that stops fitting or
13498    /// grows past a quarter a row is searched.
13499    #[test]
13500    fn a_column_that_changes_under_the_shape_is_searched_again() {
13501        let mut state = 0x9e37_79b9_7f4a_7c15_u64;
13502        let mut noise = move || {
13503            state ^= state << 13;
13504            state ^= state >> 7;
13505            state ^= state << 17;
13506            (state % 1_000_000) as i64
13507        };
13508        let mut settling = Settling::default();
13509        for part in 0..STRIPE_PARTS as i64 {
13510            let values: Vec<i64> = match part / 16 {
13511                0 => (0..2048).map(|row| (part * 2048 + row) / 300).collect(),
13512                1 => (0..2048).map(|_| noise()).collect(),
13513                2 => (0..2048).map(|row| if row % 97 == 0 { row } else { 42 }).collect(),
13514                _ => (0..2048).map(|row| 5 + (part * 2048 + row) * 1_000_000).collect(),
13515            };
13516            let settled = settling.encode(&values).unwrap();
13517            assert_eq!(integer::decode(&settled).unwrap(), values, "part {part}");
13518            let searched = integer::encode_with(&values, &Fixed).unwrap();
13519            assert!(
13520                settled.len() * 4 <= searched.len() * 5,
13521                "part {part}: {} settled against {} searched, {} against {}",
13522                settled.len(),
13523                searched.len(),
13524                integer::describe(&settled).unwrap(),
13525                integer::describe(&searched).unwrap(),
13526            );
13527        }
13528    }
13529
13530    #[test]
13531    fn checksum_matches_fixed_vectors() {
13532        assert_eq!(checksum(b""), 0xef46_db37_51d8_e999);
13533        assert_eq!(checksum(b"a"), 0xd24e_c4f1_a98c_6e5b);
13534        assert_eq!(checksum(b"abc"), 0x44bc_2cf5_ad77_0999);
13535    }
13536
13537    #[test]
13538    fn sorting_across_threads_matches_sorting_on_one() {
13539        let mut state = 0x9e37_79b9_7f4a_7c15_u64;
13540        let mut next = move || {
13541            state ^= state << 13;
13542            state ^= state >> 7;
13543            state ^= state << 17;
13544            state
13545        };
13546        let mut values = Vec::new();
13547        for at in 0..150_000_u64 {
13548            let value = match next() % 6 {
13549                0 => Vec::new(),
13550                1 => format!("https://example.com/{}", next() % 5_000).into_bytes(),
13551                2 => format!("https://example.com/path/{at}").into_bytes(),
13552                3 => b"same".to_vec(),
13553                4 => vec![0xff; (next() % 12) as usize],
13554                _ => (0..next() % 20).map(|_| (next() % 3) as u8).collect(),
13555            };
13556            values.push(value);
13557        }
13558        let value = |code: u32| values[code as usize].as_slice();
13559        for workers in [1, 2, 3, 8, 32] {
13560            let mut one = (0..values.len() as u32).rev().collect::<Vec<_>>();
13561            let mut across = one.clone();
13562            sort_by_value(&mut one, value);
13563            sort_by_value_across(&mut across, value, workers);
13564            assert_eq!(one, across, "{workers} workers");
13565        }
13566        let mut sorted = (0..values.len() as u32).collect::<Vec<_>>();
13567        sort_by_value_across(&mut sorted, value, 8);
13568        assert!(sorted.windows(2).all(|pair| value(pair[0]) <= value(pair[1])));
13569    }
13570
13571    fn path(label: &str) -> PathBuf {
13572        let stamp = SystemTime::now().duration_since(UNIX_EPOCH).expect("time advances").as_nanos();
13573        std::env::temp_dir().join(format!("rudb-native-{label}-{}-{stamp}.rdb", std::process::id()))
13574    }
13575
13576    /// Every value of a dictionary in code order, which the tests have no other way to ask for now
13577    /// that a dictionary does not keep the bytes of the values it has seen.
13578    ///
13579    /// Only valid once `finish_blocks` has run, because until then the last part block is still raw.
13580    fn dictionary_values(dictionary: &GlobalDictionary) -> Vec<Vec<u8>> {
13581        let (flat, bases) = dictionary.decoded(None).expect("the blocks decode");
13582        (0..dictionary.values())
13583            .map(|code| {
13584                let (from, to) = GlobalDictionary::value_span(&dictionary.ends, &bases, code);
13585                flat[from..to].to_vec()
13586            })
13587            .collect()
13588    }
13589
13590    /// The sections a test put in the table, which is every one the writer did not.
13591    ///
13592    /// A table now carries a summary and a sketch per column out of the write itself, and a test
13593    /// about the section table is not about those. Filtering by kind rather than by count, so a
13594    /// table that turns out to have no room for its summaries does not quietly change what these
13595    /// tests are asserting over.
13596    fn attached(table: &Table) -> Vec<&Section> {
13597        table.sections().iter().filter(|held| !held.among(section::STATISTICS_KINDS)).collect()
13598    }
13599
13600    /// A read names the offset it wants, so a cursor somebody else moved cannot reach it.
13601    #[test]
13602    fn a_read_at_an_offset_ignores_where_another_thread_left_the_cursor() {
13603        const SPANS: usize = 64;
13604        const SPAN: usize = 512;
13605        let path = path("positional");
13606        let content: Vec<u8> =
13607            (0..SPANS).flat_map(|span| std::iter::repeat_n(span as u8, SPAN)).collect();
13608        fs::write(&path, &content).expect("the file is written");
13609        let file = Arc::new(File::open(&path).expect("the file opens"));
13610        std::thread::scope(|scope| {
13611            for _ in 0..8 {
13612                let file = Arc::clone(&file);
13613                scope.spawn(move || {
13614                    for _ in 0..64 {
13615                        for span in 0..SPANS {
13616                            let mut bytes = [0_u8; SPAN];
13617                            read_at(&file, (span * SPAN) as u64, &mut bytes)
13618                                .expect("the span reads");
13619                            assert!(
13620                                bytes.iter().all(|byte| *byte == span as u8),
13621                                "span {span} came back as {}",
13622                                bytes[0],
13623                            );
13624                        }
13625                    }
13626                });
13627            }
13628        });
13629        let mut past = [0_u8; SPAN];
13630        let end = (SPANS * SPAN) as u64;
13631        let error = read_at(&file, end, &mut past).expect_err("a read past the end is refused");
13632        assert!(error.message().contains("ends before its declared length"), "{error}");
13633        drop(file);
13634        let _ = fs::remove_file(&path);
13635    }
13636
13637    /// The writer records where it put a page and puts it there.
13638    ///
13639    /// This used to move the file's cursor between the steps that record an offset, which is what
13640    /// reading the pages back to build the frequencies did on a platform with no `pread`, and the
13641    /// directory landed on top of a page. The writer's file is an `rudb_io` file now and has no
13642    /// cursor to move, so what is left is the check that every page is where the directory says.
13643    #[test]
13644    fn a_writer_puts_a_page_where_it_said_it_did_wherever_the_cursor_has_got_to() {
13645        let path = path("cursor");
13646        let mut writer = Writer::create(
13647            &path,
13648            "items",
13649            vec![
13650                Field::required("id", LogicalType::Integer),
13651                Field::new("text", LogicalType::Varchar),
13652            ],
13653        )
13654        .expect("new file");
13655        writer.append(&sample()).expect("first part");
13656        writer.append(&sample()).expect("second part");
13657        writer.finish().expect("commit");
13658        let reader = Reader::open(&path).expect("reopen from disk");
13659        assert_eq!(reader.table().rows(), 6);
13660        let ids = reader.read(0, &[0]).expect("the integer page reads back");
13661        assert_eq!(ids.value_at(0, 0), Value::Integer(4));
13662        assert_eq!(ids.value_at(2, 0), Value::Integer(-2));
13663        let text = reader.read(1, &[1]).expect("the text page reads back");
13664        assert_eq!(text.value_at(1, 0), Value::Null);
13665        assert_eq!(text.value_at(2, 0), Value::Varchar("long text after a slash".into()));
13666        // Nothing the directory points at may run past the end of the file, which is the shape the
13667        // failure took: a page recorded at an offset the directory had already been written over.
13668        let end = reader.table().stripes().iter().flat_map(|stripe| {
13669            stripe
13670                .pages
13671                .iter()
13672                .map(|page| page.offset + u64::from(page.length))
13673                .chain(std::iter::once(stripe.index.offset + u64::from(stripe.index.length)))
13674        });
13675        let last = end.fold(HEADER, u64::max);
13676        let directory = fs::metadata(&path).expect("the file is there").len();
13677        assert!(last <= directory, "a page runs to {last} in a file of {directory} bytes");
13678        fs::remove_file(path).expect("remove scratch file");
13679    }
13680
13681    /// How long a global dictionary index is, read out of the page's own header.
13682    ///
13683    /// The tests below damage a byte of the order or of the payload, so they need to know where each
13684    /// one starts, and working it out here rather than writing a number down means adding something
13685    /// to the index does not quietly turn one of them into a test that damages the index instead.
13686    fn dictionary_index_len(header: &[u8; DICTIONARY_HEADER]) -> u64 {
13687        let count = u64::from(u32::from_le_bytes(header[0..4].try_into().expect("four bytes")));
13688        let blocks = u64::from(u32::from_le_bytes(header[8..12].try_into().expect("four bytes")));
13689        let width = u32::from_le_bytes(header[12..16].try_into().expect("four bytes"));
13690        let bits = (width & !DICTIONARY_FLAGS) as usize;
13691        let payload_words = if width & DICTIONARY_SCATTERED == 0 { 2 } else { 3 };
13692        let rank_blocks = count.div_ceil(TEXT_RANK_BLOCK as u64);
13693        DICTIONARY_HEADER as u64
13694            + offset_bytes(count as usize, bits) as u64
13695            + blocks * payload_words * 8
13696            + rank_blocks * 16
13697            + if width & DICTIONARY_GRAMS == 0 { 0 } else { 8 }
13698    }
13699
13700    fn sample() -> Chunk {
13701        Chunk::new(vec![
13702            Vector::from_values(
13703                LogicalType::Integer,
13704                &[Value::Integer(4), Value::Integer(9), Value::Integer(-2)],
13705            )
13706            .expect("integers"),
13707            Vector::from_values(
13708                LogicalType::Varchar,
13709                &[
13710                    Value::Varchar("alpha".into()),
13711                    Value::Null,
13712                    Value::Varchar("long text after a slash".into()),
13713                ],
13714            )
13715            .expect("strings"),
13716        ])
13717        .expect("matching rows")
13718    }
13719
13720    fn sample_ids() -> Chunk {
13721        Chunk::new(vec![
13722            Vector::flat(LogicalType::Integer, Data::Int32(vec![7, 8, 9].into()))
13723                .expect("integers"),
13724        ])
13725        .expect("one column")
13726    }
13727
13728    #[test]
13729    fn the_planner_gets_the_null_count_off_the_same_directory_the_bounds_are_in() {
13730        // Six rows, two of them null. `IS NULL` used to get the same fifth any unreadable
13731        // condition gets, and the number was in the stripe entry next to the bounds all along.
13732        let path = path("nulls_for_the_planner");
13733        let mut writer =
13734            Writer::create(&path, "items", vec![Field::new("a", LogicalType::Integer)])
13735                .expect("new file");
13736        let rows = Chunk::new(vec![
13737            Vector::from_values(
13738                LogicalType::Integer,
13739                &[
13740                    Value::Integer(4),
13741                    Value::Null,
13742                    Value::Integer(9),
13743                    Value::Null,
13744                    Value::Integer(1),
13745                    Value::Integer(2),
13746                ],
13747            )
13748            .expect("integers"),
13749        ])
13750        .expect("one column");
13751        writer.append(&rows).expect("the only part");
13752        writer.finish().expect("commit");
13753        let reader = Reader::open(&path).expect("reopen from disk");
13754        let stripes = Stripes::new(reader);
13755        let column = stripes.column("a").expect("the file has that column");
13756        assert_eq!(stripes.nulls(column), Stat::exact(2, Provenance::NullCount));
13757        // A column the file does not have. Zero here would be a fact about a column that is not
13758        // there, which the planner would then divide by.
13759        assert_eq!(stripes.nulls(column + 1), Stat::Unknown);
13760        fs::remove_file(&path).expect("clean up");
13761    }
13762
13763    #[test]
13764    fn the_planner_gets_a_leading_count_without_a_complete_numeric_synopsis() {
13765        // Six rows hold three values. The two leading counts help equality planning, while the
13766        // omitted value keeps the directory from being a complete grouped-count result.
13767        let path = path("frequencies_for_the_planner");
13768        let mut writer =
13769            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
13770                .expect("new file");
13771        let rows = Chunk::new(vec![
13772            Vector::from_values(
13773                LogicalType::Integer,
13774                &[
13775                    Value::Integer(4),
13776                    Value::Integer(4),
13777                    Value::Integer(4),
13778                    Value::Integer(9),
13779                    Value::Integer(9),
13780                    Value::Integer(1),
13781                ],
13782            )
13783            .expect("integers"),
13784        ])
13785        .expect("one column");
13786        writer.append(&rows).expect("the only part");
13787        writer.finish().expect("commit");
13788        let reader = Reader::open(&path).expect("reopen from disk");
13789        let common = Common::new(reader);
13790        assert_eq!(common.rows(), 6);
13791        let column = common.column("id").expect("the file has that column");
13792        assert_eq!(common.column("nothing"), None);
13793        assert_eq!(
13794            common.rows_with(column, &Bound::Int(4)),
13795            Stat::exact(3, Provenance::FrequencySynopsis)
13796        );
13797        // An absent value cannot be distinguished from the omitted one by the synopsis.
13798        assert_eq!(common.rows_with(column, &Bound::Int(7)), Stat::Unknown);
13799        // A constant of another domain against an integer column. Nothing in the list compares
13800        // with it, so the zero above would be an artefact of the mismatch rather than a fact.
13801        assert_eq!(common.rows_with(column, &Bound::Bytes(b"four".to_vec())), Stat::Unknown);
13802        assert!(common.remainder(column).is_some());
13803        fs::remove_file(&path).expect("clean up");
13804    }
13805
13806    #[test]
13807    fn string_frequency_estimates_do_not_open_the_global_dictionary() {
13808        let path = path("string_frequencies_for_the_planner");
13809        let mut writer =
13810            Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
13811                .expect("new file");
13812        let rows = Chunk::new(vec![
13813            Vector::from_values(
13814                LogicalType::Varchar,
13815                &[
13816                    Value::Varchar(String::new()),
13817                    Value::Varchar("alpha".into()),
13818                    Value::Varchar(String::new()),
13819                    Value::Varchar("beta".into()),
13820                    Value::Varchar(String::new()),
13821                ],
13822            )
13823            .expect("strings"),
13824        ])
13825        .expect("one column");
13826        writer.append(&rows).expect("the only part");
13827        writer.finish().expect("commit");
13828
13829        let reader = Reader::open(&path).expect("reopen from disk");
13830        assert_eq!(reader.reads().dictionaries, 0, "open reads only the directory");
13831        let common = Common::new(reader.clone());
13832        let column = common.column("text").expect("the file has that column");
13833        assert_eq!(
13834            common.rows_with(column, &Bound::Bytes(Vec::new())),
13835            Stat::exact(3, Provenance::FrequencySynopsis)
13836        );
13837        assert_eq!(
13838            common.rows_with(column, &Bound::Bytes(b"missing".to_vec())),
13839            Stat::exact(0, Provenance::FrequencySynopsis)
13840        );
13841        assert_eq!(
13842            reader.reads().dictionaries,
13843            0,
13844            "the bounded spellings answer without opening the dictionary index"
13845        );
13846        fs::remove_file(&path).expect("clean up");
13847    }
13848
13849    #[test]
13850    fn host_groups_certify_omitted_hosts_and_keep_exact_aggregates() {
13851        let path = path("certified_host_groups");
13852        let mut writer =
13853            Writer::create(&path, "hits", vec![Field::required("Referer", LogicalType::Varchar)])
13854                .expect("new file");
13855        let mut values = vec![Value::Varchar("http://www.example.com/a".into()); 150];
13856        values.extend(vec![Value::Varchar("https://example.com/b".into()); 70]);
13857        values.extend((0..550).map(|at| Value::Varchar(format!("https://site{at}.test/x"))));
13858        values.push(Value::Varchar(String::new()));
13859        for part in values.chunks(512) {
13860            writer
13861                .append(
13862                    &Chunk::new(vec![
13863                        Vector::from_values(LogicalType::Varchar, part).expect("strings"),
13864                    ])
13865                    .expect("one column"),
13866                )
13867                .expect("part written");
13868        }
13869        writer.finish().expect("commit");
13870        let reader = Reader::open(&path).expect("reopen");
13871        assert!(reader.table.host_groups.is_none(), "no query-specific host result is stored");
13872        fs::remove_file(&path).expect("clean up");
13873    }
13874
13875    /// A table directory with nothing in it but a name and one column, for the section tests.
13876    ///
13877    /// The section table is orthogonal to everything else in a directory, so the tests that pin it
13878    /// say so by starting from the emptiest table that encodes.
13879    fn bare_table(sections: Vec<Section>) -> Table {
13880        Table {
13881            name: "linked".to_owned(),
13882            fields: vec![Field::required("id", LogicalType::Integer)],
13883            stripes: Vec::new(),
13884            rows: 0,
13885            dictionaries: vec![None],
13886            dictionary_payloads: Vec::new(),
13887            demoted: Vec::new(),
13888            distincts: vec![None],
13889            frequencies: vec![None],
13890            ordinal_bounds: Vec::new(),
13891            pair_frequencies: Vec::new(),
13892            frequency_texts: Vec::new(),
13893            host_groups: None,
13894            clustering: None,
13895            constraints: Constraints::default(),
13896            generation: 1,
13897            sections,
13898        }
13899    }
13900
13901    fn a_key_map_section() -> Section {
13902        Section {
13903            kind: *section::KEY_MAP,
13904            id: 1,
13905            generation: 3,
13906            extents: 1,
13907            extent_page: HEADER,
13908            extent_bytes: section::EXTENT_BYTES as u32,
13909            hash: 0x1234_5678_9abc_def0,
13910            flags: 0,
13911            header_bytes: 24,
13912        }
13913    }
13914
13915    #[test]
13916    fn a_section_table_round_trips_through_a_directory() {
13917        let mut later = a_key_map_section();
13918        later.kind = *b"RUDBZZ9\0";
13919        later.id = 2;
13920        let table = bare_table(vec![a_key_map_section(), later]);
13921        let directory = encode_directory(&table).expect("directory");
13922        let decoded = decode_directory(&directory, 1 << 20).expect("reopen");
13923        assert_eq!(decoded.sections(), &[a_key_map_section(), later]);
13924        // The second is a kind this build has no name for, and it survived the round trip anyway.
13925        // That is what keeps an old build from silently discarding a newer build's work when it
13926        // rewrites a directory.
13927        assert!(decoded.sections()[0].known());
13928        assert!(!decoded.sections()[1].known());
13929    }
13930
13931    #[test]
13932    fn a_directory_written_before_the_section_table_reads_as_a_table_with_none() {
13933        // The G1 exit criterion, at the directory level. A format 22 directory is exactly this
13934        // build's directory with the trailing section block cut off, so cutting it off is the
13935        // honest way to make one: no fixture to go stale, and no separate encoder to drift.
13936        let directory = encode_directory(&bare_table(Vec::new())).expect("directory");
13937        let block = SECTIONS.len() + size_of::<u64>() + size_of::<u16>();
13938        let older = &directory[..directory.len() - block];
13939        let decoded = decode_directory(older, 1 << 20).expect("a directory from before sections");
13940        assert!(decoded.sections().is_empty());
13941        assert_eq!(decoded.generation(), 0, "a format 22 table recorded no generation");
13942        assert_eq!(decoded.name(), "linked");
13943        assert_eq!(decoded.fields().len(), 1, "everything before the block still decodes");
13944    }
13945
13946    #[test]
13947    fn a_file_stamped_with_the_previous_format_still_opens_and_reads() {
13948        // The same criterion end to end, which is the one the milestone actually asks for: a build
13949        // that knows about sections opens a file written by a build that did not, with no rewrite
13950        // and no repair, and answers from it. The version field is patched rather than a file
13951        // committed by an old binary because the bytes either side of it are identical: format 22
13952        // and format 23 differ only in a trailing directory block, and a reader that stops before
13953        // that block gets a table with no sections.
13954        let path = path("format_twenty_two");
13955        let mut writer =
13956            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
13957                .expect("new file");
13958        let rows = Chunk::new(vec![
13959            Vector::from_values(
13960                LogicalType::Integer,
13961                &[Value::Integer(1), Value::Integer(2), Value::Integer(3)],
13962            )
13963            .expect("integers"),
13964        ])
13965        .expect("one column");
13966        writer.append(&rows).expect("the only part");
13967        writer.finish().expect("commit");
13968
13969        let file = OpenOptions::new().write(true).open(&path).expect("reopen to patch");
13970        write_at(&file, 8, &22_u32.to_le_bytes()).expect("stamp the older format");
13971        drop(file);
13972
13973        let reader = Reader::open(&path).expect("a format 22 file opens unchanged");
13974        assert_eq!(reader.table().rows(), 3);
13975        // The rows and not the section table, because the section block is found by the magic at
13976        // the end of the directory rather than by the number in the header, so stamping the header
13977        // back does not take away the summaries this writer put there. What the test is about is
13978        // that the version check accepts 22, and the rows coming back is what says it did.
13979        assert_eq!(reader.read(0, &[0]).expect("the part still reads").len(), 3);
13980
13981        // And a format this build has never written is still refused, so the accept set is a list
13982        // and not an absence of a check.
13983        let file = OpenOptions::new().write(true).open(&path).expect("reopen to patch");
13984        write_at(&file, 8, &21_u32.to_le_bytes()).expect("stamp an unreadable format");
13985        drop(file);
13986        let error = Reader::open(&path).expect_err("format 21 is not readable");
13987        assert!(error.to_string().contains("format 21"), "{error}");
13988
13989        fs::remove_file(&path).expect("clean up");
13990    }
13991
13992    #[test]
13993    fn a_section_whose_extent_table_is_outside_the_file_is_refused() {
13994        // The bound the format has to check and `section` cannot, because only the reader knows how
13995        // big the file is. Reading the payload a section like this names would be reading whatever
13996        // else happens to be at that offset, which is the one way a graph section could turn into a
13997        // wrong answer rather than a slow one.
13998        let mut past = a_key_map_section();
13999        past.extent_page = 1 << 30;
14000        let directory = encode_directory(&bare_table(vec![past])).expect("directory");
14001        let error = decode_directory(&directory, 1 << 20).expect_err("refused");
14002        assert!(error.to_string().contains("outside the file"), "{error}");
14003
14004        let mut inside_the_header = a_key_map_section();
14005        inside_the_header.extent_page = 8;
14006        let directory = encode_directory(&bare_table(vec![inside_the_header])).expect("directory");
14007        assert!(
14008            decode_directory(&directory, 1 << 20).is_err(),
14009            "a section may not overlap a header"
14010        );
14011    }
14012
14013    #[test]
14014    fn a_section_recorded_as_not_built_is_legal_and_names_no_bytes() {
14015        // Section 3.7: a relationship that does not fit the budget is recorded with its size so
14016        // that `rudb_links()` can report what a larger budget would buy. That record is a section
14017        // entry with no extents, so it has to survive a round trip while naming nothing.
14018        let not_built = Section {
14019            kind: *section::FORWARD_LINK,
14020            id: 9,
14021            generation: 3,
14022            extents: 0,
14023            extent_page: 0,
14024            extent_bytes: 0,
14025            hash: 0,
14026            flags: 0,
14027            header_bytes: 0,
14028        };
14029        let directory = encode_directory(&bare_table(vec![not_built])).expect("directory");
14030        let decoded = decode_directory(&directory, 1 << 20).expect("reopen");
14031        assert_eq!(decoded.sections(), &[not_built]);
14032
14033        // But a section with no extents that still names an extent table is incoherent, and an
14034        // incoherent entry is a torn directory rather than a relationship that was skipped.
14035        let mut incoherent = not_built;
14036        incoherent.extent_bytes = 28;
14037        incoherent.extent_page = HEADER;
14038        let directory = encode_directory(&bare_table(vec![incoherent])).expect("directory");
14039        assert!(decode_directory(&directory, 1 << 20).is_err());
14040    }
14041
14042    #[test]
14043    fn a_directory_naming_more_sections_than_the_bound_is_refused() {
14044        let directory = encode_directory(&bare_table(Vec::new())).expect("directory");
14045        let mut torn = directory.clone();
14046        let count_at = torn.len() - size_of::<u16>();
14047        torn[count_at..].copy_from_slice(&u16::MAX.to_le_bytes());
14048        // Not an allocation of sixty five thousand entries off a torn count: either the bound
14049        // refuses it or the bytes run out, and both are errors rather than a read past the end.
14050        assert!(decode_directory(&torn, 1 << 20).is_err());
14051    }
14052
14053    /// A committed one column file of `rows` integers, for the attach tests.
14054    fn linked_file(label: &str, rows: i32) -> PathBuf {
14055        let path = path(label);
14056        let mut writer =
14057            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
14058                .expect("new file");
14059        let values = (0..rows).map(Value::Integer).collect::<Vec<_>>();
14060        let chunk =
14061            Chunk::new(vec![Vector::from_values(LogicalType::Integer, &values).expect("integers")])
14062                .expect("one column");
14063        writer.append(&chunk).expect("the only part");
14064        writer.finish().expect("commit");
14065        path
14066    }
14067
14068    fn a_key_map_payload() -> Vec<u8> {
14069        // Shaped like one without being one: this crate never reads a payload, so what matters here
14070        // is that every byte comes back and that the header the entry measures is at the front.
14071        (0..512_u32).flat_map(u32::to_le_bytes).collect()
14072    }
14073
14074    #[test]
14075    fn a_section_attached_to_a_committed_file_reads_back_byte_for_byte() {
14076        let path = linked_file("attach", 64);
14077        let payload = a_key_map_payload();
14078        let table = attach(
14079            &path,
14080            "items",
14081            &[section::Attachment {
14082                kind: *section::KEY_MAP,
14083                id: 0,
14084                flags: 2,
14085                header_bytes: 40,
14086                bytes: &payload,
14087            }],
14088        )
14089        .expect("attach a key map");
14090        assert_eq!(attached(&table).len(), 1);
14091
14092        let reader = Reader::open(&path).expect("reopen after the attach");
14093        let held = attached(reader.table());
14094        assert_eq!(held.len(), 1);
14095        assert_eq!(held[0].kind, *section::KEY_MAP);
14096        assert_eq!(held[0].flags, 2, "the form a reader must not have to guess");
14097        assert_eq!(held[0].header_bytes, 40);
14098        // The generation is the one the pages were written at, not the one the attach committed at.
14099        // Attaching a section moved no row, so a section written by it is current, and a second
14100        // table added to this file later would not make it stale.
14101        assert_eq!(held[0].generation, 1);
14102        assert!(held[0].usable(reader.table().generation()));
14103        assert_eq!(reader.payload(held[0]).expect("read the payload"), payload);
14104        assert_eq!(reader.extents(held[0]).expect("extent table").len(), 1);
14105
14106        fs::remove_file(&path).expect("clean up");
14107    }
14108
14109    #[test]
14110    fn attaching_a_section_answers_every_row_exactly_as_before() {
14111        // Section 3.1 end to end, and the reason the whole layer is safe to build incrementally. A
14112        // file with a section in it and the same file without one have to agree row for row, so the
14113        // comparison is made against the answers taken before the attach rather than against a
14114        // constant somebody typed.
14115        let path = linked_file("attach_changes_nothing", 300);
14116        let before = Reader::open(&path).expect("open before");
14117        let rows = before.table().rows();
14118        let first = before.read(0, &[0]).expect("read before");
14119        let values = (0..rows).map(|at| first.value_at(at, 0)).collect::<Vec<_>>();
14120        let layout = before.layout().columns_total();
14121        drop(before);
14122
14123        let payload = a_key_map_payload();
14124        attach(
14125            &path,
14126            "items",
14127            &[section::Attachment {
14128                kind: *section::KEY_MAP,
14129                id: 0,
14130                flags: 0,
14131                header_bytes: 0,
14132                bytes: &payload,
14133            }],
14134        )
14135        .expect("attach");
14136
14137        let after = Reader::open(&path).expect("open after");
14138        assert_eq!(after.table().rows(), rows);
14139        let read = after.read(0, &[0]).expect("read after");
14140        for (at, value) in values.iter().enumerate() {
14141            assert_eq!(&read.value_at(at, 0), value, "row {at} moved");
14142        }
14143        assert_eq!(
14144            after.layout().columns_total(),
14145            layout,
14146            "an attach appends and does not rewrite a column page"
14147        );
14148
14149        fs::remove_file(&path).expect("clean up");
14150    }
14151
14152    #[test]
14153    fn a_rebuilt_section_replaces_the_one_it_supersedes() {
14154        // Rebuilding a key map has to be a write and not a question. If an attach added rather than
14155        // replaced, a table rebuilt a few times would name several maps for one column and a reader
14156        // would have to pick, which is a decision with no right answer in it.
14157        let path = linked_file("attach_twice", 32);
14158        let one = a_key_map_payload();
14159        let two = vec![7_u8; 1024];
14160        let entry = |bytes| section::Attachment {
14161            kind: *section::KEY_MAP,
14162            id: 4,
14163            flags: 1,
14164            header_bytes: 0,
14165            bytes,
14166        };
14167        attach(&path, "items", &[entry(&one)]).expect("first build");
14168        attach(&path, "items", &[entry(&two)]).expect("rebuild");
14169
14170        let reader = Reader::open(&path).expect("reopen");
14171        let held = attached(reader.table());
14172        assert_eq!(held.len(), 1, "one map per column and not one per build");
14173        assert_eq!(reader.payload(held[0]).expect("payload"), two);
14174
14175        fs::remove_file(&path).expect("clean up");
14176    }
14177
14178    #[test]
14179    fn an_attach_carries_through_a_kind_it_does_not_know() {
14180        // The first of section 3.2's three rules, at the point where it is easiest to break: a build
14181        // that rewrites a directory has to carry an entry it has no name for, or opening a file with
14182        // an older binary and attaching one section quietly deletes the work of a newer one.
14183        let path = linked_file("attach_unknown", 16);
14184        let payload = vec![3_u8; 96];
14185        attach(
14186            &path,
14187            "items",
14188            &[section::Attachment {
14189                kind: *b"RUDBZZ9\0",
14190                id: 1,
14191                flags: 0,
14192                header_bytes: 0,
14193                bytes: &payload,
14194            }],
14195        )
14196        .expect("a kind this build does not know still writes");
14197        let key_map = a_key_map_payload();
14198        attach(
14199            &path,
14200            "items",
14201            &[section::Attachment {
14202                kind: *section::KEY_MAP,
14203                id: 0,
14204                flags: 0,
14205                header_bytes: 0,
14206                bytes: &key_map,
14207            }],
14208        )
14209        .expect("attach beside it");
14210
14211        let reader = Reader::open(&path).expect("reopen");
14212        let held = attached(reader.table());
14213        assert_eq!(held.len(), 2, "the unfamiliar entry survived a directory rewrite");
14214        let unknown = held.iter().find(|one| !one.known()).expect("the unfamiliar one");
14215        assert_eq!(reader.payload(unknown).expect("its bytes are still there"), payload);
14216
14217        fs::remove_file(&path).expect("clean up");
14218    }
14219
14220    #[test]
14221    fn a_payload_of_nothing_is_a_relationship_recorded_as_not_built() {
14222        let path = linked_file("attach_not_built", 8);
14223        attach(
14224            &path,
14225            "items",
14226            &[section::Attachment {
14227                kind: *section::FORWARD_LINK,
14228                id: 2,
14229                flags: 0,
14230                header_bytes: 0,
14231                bytes: &[],
14232            }],
14233        )
14234        .expect("record a link that did not fit the budget");
14235
14236        let reader = Reader::open(&path).expect("reopen");
14237        let held = attached(reader.table());
14238        assert_eq!(held.len(), 1);
14239        assert_eq!(held[0].extents, 0);
14240        assert_eq!(held[0].extent_page, 0, "an entry that names no bytes points at none");
14241        assert!(reader.extents(held[0]).expect("no extent table").is_empty());
14242        assert!(reader.payload(held[0]).expect("no payload").is_empty());
14243
14244        fs::remove_file(&path).expect("clean up");
14245    }
14246
14247    #[test]
14248    fn a_payload_past_one_extent_is_split_and_joined_back() {
14249        // Issue #745's rule, exercised rather than argued. One byte past the bound is the smallest
14250        // payload that has to be two extents, and it is the case a split written for the common
14251        // size gets wrong.
14252        let path = linked_file("attach_two_extents", 8);
14253        let payload = vec![0x5a_u8; section::MAX_EXTENT as usize + 1];
14254        attach(
14255            &path,
14256            "items",
14257            &[section::Attachment {
14258                kind: *section::KEY_MAP,
14259                id: 0,
14260                flags: 0,
14261                header_bytes: 0,
14262                bytes: &payload,
14263            }],
14264        )
14265        .expect("attach a payload past the bound");
14266
14267        let reader = Reader::open(&path).expect("reopen");
14268        let held = attached(reader.table());
14269        let extents = reader.extents(held[0]).expect("extent table");
14270        assert_eq!(extents.len(), 2, "one byte past the bound is two extents");
14271        assert_eq!(extents[0].length, section::MAX_EXTENT);
14272        assert_eq!(extents[1].length, 1);
14273        assert_eq!(extents[1].first, u64::from(section::MAX_EXTENT));
14274        // And the extent the caller wants is readable on its own, which is the point of the split.
14275        assert_eq!(reader.extent(&extents[1]).expect("the last extent"), vec![0x5a]);
14276        assert_eq!(reader.payload(held[0]).expect("the whole payload").len(), payload.len());
14277
14278        fs::remove_file(&path).expect("clean up");
14279    }
14280
14281    #[test]
14282    fn a_torn_extent_is_refused_rather_than_decoded() {
14283        let path = linked_file("attach_torn", 8);
14284        let payload = a_key_map_payload();
14285        attach(
14286            &path,
14287            "items",
14288            &[section::Attachment {
14289                kind: *section::KEY_MAP,
14290                id: 0,
14291                flags: 0,
14292                header_bytes: 0,
14293                bytes: &payload,
14294            }],
14295        )
14296        .expect("attach");
14297
14298        let reader = Reader::open(&path).expect("reopen");
14299        let extent = reader.extents(&reader.table().sections()[0]).expect("extent table")[0];
14300        let file = OpenOptions::new().write(true).open(&path).expect("reopen to corrupt");
14301        write_at(&file, extent.offset + 7, &[0xff]).expect("flip a byte of the payload");
14302        drop(file);
14303
14304        let reader = Reader::open(&path).expect("the table still opens");
14305        let error = reader
14306            .payload(&reader.table().sections()[0])
14307            .expect_err("a corrupt payload is not handed out");
14308        assert!(error.to_string().contains("checksum"), "{error}");
14309        // And the table is still readable, which is section 3.1: a section that cannot be trusted
14310        // costs the query its shortcut and nothing else.
14311        assert_eq!(reader.read(0, &[0]).expect("the column is untouched").width(), 1);
14312
14313        fs::remove_file(&path).expect("clean up");
14314    }
14315
14316    #[test]
14317    fn attaching_to_a_file_of_the_previous_format_is_refused_rather_than_done() {
14318        // Readable is not writable. A format 22 directory has no section block, and adding one
14319        // without moving the number in the header would leave a file claiming a format it is not.
14320        let path = linked_file("attach_old_format", 8);
14321        let file = OpenOptions::new().write(true).open(&path).expect("reopen to patch");
14322        write_at(&file, 8, &22_u32.to_le_bytes()).expect("stamp the older format");
14323        drop(file);
14324
14325        let payload = a_key_map_payload();
14326        let error = attach(
14327            &path,
14328            "items",
14329            &[section::Attachment {
14330                kind: *section::KEY_MAP,
14331                id: 0,
14332                flags: 0,
14333                header_bytes: 0,
14334                bytes: &payload,
14335            }],
14336        )
14337        .expect_err("format 22 cannot gain a section");
14338        assert!(error.to_string().contains("format 22"), "{error}");
14339        assert!(Reader::open(&path).expect("and the file is untouched").table().rows() == 8);
14340
14341        fs::remove_file(&path).expect("clean up");
14342    }
14343
14344    #[test]
14345    fn a_section_header_longer_than_its_payload_is_refused_at_the_write() {
14346        let path = linked_file("attach_bad_header", 8);
14347        let error = attach(
14348            &path,
14349            "items",
14350            &[section::Attachment {
14351                kind: *section::KEY_MAP,
14352                id: 0,
14353                flags: 0,
14354                header_bytes: 40,
14355                bytes: &[1, 2, 3],
14356            }],
14357        )
14358        .expect_err("a writer's bug stops at the write");
14359        assert!(error.to_string().contains("header is longer"), "{error}");
14360
14361        fs::remove_file(&path).expect("clean up");
14362    }
14363
14364    #[test]
14365    fn attaching_to_a_name_the_file_does_not_hold_says_so() {
14366        let path = linked_file("attach_wrong_name", 8);
14367        let error = attach(&path, "orders", &[]).expect_err("no such table");
14368        assert!(error.to_string().contains("orders"), "{error}");
14369        fs::remove_file(&path).expect("clean up");
14370    }
14371
14372    #[test]
14373    fn the_planner_gets_an_exact_count_for_a_leading_value_of_an_incomplete_synopsis() {
14374        // The case a complete synopsis does not cover, and the one worth the most. 16,000 rows over
14375        // 601 distinct values, 10,000 of them holding a single value and the rest spread ten apiece
14376        // over six hundred more. The writer holds 512 values, so the list is a prefix and most of
14377        // the tail is outside it. The counts inside it are still exact, because the pass recounts
14378        // the candidates that survived it, so `id = 1` is ten thousand rows rather than the
14379        // twenty six a distinct count of 601 would divide its way to.
14380        let path = path("frequency_prefix_for_the_planner");
14381        let mut writer =
14382            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
14383                .expect("new file");
14384        let mut values = vec![Value::Integer(1); 10_000];
14385        for _ in 0..10 {
14386            values.extend((0..600).map(|tail| Value::Integer(1_000 + tail)));
14387        }
14388        // A vector holds 8,192 rows, so this goes in as several parts. The pass that takes the
14389        // synopsis walks the whole column rather than a part, so the counts are the same either way.
14390        for part in values.chunks(8_000) {
14391            let rows = Chunk::new(vec![
14392                Vector::from_values(LogicalType::Integer, part).expect("integers"),
14393            ])
14394            .expect("one column");
14395            writer.append(&rows).expect("a part");
14396        }
14397        writer.finish().expect("commit");
14398        let reader = Reader::open(&path).expect("reopen from disk");
14399        let prefix =
14400            reader.frequency_prefix(0).expect("a readable synopsis").expect("the column has one");
14401        // A prefix and not the whole column, and the writer said how many rows anything left out of
14402        // it can hold.
14403        assert_eq!(prefix.entries.len(), 512);
14404        assert_eq!(prefix.omitted_max, 10);
14405        let common = Common::new(reader);
14406        assert_eq!(common.rows(), 16_000);
14407        let column = common.column("id").expect("the file has that column");
14408        assert_eq!(
14409            common.rows_with(column, &Bound::Int(1)),
14410            Stat::exact(10_000, Provenance::FrequencySynopsis)
14411        );
14412        // In the prefix, because ties go to the smaller value and the prefix reaches 1,510.
14413        assert_eq!(
14414            common.rows_with(column, &Bound::Int(1_100)),
14415            Stat::exact(10, Provenance::FrequencySynopsis)
14416        );
14417        // Outside it, and a prefix says nothing about a value it does not list. Not zero, which is
14418        // what a complete list would say, and the file holds ten rows of this one.
14419        assert_eq!(common.rows_with(column, &Bound::Int(1_550)), Stat::Unknown);
14420        // Not in the file at all, and still nothing rather than a zero. A prefix cannot tell the
14421        // two apart, which is the whole of what it gives up.
14422        assert_eq!(common.rows_with(column, &Bound::Int(9_999)), Stat::Unknown);
14423        // What the prefix left out, which is what turns the unknown above into a number. The 512
14424        // entries account for 15,110 rows, so 890 are left for the 89 values the writer dropped,
14425        // and 890 over 89 is the ten rows each of them really holds.
14426        let remainder = common.remainder(column).expect("the list is a prefix");
14427        assert_eq!(remainder, Remainder { rows: 890, listed: 512, most: 10 });
14428        assert_eq!(remainder.rows / (601 - remainder.listed), 10);
14429        fs::remove_file(&path).expect("clean up");
14430    }
14431
14432    /// A file with no table in it is a file, and opening it says so rather than failing.
14433    #[test]
14434    fn a_file_holding_no_table_commits_and_opens_and_a_table_can_be_added_to_it() {
14435        let path = path("empty");
14436        Writer::empty(&path, &[], None).expect("a file with nothing in it");
14437        let catalog = Catalog::open(&path).expect("the empty file opens");
14438        assert_eq!(catalog.len(), 0);
14439        assert!(catalog.is_empty());
14440        assert_eq!(catalog.names().count(), 0);
14441        // The next generation goes over the top of it the way it goes over any other, which is what
14442        // says this is a committed file and not a special case somebody has to know about.
14443        let mut writer =
14444            Writer::open(&path, "items", vec![Field::required("id", LogicalType::Integer)])
14445                .expect("a table goes into the empty file");
14446        writer.append(&sample_ids()).expect("rows");
14447        writer.finish().expect("commit");
14448        let catalog = Catalog::open(&path).expect("the file opens again");
14449        assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
14450        fs::remove_file(&path).expect("clean up");
14451    }
14452
14453    /// A committed table with no rows is a name the next generation takes over, and one with rows
14454    /// is a name it refuses.
14455    ///
14456    /// The refusal is what it always was and it is load bearing: carrying a table that holds rows
14457    /// forward means reading and rewriting its pages, and a writer that quietly wrote a second
14458    /// entry under the same name would leave a file with two tables a reader cannot tell apart. An
14459    /// empty one has no pages and no reader, so there is nothing to carry and nothing to lose, and
14460    /// taking its place is what lets a schema committed by an earlier session be loaded by a stream
14461    /// instead of through memory.
14462    #[test]
14463    fn a_committed_empty_table_gives_up_its_name_and_one_with_rows_does_not() {
14464        let path = path("empty-name");
14465        let field = || vec![Field::required("id", LogicalType::Integer)];
14466        Writer::create(&path, "items", field()).expect("new file").finish().expect("commit");
14467        let catalog = Catalog::open(&path).expect("the file opens");
14468        assert_eq!(catalog.rows().collect::<Vec<_>>(), vec![("items", 0)]);
14469
14470        let mut writer = Writer::open(&path, "items", field()).expect("the empty name is free");
14471        writer.append(&sample_ids()).expect("rows");
14472        writer.finish().expect("commit");
14473        let catalog = Catalog::open(&path).expect("the file opens again");
14474        // One entry and not two. The generation replaced the empty table rather than joining it.
14475        assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
14476        let held = catalog.rows().collect::<Vec<_>>();
14477        assert_eq!(held.len(), 1);
14478        assert!(held[0].1 > 0, "the rows that were appended are the ones the catalog counts");
14479
14480        // The same call against the same name now that it holds rows, which is still refused.
14481        let error = Writer::open(&path, "items", field()).expect_err("a name with rows is taken");
14482        assert!(error.to_string().contains("same name"), "{error}");
14483        fs::remove_file(&path).expect("clean up");
14484    }
14485
14486    #[test]
14487    fn a_log_anchor_rides_the_catalog_after_the_card_and_goes_forward_with_every_commit() {
14488        let entry = || Entry {
14489            name: "items".to_string(),
14490            fields: vec![Field::required("id", LogicalType::Integer)],
14491            rows: 1,
14492            directory: Page { offset: HEADER, length: 8, hash: 0 },
14493            nonzero: vec![None],
14494            aggregates: vec![None],
14495            distincts: vec![None],
14496            extremes: vec![None],
14497            frequencies: vec![None],
14498        };
14499        let anchor = LogAnchor {
14500            database: 0xfeed,
14501            durable: 41,
14502            lanes: vec![LaneStart { sequence: 3, offset: 4096 }],
14503            voids: vec![43, 47],
14504        };
14505        assert!(!anchor.replays(41) && anchor.replays(42) && !anchor.replays(43));
14506        let card = KeptCard { device: "dev:42".to_string(), bytes: vec![1, 2, 3] };
14507        for card in [None, Some(&card)] {
14508            let bytes = encode_catalog(&[entry()], &[], card, Some(&anchor)).expect("encodes");
14509            let (_, _, kept, held) = decode_catalog(&bytes, HEADER + 8).expect("decodes");
14510            assert_eq!((kept.as_ref(), held.as_ref()), (card, Some(&anchor)));
14511        }
14512        let mut twice = encode_catalog(&[entry()], &[], None, Some(&anchor)).expect("encodes");
14513        anchor.encode(&mut twice).expect("encodes");
14514        assert!(decode_catalog(&twice, HEADER + 8).is_err(), "a second anchor");
14515        let mut after = encode_catalog(&[entry()], &[], None, Some(&anchor)).expect("encodes");
14516        after.extend_from_slice(DEVICE_CARD);
14517        assert!(decode_catalog(&after, HEADER + 8).is_err(), "a card after the anchor");
14518        let under = LogAnchor { voids: vec![40], ..anchor.clone() };
14519        let bytes = encode_catalog(&[entry()], &[], None, Some(&under)).expect("encodes");
14520        assert!(decode_catalog(&bytes, HEADER + 8).is_err(), "a void under the cut");
14521
14522        let path = path("anchored");
14523        let mut writer =
14524            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
14525                .expect("new file");
14526        writer.append(&sample_ids()).expect("rows");
14527        writer.with_log_anchor(anchor.clone()).finish().expect("commit");
14528        assert_eq!(Catalog::open(&path).expect("reopen").log_anchor(), Some(&anchor));
14529        Writer::restate(&path, &[sample_view("v")], None).expect("a view");
14530        assert_eq!(Catalog::open(&path).expect("reopen").log_anchor(), Some(&anchor));
14531        let mut writer =
14532            Writer::open(&path, "other", vec![Field::required("id", LogicalType::Integer)])
14533                .expect("a second table");
14534        writer.append(&sample_ids()).expect("rows");
14535        writer.finish().expect("commit");
14536        assert_eq!(Catalog::open(&path).expect("reopen").log_anchor(), Some(&anchor));
14537        let next = LogAnchor { durable: 90, voids: Vec::new(), ..anchor };
14538        Writer::restate(&path, &[], Some(&next)).expect("a new cut");
14539        assert_eq!(Catalog::open(&path).expect("reopen").log_anchor(), Some(&next));
14540        fs::remove_file(&path).expect("clean up");
14541        let empty = self::path("anchoredempty");
14542        Writer::empty(&empty, &[], Some(&next)).expect("an empty file");
14543        assert_eq!(Catalog::open(&empty).expect("reopen").log_anchor(), Some(&next));
14544        fs::remove_file(&empty).expect("clean up");
14545    }
14546
14547    #[test]
14548    fn a_device_card_rides_the_catalog_and_an_older_catalog_has_none() {
14549        let entry = || Entry {
14550            name: "items".to_string(),
14551            fields: vec![Field::required("id", LogicalType::Integer)],
14552            rows: 1,
14553            directory: Page { offset: HEADER, length: 8, hash: 0 },
14554            nonzero: vec![None],
14555            aggregates: vec![None],
14556            distincts: vec![None],
14557            extremes: vec![None],
14558            frequencies: vec![None],
14559        };
14560        let card = KeptCard { device: "dev:42".to_string(), bytes: vec![1, 2, 3] };
14561        let bytes = encode_catalog(&[entry()], &[], Some(&card), None).expect("encodes");
14562        let (entries, views, kept, _) = decode_catalog(&bytes, HEADER + 8).expect("decodes");
14563        assert_eq!((entries.len(), views.len()), (1, 0));
14564        assert_eq!(kept, Some(card));
14565        let bytes = encode_catalog(&[entry()], &[], None, None).expect("encodes");
14566        assert_eq!(decode_catalog(&bytes, HEADER + 8).expect("decodes").2, None);
14567    }
14568
14569    /// A view, with everything about it that a reopened catalog has to be able to answer from.
14570    fn sample_view(name: &str) -> ViewEntry {
14571        ViewEntry {
14572            name: name.to_string(),
14573            sql: "SELECT id FROM items WHERE id > 0".to_string(),
14574            statement: format!("CREATE VIEW {name} AS SELECT id FROM items WHERE (id > 0);"),
14575            aliases: vec!["n".to_string()],
14576            columns: vec![Field::new("n", LogicalType::Integer)],
14577        }
14578    }
14579
14580    #[test]
14581    fn a_view_written_into_the_catalog_comes_back_whole() {
14582        let path = path("views");
14583        let mut writer =
14584            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
14585                .expect("new file");
14586        writer.append(&sample_ids()).expect("rows");
14587        writer.with_views(vec![sample_view("v")]).finish().expect("commit");
14588        let catalog = Catalog::open(&path).expect("reopen");
14589        assert_eq!(catalog.views().cloned().collect::<Vec<_>>(), vec![sample_view("v")]);
14590        // The tables are still there and are still read the same way, so the section on the end did
14591        // not move anything in front of it.
14592        assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
14593        fs::remove_file(&path).expect("clean up");
14594    }
14595
14596    /// A writer opened to append a table says nothing about views and must not lose them.
14597    #[test]
14598    fn appending_a_table_carries_the_views_forward() {
14599        let path = path("viewscarry");
14600        let mut writer =
14601            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
14602                .expect("new file");
14603        writer.append(&sample_ids()).expect("rows");
14604        writer.with_views(vec![sample_view("v")]).finish().expect("commit");
14605        let mut writer =
14606            Writer::open(&path, "other", vec![Field::required("id", LogicalType::Integer)])
14607                .expect("a second table");
14608        writer.append(&sample_ids()).expect("rows");
14609        writer.finish().expect("commit");
14610        let catalog = Catalog::open(&path).expect("reopen");
14611        assert_eq!(catalog.views().count(), 1);
14612        assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items", "other"]);
14613        fs::remove_file(&path).expect("clean up");
14614    }
14615
14616    /// The whole point of [`Writer::restate`]: the views change and the pages do not move.
14617    #[test]
14618    fn restating_the_views_leaves_every_table_where_it_was() {
14619        let path = path("restate");
14620        let mut writer =
14621            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
14622                .expect("new file");
14623        writer.append(&sample_ids()).expect("rows");
14624        writer.finish().expect("commit");
14625        let before = fs::metadata(&path).expect("the file is there").len();
14626        Writer::restate(&path, &[sample_view("v"), sample_view("w")], None).expect("two views");
14627        let catalog = Catalog::open(&path).expect("reopen");
14628        assert_eq!(catalog.views().count(), 2);
14629        assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
14630        // A catalog on the end and nothing else, so what it grew by is the size of a catalog rather
14631        // than the size of the table.
14632        let after = fs::metadata(&path).expect("the file is there").len();
14633        assert!(after > before, "a generation was written");
14634        assert!(after - before < before, "the table was not written again");
14635        // The rows are still readable through the new generation, which is the part that would go
14636        // wrong if the catalog carried the wrong directory pointers forward.
14637        let reader = Catalog::open(&path).expect("reopen").table("items").expect("the table");
14638        assert_eq!(reader.table().rows, 3);
14639        // And a restate over a restate keeps working, because each one reads the slot that
14640        // checksummed rather than the highest number in the header.
14641        Writer::restate(&path, &[], None).expect("no views at all");
14642        assert_eq!(Catalog::open(&path).expect("reopen").views().count(), 0);
14643        fs::remove_file(&path).expect("clean up");
14644    }
14645
14646    /// Two entries under one name is a catalog no lookup can answer, whichever two they are.
14647    #[test]
14648    fn a_view_named_after_a_table_is_refused_when_the_catalog_is_read() {
14649        let bytes = encode_catalog(
14650            &[Entry {
14651                name: "items".to_string(),
14652                fields: vec![Field::required("id", LogicalType::Integer)],
14653                rows: 1,
14654                directory: Page { offset: HEADER, length: 8, hash: 0 },
14655                nonzero: vec![None],
14656                aggregates: vec![None],
14657                distincts: vec![None],
14658                extremes: vec![None],
14659                frequencies: vec![None],
14660            }],
14661            &[sample_view("items")],
14662            None,
14663            None,
14664        )
14665        .expect("it encodes, because encoding does not look");
14666        let error = decode_catalog(&bytes, HEADER + 8).expect_err("and decoding does");
14667        assert!(error.to_string().contains("same name"), "{error}");
14668    }
14669
14670    /// A compressed text page read at some rows is those rows of the page read whole, nulls and
14671    /// all, and a row past the end or rows out of order are refused rather than guessed at.
14672    #[test]
14673    fn a_compressed_text_page_read_at_some_rows_is_those_rows_of_the_whole() {
14674        let rows: usize = 300;
14675        let text: Vec<String> =
14676            (0..rows).map(|row| format!("a street named after number {}", row * 7)).collect();
14677        let values: Vec<&[u8]> = text.iter().map(String::as_bytes).collect();
14678        let mut page = vec![6, 2];
14679        page.extend((0..rows.div_ceil(8)).map(|byte| {
14680            (0..8).filter(|bit| (byte * 8 + bit) % 5 != 3).fold(0_u8, |mask, bit| mask | 1 << bit)
14681        }));
14682        let compressed = string::encode_only(string::Kind::Fsst, &values)
14683            .expect("encoded")
14684            .expect("text this repetitive compresses");
14685        page.extend_from_slice(&compressed);
14686        let whole = decode(&LogicalType::Varchar, rows, &page, None).expect("the whole page");
14687        let positions = [0_u32, 3, 8, 13, 200, 299];
14688        let some =
14689            decode_at(&LogicalType::Varchar, rows, &page, None, &positions).expect("some rows");
14690        assert_eq!(some.len(), positions.len());
14691        for (at, &row) in positions.iter().enumerate() {
14692            assert_eq!(some.value_at(at), whole.value_at(row as usize), "row {row}");
14693        }
14694        assert_eq!(some.value_at(1), Value::Null, "row 3 is null");
14695        assert!(decode_at(&LogicalType::Varchar, rows, &page, None, &[300]).is_err());
14696        assert!(decode_at(&LogicalType::Varchar, rows, &page, None, &[8, 3]).is_err());
14697    }
14698
14699    /// Every column of a part read at some rows is the part read whole and gathered, whatever the
14700    /// page holds.
14701    #[test]
14702    fn a_part_read_at_some_rows_is_the_part_read_whole_and_gathered() {
14703        let path = path("rows");
14704        let mut writer = Writer::create(
14705            &path,
14706            "items",
14707            vec![
14708                Field::required("id", LogicalType::Integer),
14709                Field::new("text", LogicalType::Varchar),
14710            ],
14711        )
14712        .expect("new file");
14713        let rows = 2_000;
14714        let chunk = Chunk::new(vec![
14715            Vector::from_values(
14716                LogicalType::Integer,
14717                &(0..rows).map(Value::Integer).collect::<Vec<_>>(),
14718            )
14719            .expect("integers"),
14720            Vector::from_values(
14721                LogicalType::Varchar,
14722                &(0..rows)
14723                    .map(|row| {
14724                        if row % 7 == 2 {
14725                            Value::Null
14726                        } else {
14727                            Value::Varchar(format!("a comment about order {}", row * 13))
14728                        }
14729                    })
14730                    .collect::<Vec<_>>(),
14731            )
14732            .expect("strings"),
14733        ])
14734        .expect("matching rows");
14735        writer.append(&chunk).expect("one part");
14736        writer.finish().expect("commit");
14737        let reader = Reader::open(&path).expect("reopen from disk");
14738        let positions = [1_u32, 2, 9, 1_000, 1_999];
14739        for whole in [true, false] {
14740            let some = reader.read_rows(0, &[0, 1], &positions, whole).expect("some rows");
14741            let all = reader.read(0, &[0, 1]).expect("the whole part");
14742            assert_eq!(some.len(), positions.len());
14743            for column in 0..2 {
14744                for (at, &row) in positions.iter().enumerate() {
14745                    assert_eq!(some.value_at(at, column), all.value_at(row as usize, column));
14746                }
14747            }
14748        }
14749        assert!(reader.read_rows(0, &[1], &[2_000], true).is_err());
14750    }
14751
14752    #[test]
14753    fn committed_file_reopens_and_reads_only_requested_columns() {
14754        let path = path("reopen");
14755        let mut writer = Writer::create(
14756            &path,
14757            "items",
14758            vec![
14759                Field::required("id", LogicalType::Integer),
14760                Field::new("text", LogicalType::Varchar),
14761            ],
14762        )
14763        .expect("new file");
14764        writer.append(&sample()).expect("first part");
14765        writer.append(&sample()).expect("second part");
14766        writer.finish().expect("commit");
14767        let reader = Reader::open(&path).expect("reopen from disk");
14768        assert_eq!(reader.table().rows(), 6);
14769        // Two appends below the stripe bound are two parts of one stripe, which is the whole point
14770        // of the split: the directory describes the stripe and the scan still reads a part.
14771        assert_eq!(reader.table().stripes().len(), 1);
14772        assert_eq!(reader.parts(), 2);
14773        assert_eq!(reader.part_rows(0), 3);
14774        assert_eq!(reader.part_rows(1), 3);
14775        let text = reader.read(1, &[1]).expect("only text page");
14776        assert_eq!(text.width(), 1);
14777        assert_eq!(text.value_at(1, 0), Value::Null);
14778        assert_eq!(text.value_at(2, 0), Value::Varchar("long text after a slash".into()));
14779        let sparse = reader.read_sparse(1, &[1]).expect("one part without its whole page");
14780        assert_eq!(sparse.width(), 1);
14781        assert_eq!(sparse.value_at(1, 0), Value::Null);
14782        assert_eq!(sparse.value_at(2, 0), Value::Varchar("long text after a slash".into()));
14783        assert!(!reader.skips_codes(0, 1, &[0]).expect("alpha is in the stripe"));
14784        assert!(!reader.skips_codes(0, 1, &[2]).expect("long text is in the stripe"));
14785        assert!(reader.skips_codes(0, 1, &[3]).expect("unknown code is absent"));
14786        let count = reader.read(0, &[]).expect("no page is needed for count");
14787        assert_eq!(count.len(), 3);
14788        assert!(reader.skips(0, &[Probe { column: 0, op: Op::Greater, value: Bound::Int(100) }]));
14789        assert!(!reader.skips(0, &[Probe { column: 0, op: Op::Greater, value: Bound::Int(0) }]));
14790        assert_eq!(reader.top_frequencies(0, 1).expect("valid integer synopsis"), None);
14791        let integers = reader.frequency_prefix(0).expect("valid integer synopsis").expect("kept");
14792        assert_eq!(integers.entries, vec![(Value::Integer(-2), 2), (Value::Integer(4), 2)]);
14793        assert_eq!(integers.omitted_max, 2);
14794        let strings = reader.top_frequencies(1, 1).expect("valid string synopsis").expect("kept");
14795        assert_eq!(strings.len(), 3);
14796        assert!(strings.contains(&(Value::Null, 2)));
14797        assert!(strings.contains(&(Value::Varchar("alpha".into()), 2)));
14798        assert!(strings.contains(&(Value::Varchar("long text after a slash".into()), 2)));
14799        fs::remove_file(path).expect("remove scratch file");
14800    }
14801
14802    /// Two pipeline instances handing over whole runs, which is what makes the native sink safe to
14803    /// instance.
14804    ///
14805    /// The runs arrive in the order the instances finished reading them rather than in source
14806    /// order, and the second one to finish is the one that read the earlier rows. Each run is still
14807    /// a stripe of its own and the table still reads back in source order, which is the whole of
14808    /// what the writer promises about ordering.
14809    #[test]
14810    fn runs_handed_over_out_of_order_still_read_back_in_source_order() {
14811        let path = path("interleaved-runs");
14812        let mut writer =
14813            Writer::create(&path, "interleaved", vec![Field::new("v", LogicalType::BigInt)])
14814                .expect("new file");
14815        for morsel in [2_u64, 0, 3, 1] {
14816            let parts = (0..4_u64)
14817                .map(|chunk| {
14818                    let first = i64::try_from(morsel * 32 + chunk * 8).expect("small");
14819                    let values =
14820                        (0..8_i64).map(|row| Value::BigInt(first + row)).collect::<Vec<_>>();
14821                    let column =
14822                        Vector::from_values(LogicalType::BigInt, &values).expect("a column");
14823                    ((morsel, chunk), Chunk::new(vec![column]).expect("one column"))
14824                })
14825                .collect::<Vec<_>>();
14826            writer.append_stripe(parts).expect("a stripe");
14827        }
14828        writer.finish().expect("commit");
14829
14830        let reader = Reader::open(&path).expect("valid directory");
14831        assert_eq!(reader.table().stripes().len(), 4, "a run is a stripe of its own");
14832        assert_eq!(reader.table().rows(), 128);
14833        for part in 0..16_usize {
14834            let read = reader.read(part, &[0]).expect("a part back");
14835            for row in 0..8_usize {
14836                let want = i64::try_from(part * 8 + row).expect("small");
14837                assert_eq!(read.value_at(row, 0), Value::BigInt(want), "part {part} row {row}");
14838            }
14839        }
14840        fs::remove_file(path).expect("remove scratch file");
14841    }
14842
14843    /// Runs from different callers may interleave and may not overlap, and the commit is what
14844    /// catches an overlap.
14845    #[test]
14846    fn runs_that_overlap_each_other_are_refused_at_commit() {
14847        let path = path("overlapping-runs");
14848        let mut writer =
14849            Writer::create(&path, "overlapping", vec![Field::new("v", LogicalType::BigInt)])
14850                .expect("new file");
14851        let one = |order: (u64, u64)| {
14852            let column =
14853                Vector::from_values(LogicalType::BigInt, &[Value::BigInt(1)]).expect("a column");
14854            (order, Chunk::new(vec![column]).expect("one column"))
14855        };
14856        // The second run sits inside the first rather than after it, which is a thing no instance
14857        // holding its own contiguous run can produce and a thing the file cannot represent.
14858        writer.append_stripe(vec![one((0, 0)), one((0, 2))]).expect("a stripe");
14859        writer.append_stripe(vec![one((0, 1))]).expect("a stripe");
14860        let error = writer.finish().expect_err("the runs overlap");
14861        assert!(error.message().contains("source order"), "{error}");
14862        fs::remove_file(path).expect("remove scratch file");
14863    }
14864
14865    /// A stripe holds [`STRIPE_PARTS`] parts, so a run longer than that is a caller bug rather than
14866    /// something to split, and the writer says so at the door instead of quietly cutting it in two.
14867    #[test]
14868    fn a_run_longer_than_a_stripe_is_refused() {
14869        let path = path("overlong-run");
14870        let mut writer =
14871            Writer::create(&path, "overlong", vec![Field::new("v", LogicalType::BigInt)])
14872                .expect("new file");
14873        let parts = (0..=STRIPE_PARTS)
14874            .map(|at| {
14875                let column = Vector::from_values(LogicalType::BigInt, &[Value::BigInt(1)])
14876                    .expect("a column");
14877                let chunk = Chunk::new(vec![column]).expect("one column");
14878                ((0, u64::try_from(at).expect("small")), chunk)
14879            })
14880            .collect::<Vec<_>>();
14881        let error = writer.append_stripe(parts).expect_err("one part too many");
14882        assert!(error.message().contains("more parts than it holds"), "{error}");
14883        fs::remove_file(path).expect("remove scratch file");
14884    }
14885
14886    /// Parts past the stripe bound start a new stripe, and every part stays addressable on its own.
14887    ///
14888    /// This is the shape the format exists for, so both ends of the split are checked here. The
14889    /// directory holds three stripes rather than a hundred and thirty one, and a read of any one
14890    /// part still answers with that part's rows rather than with its whole stripe's.
14891    #[test]
14892    fn parts_past_the_stripe_bound_start_a_new_stripe() {
14893        let path = path("stripe-bound");
14894        let mut writer = Writer::create(
14895            &path,
14896            "items",
14897            vec![
14898                Field::required("id", LogicalType::Integer),
14899                Field::new("text", LogicalType::Varchar),
14900            ],
14901        )
14902        .expect("new file");
14903        let parts = STRIPE_PARTS * 2 + 3;
14904        for part in 0..parts {
14905            let id = part as i32;
14906            let chunk = Chunk::new(vec![
14907                Vector::from_values(
14908                    LogicalType::Integer,
14909                    &[Value::Integer(id), Value::Integer(-id)],
14910                )
14911                .expect("integers"),
14912                Vector::from_values(
14913                    LogicalType::Varchar,
14914                    &[Value::Varchar(format!("value {part}")), Value::Null],
14915                )
14916                .expect("strings"),
14917            ])
14918            .expect("matching rows");
14919            writer.append(&chunk).expect("one part");
14920        }
14921        writer.finish().expect("commit");
14922
14923        let reader = Reader::open(&path).expect("reopen from disk");
14924        assert_eq!(reader.parts(), parts);
14925        assert_eq!(reader.table().rows(), parts * 2);
14926        assert_eq!(reader.table().stripes().len(), parts.div_ceil(STRIPE_PARTS));
14927        assert_eq!(reader.table().stripes()[0].parts(), STRIPE_PARTS);
14928        assert_eq!(reader.table().stripes()[0].rows(), STRIPE_PARTS * 2);
14929        assert_eq!(reader.table().stripes()[2].parts(), 3);
14930        // Backwards on purpose. The reader keeps four stripes a column, so a scan that walks the
14931        // table the other way is what catches a cache that only ever holds what it just read.
14932        for part in (0..parts).rev() {
14933            let dense = reader.read(part, &[0, 1]).expect("a whole page read");
14934            let sparse = reader.read_sparse(part, &[0, 1]).expect("one part read");
14935            for chunk in [&dense, &sparse] {
14936                assert_eq!(chunk.len(), 2, "part {part} has its own row count");
14937                assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
14938                assert_eq!(chunk.value_at(1, 0), Value::Integer(-(part as i32)));
14939                assert_eq!(chunk.value_at(0, 1), Value::Varchar(format!("value {part}")));
14940                assert_eq!(chunk.value_at(1, 1), Value::Null);
14941            }
14942        }
14943        // The bounds are merged over the stripe, so they answer for the range the whole stripe
14944        // covers and not for the part that was asked about.
14945        let above = [Probe { column: 0, op: Op::Greater, value: Bound::Int(100) }];
14946        assert!(reader.skips(0, &above), "the first stripe stops at 63");
14947        assert!(!reader.skips(STRIPE_PARTS * 2, &above), "the third stripe reaches 130");
14948        fs::remove_file(path).expect("remove scratch file");
14949    }
14950
14951    /// A scattered value in the column that decides `WHERE UserID = ?`.
14952    fn scattered(n: i64) -> i64 {
14953        n.wrapping_mul(-7_046_029_254_386_353_131)
14954    }
14955
14956    /// A part whose sieve does not hold the constant is skipped, and a range would skip none of them.
14957    ///
14958    /// This is ClickBench query 19 in miniature. The values are spread over the whole of `BIGINT`, so
14959    /// every stripe's bounds cover nearly all of it and rule out nothing, and the part that really
14960    /// holds the value is the only one a scan has to read.
14961    #[test]
14962    fn a_part_is_skipped_when_its_sieve_does_not_hold_the_constant() {
14963        let path = path("sieve-skip");
14964        let mut writer =
14965            Writer::create(&path, "hits", vec![Field::required("id", LogicalType::BigInt)])
14966                .expect("new file");
14967        let parts = STRIPE_PARTS + 3;
14968        // Big enough that the filter is worth its bytes. A part of eight numbers packs to under a
14969        // hundred bytes and the smallest filter there is is sixty nine, so a filter over a part
14970        // that small costs about as much to read as the rows do and is no longer written.
14971        let per_part = 128;
14972        for part in 0..parts {
14973            let held: Vec<Value> = (0..per_part)
14974                .map(|row| Value::BigInt(scattered((part * per_part + row) as i64)))
14975                .collect();
14976            let chunk =
14977                Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
14978                    .expect("one column");
14979            writer.append(&chunk).expect("one part");
14980        }
14981        writer.finish().expect("commit");
14982
14983        let reader = Reader::open(&path).expect("reopen from disk");
14984        let probe = |value: i64| Probe {
14985            column: 0,
14986            op: Op::Equal,
14987            value: Bound::Int(i128::from(scattered(value))),
14988        };
14989        for wanted in [0_i64, (per_part + 1) as i64, (parts * per_part - 1) as i64] {
14990            let tests = [probe(wanted)];
14991            let kept: Vec<usize> = (0..parts).filter(|&part| !reader.skips(part, &tests)).collect();
14992            let home = wanted as usize / per_part;
14993            assert!(kept.contains(&home), "the part holding {wanted} is read");
14994            // A filter answers maybe, so a part it keeps need not hold the value. Sixty seven parts
14995            // of a hundred and twenty eight numbers each, at a dozen bits a value, is about one
14996            // stray part across the whole file and that is what this leaves room for.
14997            assert!(kept.len() <= 2, "{wanted} keeps {kept:?}, which is more than one stray part");
14998        }
14999        let absent = [probe((parts * per_part) as i64 + 1)];
15000        let kept = (0..parts).filter(|&part| !reader.skips(part, &absent)).count();
15001        assert!(kept <= 1, "{kept} parts of {parts} kept a value no part holds");
15002        // The same probes against the bounds alone, which is what this replaces. A column of
15003        // scattered numbers has a range per stripe that covers nearly the whole type.
15004        let tests = [probe(0)];
15005        assert!(
15006            reader.table().stripes().iter().all(|stripe| !stripe.zone.skips(&tests)),
15007            "the bounds rule out no stripe at all"
15008        );
15009        fs::remove_file(path).expect("remove scratch file");
15010    }
15011
15012    /// A part whose own bounds rule out an ordered comparison is skipped where the stripe's keep it.
15013    ///
15014    /// This is the shape of ClickBench 24. Each part covers a narrow stretch of the column and the
15015    /// stripe covers all sixty four of them at once, so a comparison that lands inside the stripe
15016    /// rules out none of it and rules out all but a few parts.
15017    #[test]
15018    fn a_part_is_skipped_when_its_own_bounds_rule_out_a_comparison_the_stripe_keeps() {
15019        let path = path("part-range-skip");
15020        let mut writer =
15021            Writer::create(&path, "hits", vec![Field::required("at", LogicalType::BigInt)])
15022                .expect("new file");
15023        let parts = STRIPE_PARTS + 3;
15024        let per_part = 128;
15025        for part in 0..parts {
15026            // Scattered inside the part's own band rather than a run, because a run of
15027            // consecutive numbers encodes to a stride of a few bytes and then the page of ranges
15028            // costs more than reading the column it indexes, which is the case the writer declines.
15029            let held: Vec<Value> = (0..per_part)
15030                .map(|row| {
15031                    Value::BigInt((part * 1_000) as i64 + (scattered(row as i64).rem_euclid(900)))
15032                })
15033                .collect();
15034            let chunk =
15035                Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
15036                    .expect("one column");
15037            writer.append(&chunk).expect("one part");
15038        }
15039        writer.finish().expect("commit");
15040
15041        let reader = Reader::open(&path).expect("reopen from disk");
15042        let under = [Probe { column: 0, op: Op::Less, value: Bound::Int(3_000) }];
15043        let kept: Vec<usize> = (0..parts).filter(|&part| !reader.skips(part, &under)).collect();
15044        assert_eq!(kept, vec![0, 1, 2], "only the three parts that start under three thousand");
15045        // The same question asked of the stripe alone, which is what this replaces.
15046        assert!(!reader.stripe_skips(0, &under), "the stripe reaches from zero and keeps itself");
15047        fs::remove_file(path).expect("remove scratch file");
15048    }
15049
15050    /// The other half of the same page. A part whose own bounds put every row of it inside the
15051    /// filter is waved through, so the comparison never runs on it, where the stripe's bounds reach
15052    /// across every part and can prove nothing.
15053    #[test]
15054    fn a_part_is_waved_through_when_its_own_bounds_pass_a_comparison_the_stripe_cannot() {
15055        let path = path("part-range-certain");
15056        let mut writer =
15057            Writer::create(&path, "hits", vec![Field::required("at", LogicalType::BigInt)])
15058                .expect("new file");
15059        let parts = STRIPE_PARTS + 3;
15060        let per_part = 128;
15061        for part in 0..parts {
15062            let held: Vec<Value> = (0..per_part)
15063                .map(|row| {
15064                    Value::BigInt((part * 1_000) as i64 + (scattered(row as i64).rem_euclid(900)))
15065                })
15066                .collect();
15067            let chunk =
15068                Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
15069                    .expect("one column");
15070            writer.append(&chunk).expect("one part");
15071        }
15072        writer.finish().expect("commit");
15073
15074        let reader = Reader::open(&path).expect("reopen from disk");
15075        let under = [Probe { column: 0, op: Op::Less, value: Bound::Int(3_000) }];
15076        let waved: Vec<usize> = (0..parts).filter(|&part| reader.certain(part, &under)).collect();
15077        assert_eq!(waved, vec![0, 1, 2], "the three parts that end under three thousand");
15078        // The first stripe reaches from zero to past sixty thousand, so it straddles three thousand
15079        // and settles nothing either way. The three yeses above are the parts' own ends talking.
15080        assert!(!reader.stripe_skips(0, &under), "the stripe straddles the comparison");
15081        fs::remove_file(path).expect("remove scratch file");
15082    }
15083
15084    /// The page is worth its bytes on a column with parts to tell apart and is not written on one
15085    /// that has a single part, where the stripe bounds already are the part's.
15086    #[test]
15087    fn a_stripe_of_one_part_writes_no_range_page_and_a_stripe_of_many_does() {
15088        for (parts, wanted) in [(1_usize, false), (STRIPE_PARTS, true)] {
15089            let path = path("part-range-page");
15090            let mut writer =
15091                Writer::create(&path, "hits", vec![Field::required("at", LogicalType::BigInt)])
15092                    .expect("new file");
15093            for part in 0..parts {
15094                let held: Vec<Value> = (0..128)
15095                    .map(|row| {
15096                        Value::BigInt((part * 1_000) as i64 + scattered(row as i64).rem_euclid(900))
15097                    })
15098                    .collect();
15099                let chunk = Chunk::new(vec![
15100                    Vector::from_values(LogicalType::BigInt, &held).expect("numbers"),
15101                ])
15102                .expect("one column");
15103                writer.append(&chunk).expect("one part");
15104            }
15105            writer.finish().expect("commit");
15106            let reader = Reader::open(&path).expect("reopen from disk");
15107            let bytes = reader.layout().columns[0].part_ranges;
15108            assert_eq!(bytes > 0, wanted, "{parts} parts wrote {bytes} bytes of ranges");
15109            fs::remove_file(path).expect("remove scratch file");
15110        }
15111    }
15112
15113    /// A cut down string end is still an end on the side it was, which is the only thing that keeps
15114    /// a shortened bound from turning a skip into a wrong answer.
15115    #[test]
15116    fn a_string_end_that_is_cut_down_still_covers_the_value_it_came_from() {
15117        let long = vec![b'a'; PART_BOUND_BYTES * 2];
15118        let low = shortened(Some(Bound::Bytes(long.clone())), false).expect("a low end");
15119        let high = shortened(Some(Bound::Bytes(long.clone())), true).expect("a high end");
15120        let Bound::Bytes(low) = low else { panic!("a string stays a string") };
15121        let Bound::Bytes(high) = high else { panic!("a string stays a string") };
15122        assert!(low.len() <= PART_BOUND_BYTES && high.len() <= PART_BOUND_BYTES);
15123        assert!(low.as_slice() <= long.as_slice(), "the low end is at or under the value");
15124        assert!(high.as_slice() >= long.as_slice(), "the high end is at or over the value");
15125    }
15126
15127    /// A string of nothing but the largest byte has no prefix that can be stepped up, so the high
15128    /// end is given up rather than claimed too small. No end keeps the part, which is always safe.
15129    #[test]
15130    fn a_string_end_with_no_room_to_step_up_gives_up_the_bound() {
15131        let long = vec![u8::MAX; PART_BOUND_BYTES * 2];
15132        assert_eq!(shortened(Some(Bound::Bytes(long.clone())), true), None);
15133        let low = shortened(Some(Bound::Bytes(long)), false).expect("a low end is still a prefix");
15134        assert_eq!(low, Bound::Bytes(vec![u8::MAX; PART_BOUND_BYTES]));
15135    }
15136
15137    /// What a column is stored as, asked of two files holding the same rows in a different order.
15138    ///
15139    /// This is the question the report exists to answer and it is the one the directory cannot. The
15140    /// two files have the same rows, the same schema and the same number of parts, and the column
15141    /// comes out four times smaller in one of them, because ascending keys delta encode to a few
15142    /// bits a row and shuffled ones do not. Nothing about the file's shape says so. The page header
15143    /// says so, and reading it is what this does.
15144    ///
15145    /// It is q18 on TPC-H in miniature: clustering lineitem by ship date leaves `l_orderkey`
15146    /// ascending inside a partition but sparse, its deltas go from six bits to twelve, and the scan
15147    /// pays for the wider ones.
15148    #[test]
15149    fn what_a_column_is_stored_as_follows_the_order_the_rows_were_written_in() {
15150        let parts = 4;
15151        let per_part = 1024;
15152        let rows = parts * per_part;
15153        let written = |name: &str, keys: &[i64]| {
15154            let path = path(name);
15155            let fields = vec![Field::required("key", LogicalType::BigInt)];
15156            let mut writer = Writer::create(&path, "keys", fields).expect("new file");
15157            for part in 0..parts {
15158                let values: Vec<Value> = keys[part * per_part..(part + 1) * per_part]
15159                    .iter()
15160                    .map(|key| Value::BigInt(*key))
15161                    .collect();
15162                let chunk = Chunk::new(vec![
15163                    Vector::from_values(LogicalType::BigInt, &values).expect("numbers"),
15164                ])
15165                .expect("one column");
15166                writer.append(&chunk).expect("one part");
15167            }
15168            writer.finish().expect("commit");
15169            path
15170        };
15171        // Ascending with a small irregular step, which is what a key column in arrival order looks
15172        // like: an order has one to seven line items, so the key repeats and then moves on by one.
15173        let climbing = |step: &dyn Fn(usize) -> i64| {
15174            let mut key = 0;
15175            (0..rows)
15176                .map(|row| {
15177                    key += step(row);
15178                    key
15179                })
15180                .collect::<Vec<i64>>()
15181        };
15182        let ascending = climbing(&|row| (row % 3) as i64);
15183        // The same rows in the same direction over a range a thousand times wider, which is what a
15184        // partition of a clustered table holds: still ascending, and far enough apart that the
15185        // deltas no longer fit in a handful of bits.
15186        let sparse = climbing(&|row| ((row * 2_654_435_761) % 4096) as i64);
15187        let near_path = written("stored-near", &ascending);
15188        let far_path = written("stored-far", &sparse);
15189
15190        let one = Reader::open(&near_path).expect("reopen from disk");
15191        let other = Reader::open(&far_path).expect("reopen from disk");
15192        let near = one.stored(0).expect("the column is stored");
15193        let far = other.stored(0).expect("the column is stored");
15194        assert_eq!(near.len(), parts, "one row per part");
15195        assert_eq!(far.len(), parts);
15196        // The bytes are the same bytes the directory totals, which is the check that this is
15197        // reading the pages the file really holds rather than some other pages.
15198        let total = |stored: &[StoredPart]| stored.iter().map(|part| part.bytes).sum::<u64>();
15199        assert_eq!(total(&near), one.layout().columns[0].pages);
15200        assert_eq!(total(&far), other.layout().columns[0].pages);
15201        assert!(
15202            total(&near) * 2 < total(&far),
15203            "the sparse keys cost more, {} against {}",
15204            total(&far),
15205            total(&near)
15206        );
15207        // Every part accounted for, in order, with the row it starts at following the one before.
15208        for (at, part) in near.iter().enumerate() {
15209            assert_eq!(part.part, at);
15210            assert_eq!(part.row, at * per_part);
15211            assert_eq!(part.rows, per_part);
15212            let held = &ascending[at * per_part..(at + 1) * per_part];
15213            assert_eq!(part.low, Some(Value::BigInt(held[0])));
15214            assert_eq!(part.high, Some(Value::BigInt(held[per_part - 1])));
15215            assert_eq!(part.nulls, Some(0));
15216        }
15217        // And the encoding is a line of text that names what the encoder chose, which is the whole
15218        // point. Both are a cascade over deltas and the widths inside them are what differ.
15219        assert!(near[0].encoding.contains("DELTA"), "{}", near[0].encoding);
15220        assert!(far[0].encoding.contains("DELTA"), "{}", far[0].encoding);
15221        assert_ne!(near[0].encoding, far[0].encoding);
15222        fs::remove_file(near_path).expect("remove scratch file");
15223        fs::remove_file(far_path).expect("remove scratch file");
15224    }
15225
15226    /// A sieve bigger than the part it indexes is not written, and one smaller than it still is.
15227    ///
15228    /// Both columns hold values spread over the whole of `BIGINT`, so neither gets a bitmap and both
15229    /// reach the filter. They differ in what the part costs to read. `spread` is a thousand distinct
15230    /// numbers and packs to eight kilobytes, so a filter of about thirteen hundred bytes is a good
15231    /// trade. `repeated` is the same thousand rows over four numbers in runs and encodes to
15232    /// almost nothing, but the filter is sized for the rows rather than the values it turns out to
15233    /// hold, so it comes out larger than the data. Reading it to decide whether to read the part spends more than
15234    /// the part, every time, and that is the case this drops.
15235    #[test]
15236    fn a_sieve_larger_than_the_part_it_indexes_is_not_written() {
15237        let path = path("sieve-pays");
15238        let fields = vec![
15239            Field::required("spread", LogicalType::BigInt),
15240            Field::required("repeated", LogicalType::BigInt),
15241        ];
15242        let mut writer = Writer::create(&path, "hits", fields).expect("new file");
15243        let parts = 3;
15244        let per_part = 1024;
15245        for part in 0..parts {
15246            let base = (part * per_part) as i64;
15247            let spread: Vec<Value> =
15248                (0..per_part).map(|row| Value::BigInt(scattered(base + row as i64))).collect();
15249            let repeated: Vec<Value> =
15250                (0..per_part).map(|row| Value::BigInt(scattered((row / 256) as i64))).collect();
15251            let chunk = Chunk::new(vec![
15252                Vector::from_values(LogicalType::BigInt, &spread).expect("numbers"),
15253                Vector::from_values(LogicalType::BigInt, &repeated).expect("numbers"),
15254            ])
15255            .expect("two columns");
15256            writer.append(&chunk).expect("one part");
15257        }
15258        writer.finish().expect("commit");
15259
15260        let reader = Reader::open(&path).expect("reopen from disk");
15261        let layout = reader.layout();
15262        let spread = &layout.columns[0];
15263        let repeated = &layout.columns[1];
15264        assert!(spread.sieves > 0, "a column whose parts are worth a filter keeps one");
15265        assert_eq!(
15266            repeated.sieves, 0,
15267            "a column whose filter costs more than its parts keeps none"
15268        );
15269        // Per part this is the rule itself, so it holds over the column as well: a part without a
15270        // sieve adds to one side of this and to nothing on the other.
15271        for column in &layout.columns {
15272            assert!(
15273                column.sieves < column.pages,
15274                "{} spends {} on sieves over {} of data",
15275                column.name,
15276                column.sieves,
15277                column.pages
15278            );
15279        }
15280        // The filter that was kept still does what it is for.
15281        let absent = [Probe {
15282            column: 0,
15283            op: Op::Equal,
15284            value: Bound::Int(i128::from(scattered((parts * per_part) as i64 + 1))),
15285        }];
15286        assert!((0..parts).all(|part| reader.skips(part, &absent)), "no part holds it");
15287        fs::remove_file(path).expect("remove scratch file");
15288    }
15289
15290    /// A damaged sieve page is a part that gets read, not a query that fails.
15291    ///
15292    /// A sieve is an index over rows that are still there and still correct, so losing one costs
15293    /// time and costs no answers. That is the opposite of the membership index beside it, which is
15294    /// the only thing standing between a string page and a wrong answer.
15295    #[test]
15296    fn a_damaged_sieve_page_is_read_through_rather_than_refused() {
15297        let path = path("sieve-damaged");
15298        let mut writer =
15299            Writer::create(&path, "hits", vec![Field::required("id", LogicalType::BigInt)])
15300                .expect("new file");
15301        let rows = 128;
15302        let held: Vec<Value> = (0..rows).map(|row| Value::BigInt(scattered(row))).collect();
15303        let chunk =
15304            Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
15305                .expect("one column");
15306        writer.append(&chunk).expect("one part");
15307        writer.finish().expect("commit");
15308
15309        let page = Reader::open(&path).expect("reopen").table.stripes[0]
15310            .sieves
15311            .get(0)
15312            .expect("a sieve page");
15313        let mut file = OpenOptions::new().write(true).open(&path).expect("open the sieve page");
15314        file.seek(SeekFrom::Start(page.offset + u64::from(page.length) - 1)).expect("seek");
15315        file.write_all(&[0xff]).expect("damage one byte");
15316        drop(file);
15317
15318        let reader = Reader::open(&path).expect("reopen the damaged file");
15319        let absent =
15320            [Probe { column: 0, op: Op::Equal, value: Bound::Int(i128::from(scattered(99))) }];
15321        assert!(!reader.skips(0, &absent), "a sieve that cannot be read skips nothing");
15322        assert_eq!(
15323            reader.read(0, &[0]).expect("the rows are untouched").len(),
15324            usize::try_from(rows).expect("a small count")
15325        );
15326        fs::remove_file(path).expect("remove scratch file");
15327    }
15328
15329    /// A scan that asks for each part twice reads each page whole once and keeps only the floor.
15330    ///
15331    /// This is ClickBench 21's shape: a `LIKE` asks a compressed text part whether it can answer and
15332    /// then reads the part. Counting parts rather than asks is what stops the second ask of every
15333    /// part from looking like a second scan, which would pool every page of the column.
15334    #[test]
15335    fn a_part_asked_for_twice_in_one_scan_keeps_its_page_only_to_the_floor() {
15336        let path = path("asked-twice");
15337        let parts = STRIPE_PARTS * (CACHED_STRIPES_PER_COLUMN + 2);
15338        let mut writer =
15339            Writer::create(&path, "a", vec![Field::required("id", LogicalType::Integer)])
15340                .expect("new file");
15341        for part in 0..parts {
15342            let chunk = Chunk::new(vec![
15343                Vector::from_values(LogicalType::Integer, &[Value::Integer(part as i32)])
15344                    .expect("integers"),
15345            ])
15346            .expect("matching rows");
15347            writer.append(&chunk).expect("one part");
15348        }
15349        writer.finish().expect("commit");
15350
15351        let pool = PagePool::new(usize::MAX);
15352        let catalog = Catalog::open_in(&path, &pool).expect("the file opens");
15353        let a = catalog.table("a").expect("a");
15354        let stripes = a.table().stripes().len();
15355        for part in 0..parts {
15356            for _ in 0..2 {
15357                let chunk = a.read(part, &[0]).expect("a part");
15358                assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
15359            }
15360        }
15361        assert_eq!(
15362            a.pages.load(Atomic::Relaxed),
15363            stripes,
15364            "a page a stripe, read on the second ask"
15365        );
15366        assert_eq!(pool.bytes(), 0, "one scan puts nothing in the pool");
15367        let column = a.cache.columns[0].lock().expect("the column");
15368        assert_eq!(column.pages.iter().flatten().count(), CACHED_STRIPES_PER_COLUMN);
15369        drop(column);
15370        drop((a, catalog));
15371        fs::remove_file(path).expect("remove scratch file");
15372    }
15373
15374    /// Eight workers over one stripe read it once between them.
15375    ///
15376    /// This is the shape a scan actually has. Parts are handed out in order, so every worker on a
15377    /// column crosses into a stripe within a few parts of the others, and before [`Reader::held`]
15378    /// started sharing the read every one of them read the whole page. On the full ClickBench file
15379    /// that was a `MIN(EventDate), MAX(EventDate)` moving 3.2 GB off the disk to look at 400 MB of
15380    /// column, which is most of what a first touch costs.
15381    ///
15382    /// The workers that lose the race still answer, out of the part reads they do instead, which is
15383    /// what the values below are checking.
15384    #[test]
15385    fn workers_that_want_the_same_stripe_read_it_once() {
15386        let path = path("single-flight");
15387        let mut writer =
15388            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
15389                .expect("new file");
15390        for part in 0..STRIPE_PARTS {
15391            let id = part as i32;
15392            let chunk = Chunk::new(vec![
15393                Vector::from_values(
15394                    LogicalType::Integer,
15395                    &[Value::Integer(id), Value::Integer(-id)],
15396                )
15397                .expect("integers"),
15398            ])
15399            .expect("matching rows");
15400            writer.append(&chunk).expect("one part");
15401        }
15402        writer.finish().expect("commit");
15403
15404        let reader = Reader::open(&path).expect("reopen from disk");
15405        assert_eq!(reader.table().stripes().len(), 1, "one stripe is the point of the test");
15406        // Once through a part at a time first, since a stripe's page is only read whole the second
15407        // time a scan comes to it.
15408        for part in 0..STRIPE_PARTS {
15409            reader.read(part, &[0]).expect("a part");
15410        }
15411        assert_eq!(reader.pages.load(Atomic::Relaxed), 0, "the first pass reads no page whole");
15412        let barrier = std::sync::Barrier::new(8);
15413        std::thread::scope(|scope| {
15414            for worker in 0..8 {
15415                let reader = &reader;
15416                let barrier = &barrier;
15417                scope.spawn(move || {
15418                    barrier.wait();
15419                    for part in (worker..STRIPE_PARTS).step_by(8) {
15420                        let chunk = reader.read(part, &[0]).expect("a whole page read");
15421                        assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
15422                        assert_eq!(chunk.value_at(1, 0), Value::Integer(-(part as i32)));
15423                    }
15424                });
15425            }
15426        });
15427        assert_eq!(reader.pages.load(Atomic::Relaxed), 1, "one stripe, one page read, whoever won");
15428        fs::remove_file(path).expect("remove scratch file");
15429    }
15430
15431    /// Opening a file reads the header and the directory, and nothing that depends on the rows.
15432    ///
15433    /// `spec/stats/04-in-memory.md` section 4.2. There are no statistics in the file yet, so this
15434    /// holds today by not having anything to load, and that is exactly why it is worth pinning now.
15435    /// The change that breaks it is the reasonable looking one: summaries are a few hundred bytes,
15436    /// the next query will want them, so read them on the way past. A process that opened the
15437    /// database to run one trivial query pays for all of it and gets nothing.
15438    ///
15439    /// Two files of the same shape and a thousand times the rows in one of them, opened, and the
15440    /// two openings cost the same. The stripe count is held equal so that the directory is the same
15441    /// size in both, which leaves the rows as the only thing that changed. Anything read out of the
15442    /// data would show up here.
15443    #[test]
15444    fn opening_costs_the_same_over_a_thousand_times_the_rows() {
15445        let opened = |label: &str, rows_per_part: i32| {
15446            let path = path(label);
15447            let mut writer =
15448                Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
15449                    .expect("new file");
15450            for part in 0..STRIPE_PARTS * 3 {
15451                // Scrambled rather than sequential, so that the fat file is actually fatter. A run
15452                // of consecutive integers encodes to almost nothing and would leave the two files
15453                // the same size, which would make this test pass for the wrong reason.
15454                let values = (0..rows_per_part)
15455                    .map(|row| {
15456                        Value::Integer((part as i32 * rows_per_part + row).wrapping_mul(2_654_435))
15457                    })
15458                    .collect::<Vec<_>>();
15459                let chunk = Chunk::new(vec![
15460                    Vector::from_values(LogicalType::Integer, &values).expect("integers"),
15461                ])
15462                .expect("matching rows");
15463                writer.append(&chunk).expect("one part");
15464            }
15465            writer.finish().expect("commit");
15466            let reader = Reader::open(&path).expect("reopen from disk");
15467            let size = fs::metadata(&path).expect("the file is there").len();
15468            let out = (reader.reads(), reader.table().stripes().len(), size);
15469            fs::remove_file(path).expect("remove scratch file");
15470            out
15471        };
15472
15473        let (thin, thin_stripes, thin_size) = opened("open-thin", 1);
15474        let (fat, fat_stripes, fat_size) = opened("open-fat", 1000);
15475        assert_eq!(
15476            thin_stripes, fat_stripes,
15477            "the same stripe count is what makes this a fair ask"
15478        );
15479        assert!(
15480            fat_size > thin_size * 50,
15481            "the fat file has to actually be larger, and it is {fat_size} against {thin_size}"
15482        );
15483
15484        assert_eq!(thin.opening.reads, fat.opening.reads, "the same reads either way");
15485        assert_eq!(thin.pages, 0, "opening read a page");
15486        assert_eq!(fat.pages, 0, "opening read a page");
15487        assert_eq!(thin.indexes, 0, "opening read an index");
15488        assert_eq!(fat.indexes, 0, "opening read an index");
15489        // Not exactly equal, because a directory holds offsets and a larger file has larger ones,
15490        // and a handful of bytes of varint is not somebody loading statistics. A factor is.
15491        assert!(
15492            fat.opening.bytes < thin.opening.bytes * 2,
15493            "opening the thin file read {} bytes and the fat one read {}",
15494            thin.opening.bytes,
15495            fat.opening.bytes
15496        );
15497    }
15498
15499    /// The reads a file costs to open are fixed by its shape and not by what ran before.
15500    ///
15501    /// `spec/stats/04-in-memory.md` section 4.3, which is the rule that keeps a plan reproducible:
15502    /// the plan is a function of the data, the generation and the settings, and never of what
15503    /// happened to be in cache. Opening the same file twice in the same process has to cost the
15504    /// same, because a second open that read less would be an open that was about to plan
15505    /// differently.
15506    #[test]
15507    fn two_opens_of_one_file_cost_the_same_and_the_second_is_not_cheaper() {
15508        let path = path("open-twice");
15509        let mut writer =
15510            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
15511                .expect("new file");
15512        for part in 0..STRIPE_PARTS * 3 {
15513            let chunk = Chunk::new(vec![
15514                Vector::from_values(LogicalType::Integer, &[Value::Integer(part as i32)])
15515                    .expect("integers"),
15516            ])
15517            .expect("matching rows");
15518            writer.append(&chunk).expect("one part");
15519        }
15520        writer.finish().expect("commit");
15521
15522        let first = Reader::open(&path).expect("open");
15523        // A whole scan in between, so the operating system's page cache is as warm as it gets and
15524        // anything that consulted it would show up in the second open.
15525        for part in 0..first.parts() {
15526            first.read(part, &[0]).expect("a part");
15527        }
15528        assert!(first.reads().indexes > 0, "the scan has to have read something");
15529        let second = Reader::open(&path).expect("open again");
15530
15531        assert_eq!(first.reads().opening, second.reads().opening);
15532        assert_eq!(
15533            second.reads().pages,
15534            0,
15535            "the second open read a page off the back of the first"
15536        );
15537        assert_eq!(second.reads().indexes, 0, "the second open read an index it inherited");
15538        fs::remove_file(path).expect("remove scratch file");
15539    }
15540
15541    /// A scan reads a stripe's index once for the whole scan, not once per part that misses.
15542    ///
15543    /// The page cache holds four stripes and an index used to ride inside it, so a table with more
15544    /// stripes than that read the index again every time a stripe came back around. The index is a
15545    /// few hundred bytes and the page is a quarter of a megabyte, which is why they are now under
15546    /// different budgets. This is the test that keeps them there, since the saving is small enough
15547    /// that nothing in a benchmark would notice it going away again.
15548    #[test]
15549    fn an_index_is_read_once_per_stripe_however_often_the_page_is_evicted() {
15550        let path = path("index-cache");
15551        let mut writer =
15552            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
15553                .expect("new file");
15554        let parts = STRIPE_PARTS * (CACHED_STRIPES_PER_COLUMN + 2);
15555        for part in 0..parts {
15556            let id = part as i32;
15557            let chunk = Chunk::new(vec![
15558                Vector::from_values(LogicalType::Integer, &[Value::Integer(id)]).expect("integers"),
15559            ])
15560            .expect("matching rows");
15561            writer.append(&chunk).expect("one part");
15562        }
15563        writer.finish().expect("commit");
15564
15565        let reader = Reader::open(&path).expect("reopen from disk");
15566        let stripes = reader.table().stripes().len();
15567        assert!(stripes > CACHED_STRIPES_PER_COLUMN, "the page cache has to be too small for this");
15568        // Three times over. The first pass reads a part at a time, the second reads the pages, and
15569        // the third finds every page evicted and every index kept.
15570        for _ in 0..3 {
15571            for part in 0..parts {
15572                let chunk = reader.read(part, &[0]).expect("a part");
15573                assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
15574            }
15575        }
15576        assert_eq!(reader.indexes.load(Atomic::Relaxed), stripes, "one index read per stripe");
15577        assert!(
15578            reader.pages.load(Atomic::Relaxed) > stripes,
15579            "the pages are the ones that get read again, which is what makes the index count mean \
15580             something"
15581        );
15582        fs::remove_file(path).expect("remove scratch file");
15583    }
15584
15585    /// A page stays in memory from one scan to the next while the pool has room for it, and a
15586    /// table that is being read takes room from one that is not, down to the floor and no further.
15587    ///
15588    /// This is what the pool is for. Each reader lives as long as its database, so a second query
15589    /// over the same table should find every page it read the first time, and before the pool it
15590    /// found four stripes a column and read the rest off the file again.
15591    #[test]
15592    fn a_pool_keeps_pages_between_scans_and_gives_them_to_the_table_being_read() {
15593        let path = path("page-pool");
15594        let parts = STRIPE_PARTS * (CACHED_STRIPES_PER_COLUMN * 2 + 2);
15595        let fields = || vec![Field::required("id", LogicalType::Integer)];
15596        let mut writer = Writer::create(&path, "a", fields()).expect("new file");
15597        for table in ["a", "b"] {
15598            if table == "b" {
15599                writer = writer.next("b".to_string(), fields()).expect("a second table");
15600            }
15601            for part in 0..parts {
15602                let chunk = Chunk::new(vec![
15603                    Vector::from_values(LogicalType::Integer, &[Value::Integer(part as i32)])
15604                        .expect("integers"),
15605                ])
15606                .expect("matching rows");
15607                writer.append(&chunk).expect("one part");
15608            }
15609        }
15610        writer.finish().expect("commit");
15611
15612        let pool = PagePool::new(usize::MAX);
15613        let catalog = Catalog::open_in(&path, &pool).expect("the file opens");
15614        let (a, b) = (catalog.table("a").expect("a"), catalog.table("b").expect("b"));
15615        let stripes = a.table().stripes().len();
15616        assert!(
15617            stripes > CACHED_STRIPES_PER_COLUMN * 2,
15618            "the floor has to be smaller than a table"
15619        );
15620        let scan = |reader: &Reader| {
15621            for part in 0..parts {
15622                let chunk = reader.read(part, &[0]).expect("a part");
15623                assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
15624            }
15625        };
15626        // The first scan reads a part at a time and keeps no page, the second reads every page and
15627        // keeps it, and the third reads nothing.
15628        scan(&a);
15629        assert_eq!(a.pages.load(Atomic::Relaxed), 0, "the first scan reads no page whole");
15630        assert_eq!(pool.bytes(), 0, "a stripe read once is not the pool's");
15631        scan(&a);
15632        assert_eq!(a.pages.load(Atomic::Relaxed), stripes, "the second scan reads every page");
15633        scan(&a);
15634        assert_eq!(a.pages.load(Atomic::Relaxed), stripes, "the third scan reads nothing");
15635        let one = pool.bytes();
15636        assert!(one > 0, "the pool counts what the reader holds");
15637
15638        // Room for one table. Reading the other takes the first one's pages down to its floor.
15639        pool.budget.store(one, Atomic::Relaxed);
15640        scan(&b);
15641        scan(&b);
15642        assert_eq!(b.pages.load(Atomic::Relaxed), stripes, "a page is never let go while in use");
15643        assert_eq!(a.cache.held[0].load(Atomic::Relaxed), CACHED_STRIPES_PER_COLUMN);
15644        let column = a.cache.columns[0].lock().expect("the column");
15645        let held = column.pages.iter().flatten().count();
15646        assert_eq!(held, CACHED_STRIPES_PER_COLUMN, "the count and the slots agree");
15647        drop(column);
15648
15649        // A reader that goes takes its pages out of the count with it.
15650        drop((a, b, catalog));
15651        let c = Catalog::open_in(&path, &pool).expect("again").table("a").expect("a");
15652        scan(&c);
15653        scan(&c);
15654        assert!(pool.bytes() <= one, "only what the live reader holds is counted");
15655        fs::remove_file(path).expect("remove scratch file");
15656    }
15657
15658    /// A worker per stripe reads its stripe once, once the cache has been told how many there are.
15659    ///
15660    /// This is the shape a scan has when it hands out a whole stripe per morsel rather than a part.
15661    /// Nobody races for a page any more, but every worker holds a different one for the length of a
15662    /// stripe, so a cache that keeps four pages while eight workers are in eight stripes evicts
15663    /// every one of them before its owner has finished with it, and the owner reads a quarter of a
15664    /// megabyte again for the next part. The barrier is what makes that certain rather than likely:
15665    /// without it a worker can run a whole stripe before the next one starts and never collide.
15666    #[test]
15667    fn a_worker_per_stripe_reads_its_page_once_when_the_cache_was_told_to_expect_it() {
15668        let workers = CACHED_STRIPES_PER_COLUMN + 4;
15669        let path = path("stripe-per-worker");
15670        let mut writer =
15671            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
15672                .expect("new file");
15673        for part in 0..STRIPE_PARTS * workers {
15674            let chunk = Chunk::new(vec![
15675                Vector::from_values(LogicalType::Integer, &[Value::Integer(part as i32)])
15676                    .expect("integers"),
15677            ])
15678            .expect("matching rows");
15679            writer.append(&chunk).expect("one part");
15680        }
15681        writer.finish().expect("commit");
15682
15683        let read = |told: bool| {
15684            let reader = Reader::open(&path).expect("reopen from disk");
15685            assert_eq!(reader.table().stripes().len(), workers, "a stripe per worker");
15686            if told {
15687                reader.keep_stripes(workers);
15688            }
15689            // Through once a part at a time, so that the pass below is the one that reads pages.
15690            for part in 0..reader.parts() {
15691                reader.read(part, &[0]).expect("a part");
15692            }
15693            let barrier = std::sync::Barrier::new(workers);
15694            std::thread::scope(|scope| {
15695                for (worker, run) in reader.stripe_parts().into_iter().enumerate() {
15696                    let reader = &reader;
15697                    let barrier = &barrier;
15698                    scope.spawn(move || {
15699                        for part in run {
15700                            barrier.wait();
15701                            let chunk = reader.read(part, &[0]).expect("a part of my own stripe");
15702                            assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
15703                        }
15704                        assert!(worker < workers);
15705                    });
15706                }
15707            });
15708            reader.pages.load(Atomic::Relaxed)
15709        };
15710
15711        assert_eq!(read(true), workers, "one page read per stripe and no more");
15712        assert!(read(false) > workers, "a cache that small is read again on every part");
15713        fs::remove_file(path).expect("remove scratch file");
15714    }
15715
15716    /// A damaged index page is caught before anything decodes a part out of it.
15717    ///
15718    /// The index is the one structure a reader trusts to find bytes with, so it carries a checksum
15719    /// per column section rather than one for the page, and this is what says that check runs.
15720    #[test]
15721    fn a_damaged_index_page_is_an_error() {
15722        let path = path("damaged-index");
15723        let mut writer =
15724            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
15725                .expect("new file");
15726        writer.append(&sample_ids()).expect("first part");
15727        writer.append(&sample_ids()).expect("second part");
15728        writer.finish().expect("commit");
15729
15730        let reader = Reader::open(&path).expect("valid directory");
15731        let index = reader.table.stripes[0].index;
15732        let mut byte = [0; 1];
15733        read_at(&reader.file, index.offset, &mut byte).expect("the first part length");
15734        let mut file = OpenOptions::new().write(true).open(&path).expect("open index page");
15735        file.seek(SeekFrom::Start(index.offset)).expect("index start");
15736        file.write_all(&[!byte[0]]).expect("damage the first part length");
15737        let error = reader.read(1, &[0]).expect_err("a damaged index must not be used");
15738        assert!(error.message().contains("index page section checksum differs"), "{error}");
15739        fs::remove_file(path).expect("remove scratch file");
15740    }
15741
15742    /// Every integer width the format knows about, written and read back.
15743    ///
15744    /// The unsigned ones are the reason ClickBench can be stored at all: `hits` types `EventDate`
15745    /// as `USMALLINT`, and one unsupported column meant the whole table was refused. The extremes
15746    /// are in here on purpose, because a width that round trips through the wrong signedness only
15747    /// goes wrong at the end of its range.
15748    #[test]
15749    fn every_integer_width_round_trips_through_a_page() {
15750        let path = path("integer-widths");
15751        let columns = [
15752            (LogicalType::TinyInt, vec![Value::TinyInt(i8::MIN), Value::TinyInt(i8::MAX)]),
15753            (LogicalType::UTinyInt, vec![Value::UTinyInt(0), Value::UTinyInt(u8::MAX)]),
15754            (LogicalType::SmallInt, vec![Value::SmallInt(i16::MIN), Value::SmallInt(i16::MAX)]),
15755            (LogicalType::USmallInt, vec![Value::USmallInt(0), Value::USmallInt(u16::MAX)]),
15756            (LogicalType::Integer, vec![Value::Integer(i32::MIN), Value::Integer(i32::MAX)]),
15757            (LogicalType::UInteger, vec![Value::UInteger(0), Value::UInteger(u32::MAX)]),
15758            (LogicalType::BigInt, vec![Value::BigInt(i64::MIN), Value::BigInt(i64::MAX)]),
15759            (LogicalType::UBigInt, vec![Value::UBigInt(0), Value::UBigInt(u64::MAX)]),
15760        ];
15761        let fields = columns
15762            .iter()
15763            .enumerate()
15764            .map(|(at, (ty, _))| Field::required(format!("c{at}"), ty.clone()))
15765            .collect::<Vec<_>>();
15766        let vectors = columns
15767            .iter()
15768            .map(|(ty, values)| Vector::from_values(ty.clone(), values).expect("a vector"))
15769            .collect::<Vec<_>>();
15770        let mut writer = Writer::create(&path, "widths", fields).expect("new file");
15771        writer.append(&Chunk::new(vectors).expect("matching rows")).expect("one stripe");
15772        writer.finish().expect("commit");
15773
15774        let reader = Reader::open(&path).expect("reopen from disk");
15775        let wanted = (0..columns.len()).collect::<Vec<_>>();
15776        let read = reader.read(0, &wanted).expect("every column");
15777        assert_eq!(read.len(), 2);
15778        // row at a time: each column has its own type and its own pair of extremes.
15779        for (at, (ty, values)) in columns.iter().enumerate() {
15780            assert_eq!(read.value_at(0, at), values[0], "the low end of {ty}");
15781            assert_eq!(read.value_at(1, at), values[1], "the high end of {ty}");
15782        }
15783        fs::remove_file(path).expect("remove scratch file");
15784    }
15785
15786    /// The rest of the fixed width types, and the byte strings, written and read back.
15787    ///
15788    /// The extremes again, and for a float that means more than the ends of the range. Negative
15789    /// zero and a NaN are the two values that go through an encoder unnoticed and come back
15790    /// different, so they are here on purpose, and the NaN is compared by its bits rather than by
15791    /// `==`, which a NaN fails against itself.
15792    ///
15793    /// A blob is here beside them because it is the same round trip asked of bytes that are not
15794    /// text. The value in it is not UTF-8, so a path that reads a payload as a string on the way
15795    /// past turns this test red rather than turning a user's column into nulls.
15796    #[test]
15797    fn every_other_type_the_format_knows_round_trips_through_a_page() {
15798        let path = path("other-types");
15799        let columns = [
15800            (LogicalType::Float, vec![Value::Float(f32::MIN), Value::Float(-0.0)]),
15801            (LogicalType::Double, vec![Value::Double(f64::MIN), Value::Double(f64::MAX)]),
15802            (LogicalType::HugeInt, vec![Value::HugeInt(i128::MIN), Value::HugeInt(i128::MAX)]),
15803            (LogicalType::UHugeInt, vec![Value::UHugeInt(0), Value::UHugeInt(u128::MAX)]),
15804            (LogicalType::Time, vec![Value::Time(0), Value::Time(86_399_999_999)]),
15805            (LogicalType::TimeTz, vec![Value::TimeTz(-50_400_000_000), Value::TimeTz(0)]),
15806            (
15807                LogicalType::TimestampTz,
15808                vec![Value::TimestampTz(i64::MIN + 1), Value::TimestampTz(i64::MAX)],
15809            ),
15810            (
15811                LogicalType::Interval,
15812                vec![
15813                    Value::Interval { months: i32::MIN, days: i32::MAX, micros: i64::MIN },
15814                    Value::Interval { months: 13, days: -1, micros: 1 },
15815                ],
15816            ),
15817            (
15818                LogicalType::Blob,
15819                vec![Value::Blob(vec![0, 0xff, 0x80, 0xfe]), Value::Blob(Vec::new())],
15820            ),
15821        ];
15822        let fields = columns
15823            .iter()
15824            .enumerate()
15825            .map(|(at, (ty, _))| Field::required(format!("c{at}"), ty.clone()))
15826            .collect::<Vec<_>>();
15827        let vectors = columns
15828            .iter()
15829            .map(|(ty, values)| Vector::from_values(ty.clone(), values).expect("a vector"))
15830            .collect::<Vec<_>>();
15831        let mut writer = Writer::create(&path, "others", fields).expect("new file");
15832        writer.append(&Chunk::new(vectors).expect("matching rows")).expect("one stripe");
15833        writer.finish().expect("commit");
15834
15835        let reader = Reader::open(&path).expect("reopen from disk");
15836        let wanted = (0..columns.len()).collect::<Vec<_>>();
15837        let read = reader.read(0, &wanted).expect("every column");
15838        assert_eq!(read.len(), 2);
15839        for (at, (ty, values)) in columns.iter().enumerate() {
15840            assert_eq!(read.value_at(0, at), values[0], "the low end of {ty}");
15841            assert_eq!(read.value_at(1, at), values[1], "the high end of {ty}");
15842        }
15843        // A float keeps its sign through a zero, which `==` says nothing about because negative
15844        // zero and zero compare equal.
15845        let Value::Float(zero) = read.value_at(1, 0) else { panic!("a float stays a float") };
15846        assert!(zero.is_sign_negative(), "a negative zero came back as {zero}");
15847
15848        fs::remove_file(path).expect("remove scratch file");
15849    }
15850
15851    /// A NaN is still a NaN after a trip through a page.
15852    ///
15853    /// Apart from the other floats because it cannot be asserted the same way. A NaN is not equal
15854    /// to itself, so a comparison against the value that was written passes for every NaN and for
15855    /// nothing else, which is the one assertion that would not catch a page that lost it.
15856    #[test]
15857    fn a_nan_survives_being_written_down() {
15858        let path = path("nan");
15859        let nan = Vector::from_values(LogicalType::Double, &[Value::Double(f64::NAN)])
15860            .expect("a NaN vector");
15861        let mut writer =
15862            Writer::create(&path, "nan", vec![Field::required("d", LogicalType::Double)])
15863                .expect("new file");
15864        writer.append(&Chunk::new(vec![nan]).expect("one column")).expect("one stripe");
15865        writer.finish().expect("commit");
15866        let read = Reader::open(&path).expect("reopen").read(0, &[0]).expect("the column");
15867        let Value::Double(back) = read.value_at(0, 0) else { panic!("a double stays a double") };
15868        assert!(back.is_nan(), "a NaN came back as {back}");
15869        fs::remove_file(path).expect("remove scratch file");
15870    }
15871
15872    /// A uuid and a bit string, which have no `Value` arm of their own and are checked as bits.
15873    ///
15874    /// A uuid is the 128 bit lane and a bit string is bytes, and neither of them reads back as
15875    /// anything in `Value` today, so asking for a value here would compare two nulls and pass
15876    /// whatever the file held. The data underneath is what the storage promise is about, so that is
15877    /// what this reads.
15878    #[test]
15879    fn a_uuid_and_a_bit_string_come_back_as_the_bits_that_went_in() {
15880        let path = path("uuid-and-bit");
15881        let uuids = vec![0_i128, i128::MIN, -1];
15882        let mut bits = StringColumn::new();
15883        for value in [&b"\x02\xff"[..], &b""[..], &b"\x00\x01\x02\x03\x04\x05"[..]] {
15884            bits.push_bytes(value);
15885        }
15886        let expected = bits.clone();
15887        let fields =
15888            vec![Field::required("u", LogicalType::Uuid), Field::required("b", LogicalType::Bit)];
15889        let vectors = vec![
15890            Vector::flat(LogicalType::Uuid, Data::Int128(uuids.clone().into())).expect("uuids"),
15891            Vector::flat(LogicalType::Bit, Data::Varlen(bits)).expect("bit strings"),
15892        ];
15893        let mut writer = Writer::create(&path, "ids", fields).expect("new file");
15894        writer.append(&Chunk::new(vectors).expect("matching rows")).expect("one stripe");
15895        writer.finish().expect("commit");
15896
15897        let reader = Reader::open(&path).expect("reopen from disk");
15898        let read = reader.read(0, &[0, 1]).expect("both columns").flatten().expect("flat");
15899        let Some(Data::Int128(back)) = read.column(0).expect("the uuids").data() else {
15900            panic!("a uuid column is the 128 bit lane")
15901        };
15902        assert_eq!(back.as_slice(), uuids.as_slice());
15903        let Some(Data::Varlen(back)) = read.column(1).expect("the bits").data() else {
15904            panic!("a bit column is bytes")
15905        };
15906        for row in 0..expected.len() {
15907            assert_eq!(back.bytes(row), expected.bytes(row), "row {row} of the bit column");
15908        }
15909        fs::remove_file(path).expect("remove scratch file");
15910    }
15911
15912    /// Counting a run at once has to leave the candidate table exactly where counting its rows one
15913    /// at a time would, including once the table is full and a run is turned away row by row.
15914    #[test]
15915    fn a_run_counted_at_once_leaves_the_candidates_a_row_at_a_time_would() {
15916        let mut rows: Vec<Option<u64>> = Vec::new();
15917        let mut state = 0x2545_f491_4f6c_dd1d_u64;
15918        for index in 0..400_000_u64 {
15919            state ^= state << 13;
15920            state ^= state >> 7;
15921            state ^= state << 17;
15922            let times = 1 + (state % 7) as usize;
15923            let bits = match state % 11 {
15924                0 => None,
15925                1..=3 => Some(state % 16),
15926                _ => Some(index.wrapping_mul(0x9e37_79b9_7f4a_7c15)),
15927            };
15928            rows.extend(std::iter::repeat_n(bits, times));
15929        }
15930        let mut by_row = Candidates::default();
15931        for &bits in &rows {
15932            by_row.add(bits, 1);
15933        }
15934        let mut by_run = Candidates::default();
15935        let mut run = Run::default();
15936        let mut runs = 0_usize;
15937        for &bits in &rows {
15938            if let Some((bits, times)) = run.push(bits) {
15939                by_run.add(bits, times);
15940                runs += 1;
15941            }
15942        }
15943        if let Some((bits, times)) = run.take() {
15944            by_run.add(bits, times);
15945        }
15946        assert!(runs < rows.len() / 2, "the rows came in runs");
15947        assert!(by_row.decrements > 0, "the table filled and turned values away");
15948        assert_eq!(sorted_candidates(&by_run), sorted_candidates(&by_row));
15949        assert_eq!(by_run.nulls, by_row.nulls);
15950        assert_eq!(by_run.decrements, by_row.decrements);
15951    }
15952
15953    fn sorted_candidates(candidates: &Candidates) -> Vec<(u64, u32)> {
15954        let mut pairs = candidates.pairs().collect::<Vec<_>>();
15955        pairs.sort_unstable();
15956        assert_eq!(pairs.len(), candidates.held, "the count of held slots drifted");
15957        pairs
15958    }
15959
15960    /// The Misra-Gries table as it was written over a `HashMap`, kept as the oracle the open
15961    /// addressed one has to agree with.
15962    #[derive(Default)]
15963    struct MapCandidates {
15964        counts: HashMap<u64, u32>,
15965        nulls: u32,
15966        decrements: u64,
15967    }
15968
15969    impl MapCandidates {
15970        fn add(&mut self, bits: Option<u64>, mut times: u32) {
15971            while times > 0 {
15972                let held = match bits {
15973                    Some(bits) => self.counts.get_mut(&bits),
15974                    None if self.nulls != 0 => Some(&mut self.nulls),
15975                    None => None,
15976                };
15977                if let Some(count) = held {
15978                    *count = count.saturating_add(times);
15979                    return;
15980                }
15981                if self.counts.len() + usize::from(self.nulls != 0) < FREQUENCY_CANDIDATES {
15982                    match bits {
15983                        Some(bits) => {
15984                            self.counts.insert(bits, times);
15985                        }
15986                        None => self.nulls = times,
15987                    }
15988                    return;
15989                }
15990                self.counts.retain(|_, count| {
15991                    *count -= 1;
15992                    *count != 0
15993                });
15994                self.nulls = self.nulls.saturating_sub(1);
15995                self.decrements = self.decrements.saturating_add(1);
15996                times -= 1;
15997            }
15998        }
15999    }
16000
16001    /// Near unique values, a few heavy ones, nulls, and runs, through enough rows that the table
16002    /// fills, grows through every size and is decremented many times over. Both tables have to hold
16003    /// the same candidates with the same counts at the end, and at points along the way.
16004    #[test]
16005    fn the_open_addressed_candidates_agree_with_the_map_they_replaced() {
16006        for seed in [0x2545_f491_4f6c_dd1d_u64, 0x9e37_79b9_7f4a_7c15, 7] {
16007            let mut table = Candidates::default();
16008            let mut oracle = MapCandidates::default();
16009            let mut state = seed;
16010            for index in 0..300_000_u64 {
16011                state ^= state << 13;
16012                state ^= state >> 7;
16013                state ^= state << 17;
16014                let bits = match state % 13 {
16015                    0 => None,
16016                    1..=4 => Some(state % 40),
16017                    5 => Some((index % 1000) * 1_000_000),
16018                    _ => Some(state),
16019                };
16020                let times = 1 + (state >> 60) as u32 % 3;
16021                table.add(bits, times);
16022                oracle.add(bits, times);
16023                if index % 50_000 == 0 {
16024                    let mut expected =
16025                        oracle.counts.iter().map(|(&b, &c)| (b, c)).collect::<Vec<_>>();
16026                    expected.sort_unstable();
16027                    assert_eq!(sorted_candidates(&table), expected, "seed {seed} row {index}");
16028                }
16029            }
16030            let mut expected = oracle.counts.iter().map(|(&b, &c)| (b, c)).collect::<Vec<_>>();
16031            expected.sort_unstable();
16032            assert_eq!(sorted_candidates(&table), expected, "seed {seed}");
16033            assert_eq!(table.nulls, oracle.nulls, "seed {seed}");
16034            assert_eq!(table.decrements, oracle.decrements, "seed {seed}");
16035            assert!(table.decrements > 0, "seed {seed} never filled the table");
16036            for &(bits, _) in &expected {
16037                assert!(table.position(bits).is_some(), "seed {seed} lost {bits}");
16038            }
16039        }
16040    }
16041
16042    #[test]
16043    fn numeric_frequency_candidates_keep_bounded_row_ordinals() {
16044        let path = path("frequency-ordinals");
16045        let mut writer =
16046            Writer::create(&path, "items", vec![Field::required("id", LogicalType::BigInt)])
16047                .expect("new file");
16048        let mut values = Vec::new();
16049        for leader in 0..10_i64 {
16050            values.extend(std::iter::repeat_n(leader, 100));
16051        }
16052        values.extend(1_000_i64..41_000);
16053        for part in values.chunks(1_024) {
16054            let vector = Vector::flat(LogicalType::BigInt, Data::Int64(part.to_vec().into()))
16055                .expect("big integers");
16056            writer.append(&Chunk::new(vec![vector]).expect("one column")).expect("one stripe");
16057        }
16058        writer.finish().expect("commit");
16059
16060        let reader = Reader::open(&path).expect("reopen from disk");
16061        let occurrences =
16062            reader.frequency_occurrences(0).expect("valid metadata").expect("bounded ordinals");
16063        assert!(occurrences.omitted_max < 100);
16064        assert!(occurrences.ordinals.len() <= FREQUENCY_ORDINALS);
16065        assert_eq!(occurrences.anchor_indices.len(), occurrences.ordinals.len());
16066        assert!(occurrences.ordinals.windows(2).all(|pair| pair[0] < pair[1]));
16067        assert_eq!(&occurrences.ordinals[..1_000], &(0_u64..1_000).collect::<Vec<_>>());
16068        assert_eq!(
16069            &occurrences.anchor_indices[..1_000]
16070                .iter()
16071                .map(|&entry| occurrences.anchors[entry as usize].clone())
16072                .collect::<Vec<_>>(),
16073            &(0_i64..10)
16074                .flat_map(|leader| std::iter::repeat_n(Value::BigInt(leader), 100))
16075                .collect::<Vec<_>>()
16076        );
16077        fs::remove_file(path).expect("remove scratch file");
16078    }
16079
16080    #[test]
16081    fn numeric_frequencies_count_nulls_and_values_past_the_top_of_bigint() {
16082        // Ten leaders, then more unique values than the candidate table holds, so the first pass
16083        // has to decrement and the counts come from the recount. The unsigned leaders sit above
16084        // `i64::MAX`, where reading the bits as signed would give a different value, and the signed
16085        // ones are negative, where reading them as unsigned would.
16086        let path = path("frequency-bits");
16087        let mut writer = Writer::create(
16088            &path,
16089            "items",
16090            vec![Field::new("u", LogicalType::UBigInt), Field::new("s", LogicalType::BigInt)],
16091        )
16092        .expect("new file");
16093        let mut rows = Vec::new();
16094        let mut leaders = Vec::new();
16095        for leader in 0..10_u64 {
16096            let count = 300 - leader * 10;
16097            let (unsigned, signed) = if leader == 0 {
16098                (Value::Null, Value::Null)
16099            } else {
16100                (Value::UBigInt(u64::MAX - leader), Value::BigInt(-(leader as i64)))
16101            };
16102            rows.extend(std::iter::repeat_n((unsigned.clone(), signed.clone()), count as usize));
16103            leaders.push(((unsigned, count), (signed, count)));
16104        }
16105        rows.extend((1_000..41_000_u64).map(|id| (Value::UBigInt(id), Value::BigInt(id as i64))));
16106        for part in rows.chunks(1_024) {
16107            let unsigned = part.iter().map(|(value, _)| value.clone()).collect::<Vec<_>>();
16108            let signed = part.iter().map(|(_, value)| value.clone()).collect::<Vec<_>>();
16109            let chunk = Chunk::new(vec![
16110                Vector::from_values(LogicalType::UBigInt, &unsigned).expect("unsigned"),
16111                Vector::from_values(LogicalType::BigInt, &signed).expect("signed"),
16112            ])
16113            .expect("matching columns");
16114            writer.append(&chunk).expect("rows");
16115        }
16116        writer.finish().expect("commit");
16117
16118        let reader = Reader::open(&path).expect("reopen from disk");
16119        for column in 0..2 {
16120            let prefix =
16121                reader.frequency_prefix(column).expect("valid metadata").expect("a synopsis");
16122            let wanted = leaders
16123                .iter()
16124                .map(|(unsigned, signed)| if column == 0 { unsigned } else { signed })
16125                .cloned()
16126                .collect::<Vec<_>>();
16127            assert_eq!(&prefix.entries[..10], &wanted[..], "column {column}");
16128            assert!(prefix.omitted_max < 210, "column {column}");
16129            assert_eq!(
16130                reader.distinct_values(column).expect("valid metadata"),
16131                Some(9 + 40_000),
16132                "column {column}"
16133            );
16134        }
16135        fs::remove_file(path).expect("remove scratch file");
16136    }
16137
16138    #[test]
16139    fn a_narrow_column_takes_its_frequencies_from_the_tally_and_they_match_the_rows() {
16140        // Every column here has fewer distinct values than the tally holds, so the close takes its
16141        // counts from the gather rather than reading the pages back. The types are the ones whose
16142        // bits could come out wrong on that road: a negative tiny integer that has to be sign
16143        // extended, an unsigned one past the top of `INTEGER`, a date and a timestamp. A null every
16144        // thirteenth row checks that the nulls come from the pass and not from the list.
16145        let path = path("frequency-tally");
16146        let types = [
16147            LogicalType::TinyInt,
16148            LogicalType::UInteger,
16149            LogicalType::Date,
16150            LogicalType::Timestamp,
16151        ];
16152        let value = |ty: &LogicalType, at: i64| match ty {
16153            LogicalType::TinyInt => Value::TinyInt((at % 250 - 125) as i8),
16154            LogicalType::UInteger => Value::UInteger(u32::MAX - at as u32),
16155            LogicalType::Date => Value::Date(19_000 - at as i32),
16156            _ => Value::Timestamp(1_700_000_000_000_000 - at * 1_000_003),
16157        };
16158        let fields = types
16159            .iter()
16160            .enumerate()
16161            .map(|(at, ty)| Field::new(format!("c{at}"), ty.clone()))
16162            .collect::<Vec<_>>();
16163        let mut writer = Writer::create(&path, "items", fields).expect("new file");
16164        let mut rows = Vec::new();
16165        for at in 0..250_i64 {
16166            for _ in 0..=(at % 37) {
16167                rows.push(if rows.len() % 13 == 0 { None } else { Some(at) });
16168            }
16169        }
16170        for part in rows.chunks(1_000) {
16171            let columns = types
16172                .iter()
16173                .map(|ty| {
16174                    let values = part
16175                        .iter()
16176                        .map(|row| row.map_or(Value::Null, |at| value(ty, at)))
16177                        .collect::<Vec<_>>();
16178                    Vector::from_values(ty.clone(), &values).expect("a column")
16179                })
16180                .collect();
16181            writer.append(&Chunk::new(columns).expect("matching columns")).expect("rows");
16182        }
16183        writer.finish().expect("commit");
16184
16185        let reader = Reader::open(&path).expect("reopen from disk");
16186        for (column, ty) in types.iter().enumerate() {
16187            let mut counts = HashMap::<Option<i64>, u64>::new();
16188            for row in &rows {
16189                *counts.entry(*row).or_default() += 1;
16190            }
16191            let wanted = counts
16192                .into_iter()
16193                .map(|(row, count)| (row.map_or(Value::Null, |at| value(ty, at)), count))
16194                .collect::<Vec<_>>();
16195            let prefix =
16196                reader.frequency_prefix(column).expect("valid metadata").expect("a synopsis");
16197            assert_eq!(prefix.entries.len(), 2, "column {column}");
16198            assert!(prefix.omitted_max > 0, "column {column}");
16199            for (value, count) in &prefix.entries {
16200                let held =
16201                    wanted.iter().find(|(wanted, _)| wanted == value).map(|(_, count)| count);
16202                assert_eq!(held, Some(count), "column {column} value {value:?}");
16203            }
16204            assert!(prefix.entries.windows(2).all(|pair| pair[0].1 >= pair[1].1));
16205            assert_eq!(
16206                reader.distinct_values(column).expect("valid metadata"),
16207                Some(wanted.len() as u64 - 1),
16208                "column {column}"
16209            );
16210        }
16211        fs::remove_file(path).expect("remove scratch file");
16212    }
16213
16214    #[test]
16215    fn distinct_counts_are_exact_either_side_of_a_full_candidate_table() {
16216        // The count comes from the candidate table while it has room and from the set once it
16217        // fills, so the sizes around the fill, with and without a null taking a place, are where a
16218        // value could be counted twice or missed. Zero is in every column because the set keeps it
16219        // apart from the other values, and every value comes back later to be counted again.
16220        let edge = FREQUENCY_CANDIDATES as i64;
16221        for distinct in [0, 1, 7, edge - 2, edge - 1, edge, edge + 1, edge + 2, 3 * edge] {
16222            for with_null in [false, true] {
16223                let path = path("distinct-edge");
16224                let mut writer =
16225                    Writer::create(&path, "items", vec![Field::new("id", LogicalType::BigInt)])
16226                        .expect("new file");
16227                let mut values = Vec::new();
16228                for round in 0..2 {
16229                    for value in 0..distinct {
16230                        let repeat = if round == 0 { 1 + (value % 3) as usize } else { 1 };
16231                        values.extend(std::iter::repeat_n(
16232                            Value::BigInt(value * 7_919 % distinct),
16233                            repeat,
16234                        ));
16235                        if with_null && value % 1_000 == 0 {
16236                            values.push(Value::Null);
16237                        }
16238                    }
16239                }
16240                if with_null {
16241                    values.push(Value::Null);
16242                }
16243                for part in values.chunks(1_024) {
16244                    let chunk = Chunk::new(vec![
16245                        Vector::from_values(LogicalType::BigInt, part).expect("ids"),
16246                    ])
16247                    .expect("one column");
16248                    writer.append(&chunk).expect("rows");
16249                }
16250                writer.finish().expect("commit");
16251                let reader = Reader::open(&path).expect("reopen from disk");
16252                assert_eq!(
16253                    reader.distinct_values(0).expect("valid metadata"),
16254                    Some(distinct as u64),
16255                    "{distinct} values, null {with_null}"
16256                );
16257                fs::remove_file(path).expect("remove scratch file");
16258            }
16259        }
16260    }
16261
16262    #[test]
16263    fn narrow_nonzero_count_matches_the_full_reader_across_stripes() {
16264        let path = path("quick-nonzero");
16265        let mut writer = Writer::create(
16266            &path,
16267            "items",
16268            vec![Field::new("label", LogicalType::Varchar), Field::new("id", LogicalType::Integer)],
16269        )
16270        .expect("create");
16271        for ids in [
16272            &[Value::Integer(0), Value::Null, Value::Integer(3)][..],
16273            &[Value::Integer(0), Value::Integer(7), Value::Null][..],
16274        ] {
16275            let labels = vec![Value::Varchar("same".into()); ids.len()];
16276            writer
16277                .append(
16278                    &Chunk::new(vec![
16279                        Vector::from_values(LogicalType::Varchar, &labels).expect("labels"),
16280                        Vector::from_values(LogicalType::Integer, ids).expect("ids"),
16281                    ])
16282                    .expect("chunk"),
16283                )
16284                .expect("append");
16285        }
16286        writer.finish().expect("finish");
16287        let catalog = Catalog::open(&path).expect("catalog");
16288        assert_eq!(catalog.entries[0].nonzero, vec![None, None]);
16289        assert_eq!(catalog.entries[0].aggregates, vec![None, Some((10, 4))]);
16290        assert_eq!(catalog.entries[0].distincts, vec![Some(1), Some(3)]);
16291        assert_eq!(catalog.exact_numeric_frequencies("items", 1).expect("frequencies"), None);
16292        let prefix = catalog
16293            .table("items")
16294            .expect("reader")
16295            .frequency_prefix(1)
16296            .expect("valid metadata")
16297            .expect("partial frequencies");
16298        assert_eq!(prefix.entries, vec![(Value::Null, 2), (Value::Integer(0), 2)]);
16299        assert_eq!(prefix.omitted_max, 1);
16300        assert_eq!(catalog.distinct_count("items", 1).expect("distinct count"), Some(3));
16301        assert_eq!(
16302            catalog.integer_extremes("items", 1).expect("extremes"),
16303            Some(IntegerExtremes::Values { low: 0, high: 7 })
16304        );
16305        assert_eq!(
16306            catalog.aggregate_sums("items", &[1]).expect("catalog sums"),
16307            Some(CertifiedSums { columns: vec![(10, 4)], rows: 6 })
16308        );
16309        assert_eq!(catalog.nonzero_count("items", 1).expect("quick count"), Some(2));
16310        let mut legacy = catalog.clone();
16311        Arc::make_mut(&mut legacy.entries)[0].nonzero[1] = Some(999);
16312        assert_eq!(legacy.nonzero_count("items", 1).expect("ignore legacy count"), Some(2));
16313        Arc::make_mut(&mut legacy.entries)[0].frequencies[1] = None;
16314        assert_eq!(legacy.nonzero_count("items", 1).expect("directory fallback"), Some(2));
16315        Writer::certify_counts(&path).expect("recertify");
16316        assert_eq!(
16317            Catalog::open(&path).expect("reopen").nonzero_count("items", 1).expect("count"),
16318            Some(2)
16319        );
16320        assert_eq!(
16321            Catalog::open(&path).expect("reopen").aggregate_sums("items", &[1]).expect("sums"),
16322            Some(CertifiedSums { columns: vec![(10, 4)], rows: 6 })
16323        );
16324        assert_eq!(
16325            Catalog::open(&path).expect("reopen").distinct_count("items", 1).expect("distinct"),
16326            Some(3)
16327        );
16328        assert_eq!(
16329            Catalog::open(&path).expect("reopen").integer_extremes("items", 1).expect("ends"),
16330            Some(IntegerExtremes::Values { low: 0, high: 7 })
16331        );
16332        assert_eq!(
16333            Catalog::open(&path)
16334                .expect("reopen")
16335                .exact_numeric_frequencies("items", 1)
16336                .expect("frequencies"),
16337            None
16338        );
16339        assert_eq!(catalog.table("items").expect("reader").null_count(1).expect("nulls"), 2);
16340        fs::remove_file(path).expect("remove scratch file");
16341    }
16342
16343    #[test]
16344    fn numeric_string_pair_leaders_are_certified_in_the_directory() {
16345        let path = path("pair-frequencies");
16346        let mut pairs = Vec::new();
16347        pairs.extend(std::iter::repeat_n((1_i64, "alpha".to_string()), 100));
16348        pairs.extend(std::iter::repeat_n((1_i64, "beta".to_string()), 50));
16349        pairs.extend(std::iter::repeat_n((2_i64, "gamma".to_string()), 40));
16350        pairs.extend((1_000_i64..1_600).map(|id| (id, format!("tail {id}"))));
16351        let mut writer = Writer::create(
16352            &path,
16353            "items",
16354            vec![
16355                Field::required("id", LogicalType::BigInt),
16356                Field::required("phrase", LogicalType::Varchar),
16357            ],
16358        )
16359        .expect("new file");
16360        for part in pairs.chunks(1_024) {
16361            let ids = part.iter().map(|(id, _)| Value::BigInt(*id)).collect::<Vec<_>>();
16362            let phrases =
16363                part.iter().map(|(_, phrase)| Value::Varchar(phrase.clone())).collect::<Vec<_>>();
16364            writer
16365                .append(
16366                    &Chunk::new(vec![
16367                        Vector::from_values(LogicalType::BigInt, &ids).expect("ids"),
16368                        Vector::from_values(LogicalType::Varchar, &phrases).expect("phrases"),
16369                    ])
16370                    .expect("matching columns"),
16371                )
16372                .expect("rows");
16373        }
16374        writer.finish().expect("commit");
16375
16376        let reader = Reader::open(&path).expect("reopen from disk");
16377        assert!(
16378            reader.table.pair_frequencies.is_empty(),
16379            "no query-specific pair result is stored"
16380        );
16381        fs::remove_file(path).expect("remove scratch file");
16382    }
16383
16384    #[test]
16385    fn legacy_group_answers_are_ignored() {
16386        let path = path("legacy-group-answers");
16387        let mut writer = Writer::create(
16388            &path,
16389            "items",
16390            vec![
16391                Field::required("id", LogicalType::BigInt),
16392                Field::required("text", LogicalType::Varchar),
16393            ],
16394        )
16395        .expect("new file");
16396        writer
16397            .append(
16398                &Chunk::new(vec![
16399                    Vector::from_values(LogicalType::BigInt, &[Value::BigInt(1)]).expect("id"),
16400                    Vector::from_values(LogicalType::Varchar, &[Value::Varchar("x".into())])
16401                        .expect("text"),
16402                ])
16403                .expect("row"),
16404            )
16405            .expect("append");
16406        writer.finish().expect("commit");
16407        let mut reader = Reader::open(&path).expect("reopen");
16408        let table = Arc::make_mut(&mut reader.table);
16409        table.pair_frequencies.push(PairFrequencySummary {
16410            first: 0,
16411            second: 1,
16412            entries: vec![PairFrequencyEntry { first_entry: 0, second: Some(0), count: 999 }],
16413            omitted_max: 0,
16414        });
16415        table.host_groups = Some(host::HostSummary {
16416            column: 1,
16417            omitted_max: 0,
16418            entries: vec![host::HostEntry {
16419                host: "fake.test".into(),
16420                count: 999,
16421                bytes_sum: 999,
16422                minimum: "x".into(),
16423            }],
16424        });
16425        assert_eq!(reader.top_pair_frequencies(0, 1, 1).expect("legacy pair"), None);
16426        assert_eq!(reader.host_groups(1, 1).expect("legacy host"), None);
16427        fs::remove_file(path).expect("remove scratch file");
16428    }
16429
16430    /// The bug this is here for cost a 43 GB ClickBench table and an hour of reloading it. The
16431    /// format went from 11 to 12, every binary built after that said "magic or major version is
16432    /// unsupported" about the file, and there was no way to tell from the message whether the path
16433    /// was wrong, the file was truncated, or it was ours and simply older. The number this build
16434    /// wants is the whole answer and it was the one thing the message did not carry.
16435    #[test]
16436    fn a_file_from_another_format_says_which_format_it_is() {
16437        let older = path("older-format");
16438        let mut writer =
16439            Writer::create(&older, "items", vec![Field::new("id", LogicalType::Integer)])
16440                .expect("new file");
16441        let chunk = Chunk::new(vec![
16442            Vector::flat(LogicalType::Integer, Data::Int32(vec![1, 2, 3].into()))
16443                .expect("integers"),
16444        ])
16445        .expect("chunk");
16446        writer.append(&chunk).expect("page written");
16447        writer.finish().expect("commit");
16448
16449        // A format below the whole readable set, rather than `FORMAT - 1`, because the set has
16450        // more than one member now: format 22 is deliberately still readable, so the version that
16451        // has to be refused is the one under the oldest one accepted.
16452        let unreadable =
16453            READABLE.iter().copied().min().expect("at least one format is readable") - 1;
16454        let mut file = OpenOptions::new().write(true).open(&older).expect("open for the header");
16455        file.seek(SeekFrom::Start(8)).expect("the version follows the magic");
16456        file.write_all(&unreadable.to_le_bytes()).expect("write an older version");
16457        drop(file);
16458        let complaint = Reader::open(&older).expect_err("an older format is refused").to_string();
16459        assert!(complaint.contains(&format!("format {unreadable}")), "{complaint}");
16460        assert!(complaint.contains(&format!("format {FORMAT}")), "{complaint}");
16461
16462        let mut file = OpenOptions::new().write(true).open(&older).expect("open for the header");
16463        file.seek(SeekFrom::Start(0)).expect("the magic is first");
16464        file.write_all(b"NOTRUDB!").expect("write another engine's magic");
16465        drop(file);
16466        let complaint = Reader::open(&older).expect_err("a foreign file is refused").to_string();
16467        assert!(complaint.contains("magic"), "{complaint}");
16468        assert!(!complaint.contains("format"), "a version has nothing to do with it: {complaint}");
16469        fs::remove_file(older).expect("remove scratch file");
16470    }
16471
16472    #[test]
16473    fn an_unfinished_or_damaged_file_does_not_answer_with_partial_rows() {
16474        let unfinished = path("unfinished");
16475        let mut writer =
16476            Writer::create(&unfinished, "items", vec![Field::new("id", LogicalType::Integer)])
16477                .expect("new file");
16478        let chunk = Chunk::new(vec![
16479            Vector::flat(LogicalType::Integer, Data::Int32(vec![1, 2, 3].into()))
16480                .expect("integers"),
16481        ])
16482        .expect("chunk");
16483        writer.append(&chunk).expect("page written");
16484        drop(writer);
16485        assert!(Reader::open(&unfinished).is_err(), "no directory was committed");
16486        fs::remove_file(unfinished).expect("remove scratch file");
16487
16488        let damaged = path("damaged");
16489        let mut writer =
16490            Writer::create(&damaged, "items", vec![Field::new("id", LogicalType::Integer)])
16491                .expect("new file");
16492        writer.append(&chunk).expect("page written");
16493        writer.finish().expect("commit");
16494        let reader = Reader::open(&damaged).expect("valid directory");
16495        let mut file =
16496            OpenOptions::new().write(true).open(&damaged).expect("open for a damaged page");
16497        file.seek(SeekFrom::Start(HEADER + 1)).expect("inside first page");
16498        file.write_all(&[255]).expect("damage one byte");
16499        assert!(reader.read(0, &[0]).is_err(), "page checksum rejects corruption");
16500        fs::remove_file(damaged).expect("remove scratch file");
16501    }
16502
16503    #[test]
16504    fn damaged_lazy_dictionary_payload_is_an_error() {
16505        let path = path("damaged-dictionary");
16506        let mut writer = Writer::create(
16507            &path,
16508            "items",
16509            vec![
16510                Field::required("id", LogicalType::Integer),
16511                Field::new("text", LogicalType::Varchar),
16512            ],
16513        )
16514        .expect("new file");
16515        writer.append(&sample()).expect("stripe written");
16516        writer.finish().expect("commit");
16517
16518        let reader = Reader::open(&path).expect("valid directory");
16519        let dictionary = reader.table.dictionaries[1].expect("string dictionary page");
16520        // Read the count out of the page rather than writing it here, so that adding something
16521        // else to the index does not silently turn this into a test that damages the index.
16522        let mut header = [0; DICTIONARY_HEADER];
16523        read_at(&reader.file, dictionary.offset, &mut header).expect("dictionary header");
16524        // The first block's start is the first word after the offsets, since the blocks are written
16525        // during the load and are wherever the writer was when each was encoded.
16526        let count = u32::from_le_bytes(header[0..4].try_into().expect("four bytes")) as usize;
16527        let width = u32::from_le_bytes(header[12..16].try_into().expect("four bytes"));
16528        assert_ne!(width & DICTIONARY_SCATTERED, 0, "the blocks say where they are");
16529        let bits = (width & !DICTIONARY_FLAGS) as usize;
16530        let mut start = [0; 8];
16531        let at = dictionary.offset + (DICTIONARY_HEADER + offset_bytes(count, bits)) as u64;
16532        read_at(&reader.file, at, &mut start).expect("the first block's start");
16533        let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
16534        file.seek(SeekFrom::Start(u64::from_le_bytes(start))).expect("inside dictionary payload");
16535        file.write_all(&[255]).expect("damage dictionary payload");
16536
16537        let chunk = reader.read(0, &[1]).expect("code page and dictionary index remain valid");
16538        let error =
16539            chunk.validate_external().expect_err("payload corruption must reach the caller");
16540        assert!(error.message().contains("payload checksum differs"), "{error}");
16541        fs::remove_file(path).expect("remove scratch file");
16542    }
16543
16544    /// A column whose values are all different is written without a dictionary, and one whose
16545    /// values repeat keeps it.
16546    ///
16547    /// The two columns go in the same table and hold the same number of rows, so the only thing
16548    /// separating them is how much of the first stripe was a value it had not seen before. Both have
16549    /// to read back the values that were written, because the decision is about cost and nothing
16550    /// else. The file size is the other half of it: a column written without a dictionary goes
16551    /// through the string cascade instead, so dropping the dictionary must not turn into storing the
16552    /// column raw.
16553    #[test]
16554    fn a_column_of_all_different_values_is_written_without_a_dictionary() {
16555        let path = path("dictionary-decide");
16556        let rows = 20_000;
16557        // Long enough that storing it raw would show, and different in every row.
16558        let unique =
16559            |row: usize| format!("{row:09} a value that appears exactly once in the table");
16560        // The same values in the same shape, each one used forty times over.
16561        let repeated = |row: usize| unique(row / 40);
16562        let mut writer = Writer::create(
16563            &path,
16564            "items",
16565            vec![
16566                Field::required("unique", LogicalType::Varchar),
16567                Field::required("repeated", LogicalType::Varchar),
16568            ],
16569        )
16570        .expect("new file");
16571        for part in (0..rows).step_by(1_000) {
16572            let span = part..(part + 1_000).min(rows);
16573            let left = span.clone().map(|row| Value::Varchar(unique(row))).collect::<Vec<_>>();
16574            let right = span.map(|row| Value::Varchar(repeated(row))).collect::<Vec<_>>();
16575            writer
16576                .append(
16577                    &Chunk::new(vec![
16578                        Vector::from_values(LogicalType::Varchar, &left).expect("strings"),
16579                        Vector::from_values(LogicalType::Varchar, &right).expect("strings"),
16580                    ])
16581                    .expect("two columns"),
16582                )
16583                .expect("a part");
16584        }
16585        writer.finish().expect("commit");
16586
16587        let reader = Reader::open(&path).expect("reopen from disk");
16588        assert!(
16589            reader.table.dictionaries[0].is_none(),
16590            "a column with no repeats has nothing to say twice"
16591        );
16592        assert!(
16593            reader.table.dictionaries[1].is_some(),
16594            "a column whose values come round again keeps its dictionary"
16595        );
16596        let mut first = 0;
16597        for part in 0..reader.parts() {
16598            let chunk = reader.read(part, &[0, 1]).expect("a part");
16599            for row in 0..chunk.len() {
16600                assert_eq!(chunk.value_at(row, 0), Value::Varchar(unique(first + row)));
16601                assert_eq!(chunk.value_at(row, 1), Value::Varchar(repeated(first + row)));
16602            }
16603            first += chunk.len();
16604        }
16605        assert_eq!(first, rows, "every row was read back");
16606        let raw = (0..rows).map(|row| unique(row).len()).sum::<usize>();
16607        let size = fs::metadata(&path).expect("the file is there").len() as usize;
16608        assert!(size < raw, "a column without a dictionary is still encoded: {size} against {raw}");
16609        fs::remove_file(path).expect("remove scratch file");
16610    }
16611
16612    /// A payload of many blocks reads and checks every block of it.
16613    ///
16614    /// The test above has a dictionary of three values, which is one block, so it says nothing
16615    /// about a reader finding the right block among many. This one has thirty two thousand values,
16616    /// which is thirty two blocks, and it reads a value out of the first block and a value out of
16617    /// the last and then damages the last and asks for it again.
16618    ///
16619    /// Forty thousand rows over those thirty two thousand values, because a column the writer finds
16620    /// to be all distinct does not get a dictionary at all and there would be nothing here to test.
16621    /// Four rows in five holding a value the stripe has not seen before is a column that keeps one.
16622    /// The repeats are put at the front so that the values still arrive in order after them, which
16623    /// is what keeps the last part of the table on the last block of the payload.
16624    #[test]
16625    fn a_dictionary_over_many_blocks_checks_every_block_of_it() {
16626        let path = path("dictionary-blocks");
16627        let value = |row: usize| {
16628            let row = row.saturating_sub(8_000);
16629            format!("{row:07} a value long enough to be worth a payload block")
16630        };
16631        let parts = 40;
16632        let per_part = 1000;
16633        let mut writer =
16634            Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
16635                .expect("new file");
16636        for part in 0..parts {
16637            let values = (0..per_part)
16638                .map(|row| Value::Varchar(value(part * per_part + row)))
16639                .collect::<Vec<_>>();
16640            let chunk = Chunk::new(vec![
16641                Vector::from_values(LogicalType::Varchar, &values).expect("strings"),
16642            ])
16643            .expect("matching rows");
16644            writer.append(&chunk).expect("a part");
16645        }
16646        writer.finish().expect("commit");
16647
16648        let reader = Reader::open(&path).expect("reopen from disk");
16649        let dictionary = reader.table.dictionaries[0].expect("string dictionary page");
16650        assert!(
16651            parts * per_part > TEXT_PAYLOAD_VALUES * 4,
16652            "the dictionary has to be several blocks for this to be testing anything"
16653        );
16654        for part in [0, parts - 1] {
16655            let chunk = reader.read(part, &[0]).expect("a part");
16656            chunk.validate_external().expect("every payload block checks out");
16657            assert_eq!(chunk.value_at(0, 0), Value::Varchar(value(part * per_part)));
16658        }
16659
16660        // The last block is wherever the writer was when it was encoded, which the index says.
16661        let mut header = [0; DICTIONARY_HEADER];
16662        read_at(&reader.file, dictionary.offset, &mut header).expect("dictionary header");
16663        let count = u32::from_le_bytes(header[0..4].try_into().expect("four bytes")) as usize;
16664        let blocks = u32::from_le_bytes(header[8..12].try_into().expect("four bytes")) as usize;
16665        let width = u32::from_le_bytes(header[12..16].try_into().expect("four bytes"));
16666        let bits = (width & !DICTIONARY_FLAGS) as usize;
16667        let mut place = [0; 16];
16668        let at = DICTIONARY_HEADER + offset_bytes(count, bits) + (blocks - 1) * 16;
16669        read_at(&reader.file, dictionary.offset + at as u64, &mut place).expect("its place");
16670        let start = u64::from_le_bytes(place[..8].try_into().expect("eight bytes"));
16671        let length = u64::from_le_bytes(place[8..].try_into().expect("eight bytes"));
16672        let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
16673        file.seek(SeekFrom::Start(start + length - 4)).expect("the last bytes of the last block");
16674        file.write_all(&[255]).expect("damage the last payload block");
16675        let reader = Reader::open(&path).expect("the directory and the index are untouched");
16676        let chunk = reader.read(parts - 1, &[0]).expect("the code page remains valid");
16677        let error = chunk.validate_external().expect_err("the damage must reach the caller");
16678        assert!(error.message().contains("payload checksum differs"), "{error}");
16679        fs::remove_file(path).expect("remove scratch file");
16680    }
16681
16682    /// Values of different lengths read back where the offsets say they do.
16683    ///
16684    /// The offsets are packed at one width for the column, they are relative to the payload block a
16685    /// value lands in, and they go in runs of half a block, so there are two boundaries where the
16686    /// arithmetic could be off by one and neither shows up on values that are all the same length.
16687    /// This writes 5,000 values whose lengths cycle through a wide range and reads every one back,
16688    /// so the first value of a block, the last value of a run and the last value of a block are all
16689    /// covered several times over. An empty value is in the cycle because a zero length span is the
16690    /// case the reader short circuits.
16691    ///
16692    /// Six thousand rows over those 5,000 values, because a column the writer finds to be all
16693    /// distinct is written without a dictionary and then there are no packed offsets to be off by
16694    /// one in.
16695    #[test]
16696    fn values_of_different_lengths_read_back_out_of_packed_offsets() {
16697        let path = path("dictionary-offsets");
16698        let value = |row: usize| {
16699            let row = row % 5_000;
16700            if row % 511 == 3 { String::new() } else { "x".repeat(row % 97) + &format!("{row:05}") }
16701        };
16702        let rows = 6_000;
16703        let mut writer =
16704            Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
16705                .expect("new file");
16706        let values = (0..rows).map(|row| Value::Varchar(value(row))).collect::<Vec<_>>();
16707        for part in values.chunks(1_000) {
16708            let chunk =
16709                Chunk::new(vec![Vector::from_values(LogicalType::Varchar, part).expect("strings")])
16710                    .expect("matching rows");
16711            writer.append(&chunk).expect("a part");
16712        }
16713        writer.finish().expect("commit");
16714
16715        let reader = Reader::open(&path).expect("reopen from disk");
16716        assert!(
16717            rows > TEXT_PAYLOAD_VALUES * 4,
16718            "the dictionary has to be several blocks for this to be testing anything"
16719        );
16720        for part in 0..rows / 1_000 {
16721            let chunk = reader.read(part, &[0]).expect("a part");
16722            for row in 0..1_000 {
16723                let row = part * 1_000 + row;
16724                assert_eq!(
16725                    chunk.value_at(row % 1_000, 0),
16726                    Value::Varchar(value(row)),
16727                    "value {row}"
16728                );
16729            }
16730        }
16731        // The lengths a vector at a time, twice over, because the first pass is what makes the
16732        // table of ends worth building and the second is read out of the lengths worked out of it.
16733        for _ in 0..2 {
16734            for part in 0..rows / 1_000 {
16735                let chunk = reader.read(part, &[0]).expect("a part");
16736                let mut lens = vec![0_i64; 1_000];
16737                let column = chunk.column(0).expect("one column");
16738                assert!(column.try_bytes_lens(&mut lens).expect("lengths"), "a stored column");
16739                for (row, &len) in lens.iter().enumerate() {
16740                    let row = part * 1_000 + row;
16741                    assert_eq!(len as usize, value(row).len(), "the length of value {row}");
16742                }
16743            }
16744        }
16745        fs::remove_file(path).expect("remove scratch file");
16746    }
16747
16748    /// Lengths start again at every block, and ends that go backwards inside one give no table.
16749    #[test]
16750    fn lengths_restart_at_each_block_and_refuse_ends_that_go_backwards() {
16751        let mut ends: Vec<u32> = (1..=TEXT_PAYLOAD_VALUES as u32).map(|at| at * 2).collect();
16752        ends.extend([3, 3, 10]);
16753        let Some(Lengths::Narrow(lens)) = lengths_of(&ends) else { panic!("short ordered ends") };
16754        assert!(lens[..TEXT_PAYLOAD_VALUES].iter().all(|&len| len == 2));
16755        assert_eq!(&lens[TEXT_PAYLOAD_VALUES..], &[3, 0, 7]);
16756        // One value longer than sixteen bits keeps every length at four bytes.
16757        let long = [5, 70_005, 70_006];
16758        let Some(Lengths::Wide(lens)) = lengths_of(&long) else { panic!("long ordered ends") };
16759        assert_eq!(lens, [5, 70_000, 1]);
16760        let mut read = Vec::new();
16761        Lengths::Wide(lens).extend_at(&[1, 9, 0], &mut read);
16762        assert_eq!(read, [70_000, 0, 5], "a position past the end is no length");
16763        ends.push(9);
16764        assert!(lengths_of(&ends).is_none());
16765    }
16766
16767    /// Every worker of a scan wants the dictionary at the same moment and one of them fetches it.
16768    ///
16769    /// Asking a `OnceLock` whether it holds something answers the question a worker that already has
16770    /// the dictionary is asking and not the one a worker without it is asking, which is whether
16771    /// somebody is already on their way with it. Sixteen workers that all miss will all read the
16772    /// page, all verify it and all decode it, and fifteen will drop the result. Nothing about that
16773    /// is incorrect, which is why it went unnoticed, and it showed up as ClickBench 38 getting
16774    /// slower when the scan in front of it got faster and stopped staggering the arrivals.
16775    ///
16776    /// The barrier is what makes the test about that rather than about luck. Without it the first
16777    /// thread is usually finished before the last one starts and the count is one either way.
16778    #[test]
16779    fn a_global_dictionary_is_opened_once_however_many_workers_ask_at_once() {
16780        let path = path("dictionary-once");
16781        let parts = 8;
16782        let per_part = 500;
16783        let value =
16784            |row: usize| format!("{row:07} a value long enough to be worth a payload block");
16785        let mut writer =
16786            Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
16787                .expect("new file");
16788        for part in 0..parts {
16789            let values = (0..per_part)
16790                .map(|row| Value::Varchar(value(part * per_part + row)))
16791                .collect::<Vec<_>>();
16792            let chunk = Chunk::new(vec![
16793                Vector::from_values(LogicalType::Varchar, &values).expect("strings"),
16794            ])
16795            .expect("matching rows");
16796            writer.append(&chunk).expect("a part");
16797        }
16798        writer.finish().expect("commit");
16799
16800        let reader = Reader::open(&path).expect("reopen from disk");
16801        assert!(reader.table.dictionaries[0].is_some(), "the column has to have one to share");
16802        assert_eq!(reader.reads().dictionaries, 0, "opening the file does not open a dictionary");
16803
16804        let workers = 16;
16805        let gate = std::sync::Barrier::new(workers);
16806        std::thread::scope(|scope| {
16807            for worker in 0..workers {
16808                let reader = reader.clone();
16809                let gate = &gate;
16810                scope.spawn(move || {
16811                    gate.wait();
16812                    let chunk = reader.read(worker % parts, &[0]).expect("a part");
16813                    assert_eq!(
16814                        chunk.value_at(0, 0),
16815                        Value::Varchar(value((worker % parts) * per_part))
16816                    );
16817                });
16818            }
16819        });
16820
16821        assert_eq!(reader.reads().dictionaries, 1, "sixteen workers, one dictionary, one open");
16822        fs::remove_file(path).expect("remove scratch file");
16823    }
16824
16825    /// The sorted order sits outside the index the page checksum covers, because a query that
16826    /// never searches a dictionary should not read it, so it carries its own checksums and this is
16827    /// what says they are checked. A search that trusted a damaged order would give a wrong answer
16828    /// rather than a slow one.
16829    #[test]
16830    fn a_damaged_sorted_order_is_an_error() {
16831        let path = path("damaged-order");
16832        let mut writer = Writer::create(
16833            &path,
16834            "items",
16835            vec![
16836                Field::required("id", LogicalType::Integer),
16837                Field::new("text", LogicalType::Varchar),
16838            ],
16839        )
16840        .expect("new file");
16841        writer.append(&sample()).expect("stripe written");
16842        writer.finish().expect("commit");
16843
16844        let reader = Reader::open(&path).expect("valid directory");
16845        let page = reader.table.dictionaries[1].expect("string dictionary page");
16846        let mut header = [0; DICTIONARY_HEADER];
16847        read_at(&reader.file, page.offset, &mut header).expect("dictionary header");
16848        let index_len = dictionary_index_len(&header);
16849        let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
16850        file.seek(SeekFrom::Start(page.offset + index_len)).expect("the first head");
16851        file.write_all(&[255]).expect("damage the order");
16852
16853        let dictionary = reader.dictionary(1).expect("read").expect("a string column has one");
16854        let error = dictionary.compare_rank(0, b"anything").expect_err("a damaged order is caught");
16855        assert!(error.message().contains("rank checksum differs"), "{error}");
16856        fs::remove_file(path).expect("remove scratch file");
16857    }
16858
16859    /// Codes stay in first appearance order and the sorted order is written beside them, so a
16860    /// reader can put the values back in order without the writer having had to know them all
16861    /// before it handed out the first code.
16862    #[test]
16863    fn a_global_dictionary_carries_the_sorted_order_of_its_values() {
16864        // Chosen so the sort cannot be decided on the first eight bytes alone. Three values share
16865        // a nine byte prefix, one is a prefix of another, and one is empty.
16866        let spellings = ["overlong1z", "b", "", "overlong1a", "overlong", "ab", "a", "overlong1"];
16867        let path = path("dictionary-order");
16868        let mut writer =
16869            Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
16870                .expect("new file");
16871        writer
16872            .append(
16873                &Chunk::new(vec![
16874                    Vector::from_values(
16875                        LogicalType::Varchar,
16876                        &spellings.map(|text| Value::Varchar(text.into())),
16877                    )
16878                    .expect("strings"),
16879                ])
16880                .expect("one column"),
16881            )
16882            .expect("stripe written");
16883        writer.finish().expect("commit");
16884
16885        let reader = Reader::open(&path).expect("valid directory");
16886        let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
16887        let count = dictionary.ranks().expect("a v10 file stores one");
16888        assert_eq!(count, spellings.len(), "every distinct value has a rank");
16889        let order = (0..count)
16890            .map(|rank| dictionary.code_at_rank(rank).expect("a code"))
16891            .collect::<Vec<_>>();
16892        let mut seen = order.clone();
16893        seen.sort_unstable();
16894        assert_eq!(seen, (0..spellings.len() as u32).collect::<Vec<_>>(), "a permutation of codes");
16895
16896        let ranked = order
16897            .iter()
16898            .map(|&code| {
16899                dictionary.try_bytes_at(code as usize).expect("read").expect("a value").to_vec()
16900            })
16901            .collect::<Vec<_>>();
16902        let mut expected = spellings.map(|text| text.as_bytes().to_vec()).to_vec();
16903        expected.sort();
16904        assert_eq!(ranked, expected, "rank order is value order");
16905
16906        // What a search asks, on the values themselves rather than through a kernel, so that a
16907        // file whose heads disagree with its bytes is caught here rather than as a wrong answer.
16908        for (rank, value) in expected.iter().enumerate() {
16909            assert_eq!(
16910                dictionary.compare_rank(rank, value).expect("compare"),
16911                Ordering::Equal,
16912                "rank {rank} is its own value"
16913            );
16914            if rank > 0 {
16915                assert_eq!(
16916                    dictionary.compare_rank(rank - 1, value).expect("compare"),
16917                    Ordering::Less,
16918                    "rank {rank} follows the one before it"
16919                );
16920            }
16921        }
16922        fs::remove_file(path).expect("remove scratch file");
16923    }
16924
16925    /// Five text columns of different sizes close at the same time, and each comes back with its
16926    /// own values in its own order.
16927    ///
16928    /// The sizes differ so that the columns are taken in an order that is not the column order, and
16929    /// the values of each column are spelled with its number so that one column's page written in
16930    /// another's place would read back as the wrong strings rather than the right ones by chance.
16931    #[test]
16932    fn text_columns_closed_at_once_each_keep_their_own_dictionary() {
16933        let sizes = [300_usize, 5_000, 40, 2_000, 1_200];
16934        let path = path("dictionaries-at-once");
16935        let fields = (0..sizes.len())
16936            .map(|column| Field::new(format!("text{column}"), LogicalType::Varchar))
16937            .collect::<Vec<_>>();
16938        let mut writer = Writer::create(&path, "items", fields).expect("new file");
16939        let rows = 10_000_usize;
16940        for start in (0..rows).step_by(1_024) {
16941            let columns = sizes
16942                .iter()
16943                .enumerate()
16944                .map(|(column, &size)| {
16945                    let values = (start..(start + 1_024).min(rows))
16946                        .map(|row| Value::Varchar(format!("c{column}-{:05}", (row * 7919) % size)))
16947                        .collect::<Vec<_>>();
16948                    Vector::from_values(LogicalType::Varchar, &values).expect("strings")
16949                })
16950                .collect::<Vec<_>>();
16951            writer.append(&Chunk::new(columns).expect("five columns")).expect("stripe written");
16952        }
16953        writer.finish().expect("commit");
16954
16955        let reader = Reader::open(&path).expect("valid directory");
16956        for (column, &size) in sizes.iter().enumerate() {
16957            let dictionary =
16958                reader.dictionary(column).expect("read").expect("a string column has one");
16959            let count = dictionary.ranks().expect("a v10 file stores one");
16960            assert_eq!(count, size, "column {column} has its own distinct count");
16961            let ranked = (0..count)
16962                .map(|rank| {
16963                    let code = dictionary.code_at_rank(rank).expect("a code");
16964                    dictionary.try_bytes_at(code as usize).expect("read").expect("a value").to_vec()
16965                })
16966                .collect::<Vec<_>>();
16967            let expected = (0..size)
16968                .map(|value| format!("c{column}-{value:05}").into_bytes())
16969                .collect::<Vec<_>>();
16970            assert_eq!(ranked, expected, "column {column} ranks its own values in order");
16971        }
16972        fs::remove_file(path).expect("remove scratch file");
16973    }
16974
16975    /// A dictionary large enough to be decoded and sorted on several threads ranks the way one small
16976    /// enough for one thread does.
16977    ///
16978    /// Seventy thousand values over sixty nine blocks, in no order and each four times over so the
16979    /// column is worth a dictionary, written and ranked in the close.
16980    /// Some share a long prefix and some differ only in the last byte, so the buckets of the sort cut
16981    /// through runs of values that agree for a long way.
16982    #[test]
16983    fn a_large_dictionary_ranks_in_value_order() {
16984        let path = path("dictionary-large-rank");
16985        let value = |row: u64| {
16986            let mixed = row.wrapping_mul(0x9e37_79b9_7f4a_7c15) >> 40;
16987            match row % 3 {
16988                0 => format!("https://example.com/a/long/shared/path/{mixed:08}"),
16989                1 => format!("{mixed}"),
16990                _ => format!("x{}", row % 1000).repeat(1 + (row % 4) as usize) + &row.to_string(),
16991            }
16992        };
16993        let distinct = 70_000;
16994        let parts = 4 * distinct / 1000;
16995        let per_part = 1000;
16996        let mut writer =
16997            Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
16998                .expect("new file");
16999        for part in 0..parts {
17000            let values = (0..per_part)
17001                .map(|row| Value::Varchar(value((part * per_part + row) / 4)))
17002                .collect::<Vec<_>>();
17003            let chunk = Chunk::new(vec![
17004                Vector::from_values(LogicalType::Varchar, &values).expect("strings"),
17005            ])
17006            .expect("matching rows");
17007            writer.append(&chunk).expect("a part");
17008        }
17009        writer.finish().expect("commit");
17010
17011        let reader = Reader::open(&path).expect("reopen from disk");
17012        let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
17013        let count = dictionary.ranks().expect("a ranked dictionary");
17014        assert_eq!(count, distinct as usize, "every distinct value has a rank");
17015        assert!(count >= PARALLEL_SORT_MIN, "too few values to be sorted on more than one thread");
17016        let ranked = (0..count)
17017            .map(|rank| {
17018                let code = dictionary.code_at_rank(rank).expect("a code");
17019                dictionary.try_bytes_at(code as usize).expect("read").expect("a value").to_vec()
17020            })
17021            .collect::<Vec<_>>();
17022        let mut expected = (0..distinct).map(|row| value(row).into_bytes()).collect::<Vec<_>>();
17023        expected.sort();
17024        assert_eq!(ranked, expected, "rank order is value order");
17025        fs::remove_file(path).expect("remove scratch file");
17026    }
17027
17028    /// A string column's synopsis is turned into values without keeping the blocks it went through.
17029    ///
17030    /// Three thousand values, every fifth of them four times over, so the synopsis is a prefix of
17031    /// five hundred and twelve codes spread over all three payload blocks. Reading it used to leave
17032    /// all three decoded for as long as the reader lived. It leaves none of them now, and the second
17033    /// read answers out of what the first remembered.
17034    /// A directory read out of the file a window at a time is the directory read whole.
17035    ///
17036    /// The windows here are far smaller than any field is long, so every kind of field is split
17037    /// across a refill somewhere, and a bound is offered to its codec short more than once. The
17038    /// synopses are left in the file, and each one read back from where it was left is the one the
17039    /// whole read decoded.
17040    #[test]
17041    fn a_directory_read_a_window_at_a_time_is_the_directory_read_whole() {
17042        let path = path("windowed-directory");
17043        let fields = vec![
17044            Field::required("id", LogicalType::BigInt),
17045            Field::required("word", LogicalType::Varchar),
17046            Field::new("score", LogicalType::Double),
17047        ];
17048        let mut writer = Writer::create(&path, "items", fields).expect("new file");
17049        for part in 0..70_i64 {
17050            let ids = (0..100).map(|row| Value::BigInt(part * 100 + row % 7)).collect::<Vec<_>>();
17051            let words = (0..100)
17052                .map(|row| Value::Varchar(format!("word {}", row % 13)))
17053                .collect::<Vec<_>>();
17054            let scores = (0..100)
17055                .map(|row| if row % 4 == 0 { Value::Null } else { Value::Double(row as f64) })
17056                .collect::<Vec<_>>();
17057            let chunk = Chunk::new(vec![
17058                Vector::from_values(LogicalType::BigInt, &ids).expect("integers"),
17059                Vector::from_values(LogicalType::Varchar, &words).expect("strings"),
17060                Vector::from_values(LogicalType::Double, &scores).expect("doubles"),
17061            ])
17062            .expect("three columns");
17063            writer.append(&chunk).expect("a part");
17064        }
17065        writer.finish().expect("commit");
17066
17067        let catalog = Catalog::open(&path).expect("reopen");
17068        let entry = catalog.entries.first().expect("one table").directory;
17069        let (offset, length) = (entry.offset, entry.length as usize);
17070        let mut bytes = vec![0; length];
17071        read_at(&catalog.file, offset, &mut bytes).expect("the directory");
17072        assert_eq!(file_checksum(&catalog.file, offset, length).expect("checksum"), entry.hash);
17073        let whole = decode_directory(&bytes, catalog.size).expect("whole");
17074        assert!(whole.stripes.len() > 1, "the table should span stripes");
17075        for size in [1, 7, 33, 4_096] {
17076            let mut cursor = Cursor::over(&catalog.file, offset, length);
17077            cursor.window.as_mut().expect("a window").size = size;
17078            let windowed = read_directory(cursor, catalog.size, Some(offset)).expect("windowed");
17079            assert_eq!(format!("{:?}", windowed.stripes), format!("{:?}", whole.stripes));
17080            assert_eq!(format!("{:?}", windowed.fields), format!("{:?}", whole.fields));
17081            let mut stored = 0;
17082            for (column, (left, held)) in
17083                windowed.frequencies.iter().zip(&whole.frequencies).enumerate()
17084            {
17085                match (left, held) {
17086                    (None, None) => {}
17087                    (
17088                        Some(super::Frequencies::Stored { span, values, entries }),
17089                        Some(super::Frequencies::Held(summary)),
17090                    ) => {
17091                        let mut one = vec![0; span.length as usize];
17092                        read_at(&catalog.file, span.offset, &mut one).expect("a synopsis");
17093                        let read = decode_summary(
17094                            &mut Cursor::new(&one),
17095                            &whole.fields[column],
17096                            whole.rows,
17097                            *values,
17098                        )
17099                        .expect("a valid synopsis")
17100                        .expect("one is there");
17101                        assert_eq!(*entries, read.entries.len());
17102                        assert_eq!(format!("{read:?}"), format!("{summary:?}"));
17103                        stored += 1;
17104                    }
17105                    other => panic!("column {column} came back as {other:?}"),
17106                }
17107            }
17108            assert!(stored >= 2, "only {stored} synopses were left in the file");
17109        }
17110        let reader = catalog.table("items").expect("the table");
17111        assert!(reader.frequency_heads[1].get().is_none());
17112        assert!(reader.top_frequencies(1, 1).expect("a readable synopsis").is_some());
17113        let first = reader.frequency_heads[1].get().expect("decoded synopsis");
17114        let clone = reader.clone();
17115        assert!(clone.top_frequencies(1, 1).expect("cached synopsis").is_some());
17116        assert!(Arc::ptr_eq(first, clone.frequency_heads[1].get().expect("same synopsis")));
17117        fs::remove_file(path).expect("remove scratch file");
17118    }
17119
17120    #[test]
17121    fn a_checksum_carried_across_reads_is_the_checksum_of_the_whole() {
17122        let path = path("file-checksum");
17123        let bytes = (0..200_000_u32)
17124            .map(|at| (at.wrapping_mul(2_654_435_761) >> 13) as u8)
17125            .collect::<Vec<_>>();
17126        fs::write(&path, &bytes).expect("scratch file");
17127        let file = File::open(&path).expect("open");
17128        for (offset, length) in [
17129            (0, 0),
17130            (3, 1),
17131            (5, 31),
17132            (0, 32),
17133            (9, 33),
17134            (1, 65_536),
17135            (7, 65_567),
17136            (0, 200_000),
17137            (11, 131_101),
17138        ] {
17139            let whole = checksum(&bytes[offset..offset + length]);
17140            assert_eq!(
17141                file_checksum(&file, offset as u64, length).expect("read"),
17142                whole,
17143                "{offset} {length}"
17144            );
17145        }
17146        fs::remove_file(path).expect("remove scratch file");
17147    }
17148
17149    #[test]
17150    fn a_string_synopsis_is_read_without_keeping_the_dictionary_blocks() {
17151        let path = path("synopsis-keeps-no-block");
17152        let spelled = |index: usize| Value::Varchar(format!("phrase {index:05}"));
17153        let mut values = (0..3_000).map(spelled).collect::<Vec<_>>();
17154        for _ in 0..3 {
17155            values.extend((0..3_000).step_by(5).map(spelled));
17156        }
17157        let mut writer =
17158            Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
17159                .expect("new file");
17160        for part in values.chunks(1_024) {
17161            writer
17162                .append(
17163                    &Chunk::new(vec![
17164                        Vector::from_values(LogicalType::Varchar, part).expect("strings"),
17165                    ])
17166                    .expect("one column"),
17167                )
17168                .expect("a part");
17169        }
17170        writer.finish().expect("commit");
17171
17172        let reader = Reader::open(&path).expect("reopen from disk");
17173        let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
17174        let resting = dictionary.footprint();
17175        let prefix = reader.frequency_prefix(0).expect("a readable synopsis").expect("one");
17176        assert_eq!(prefix.entries.len(), 512);
17177        for (value, count) in &prefix.entries {
17178            let Value::Varchar(text) = value else { panic!("a string column gave {value:?}") };
17179            let index = text["phrase ".len()..].parse::<usize>().expect("a spelled number");
17180            assert_eq!((index % 5, *count), (0, 4), "{text} came back with {count}");
17181        }
17182        assert_eq!(dictionary.footprint(), resting, "reading the synopsis kept a decoded block");
17183        let again = reader.frequency_prefix(0).expect("a readable synopsis").expect("one");
17184        assert_eq!(again.entries, prefix.entries);
17185        fs::remove_file(path).expect("remove scratch file");
17186    }
17187
17188    /// `length` over a stored column keeps a count a value rather than the blocks it counted.
17189    ///
17190    /// Reading the bytes a row at a time keeps every block it touches, so a scan of `length` over a
17191    /// whole column used to end up holding the column decoded. The counts are what is kept now, and
17192    /// they have to be the counts of characters rather than bytes, which is why the values here are
17193    /// not ASCII.
17194    #[test]
17195    fn character_lengths_are_counted_without_keeping_the_dictionary_blocks() {
17196        let path = path("character-lengths");
17197        let spellings = (0..2_500)
17198            .map(|index| Value::Varchar(format!("héllo {index:05} {}", "ü".repeat(index % 30))))
17199            .collect::<Vec<_>>();
17200        let mut writer =
17201            Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
17202                .expect("new file");
17203        for part in spellings.chunks(1_024) {
17204            writer
17205                .append(
17206                    &Chunk::new(vec![
17207                        Vector::from_values(LogicalType::Varchar, part).expect("strings"),
17208                    ])
17209                    .expect("one column"),
17210                )
17211                .expect("a part");
17212        }
17213        writer.finish().expect("commit");
17214
17215        let reader = Reader::open(&path).expect("reopen from disk");
17216        let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
17217        let resting = dictionary.footprint();
17218        let mut lens = Vec::new();
17219        assert!(dictionary.try_chars_lens(&mut lens).expect("counted"), "a stored source counts");
17220        let counted = dictionary.footprint() - resting;
17221        let blocks = dictionary.len().div_ceil(TEXT_PAYLOAD_VALUES);
17222        assert!(
17223            counted <= blocks * TEXT_PAYLOAD_VALUES * size_of::<u32>(),
17224            "counting kept {counted} bytes, more than a count a value"
17225        );
17226        let expected = (0..dictionary.len())
17227            .map(|code| {
17228                let bytes = dictionary.try_bytes_at(code).expect("read").expect("a value");
17229                i64::try_from(std::str::from_utf8(bytes).expect("utf-8").chars().count())
17230                    .expect("small")
17231            })
17232            .collect::<Vec<_>>();
17233        assert_eq!(lens, expected, "a count is the number of characters, not of bytes");
17234        let mut again = Vec::new();
17235        assert!(dictionary.try_chars_lens(&mut again).expect("counted"));
17236        assert_eq!(again, lens, "the kept counts answer the second time");
17237        fs::remove_file(path).expect("remove scratch file");
17238    }
17239
17240    /// Writes one column of strings whose code is where they sit in `spellings`, and reopens it.
17241    fn stored_spellings(label: &str, spellings: &[String]) -> (PathBuf, Reader) {
17242        let path = path(label);
17243        let values = spellings.iter().map(|text| Value::Varchar(text.clone())).collect::<Vec<_>>();
17244        let mut writer =
17245            Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
17246                .expect("new file");
17247        for part in values.chunks(1_024) {
17248            writer
17249                .append(
17250                    &Chunk::new(vec![
17251                        Vector::from_values(LogicalType::Varchar, part).expect("strings"),
17252                    ])
17253                    .expect("one column"),
17254                )
17255                .expect("a part");
17256        }
17257        writer.finish().expect("commit");
17258        let reader = Reader::open(&path).expect("reopen from disk");
17259        (path, reader)
17260    }
17261
17262    /// Codes that go all over a dictionary of `len` values, and every seventh row null.
17263    ///
17264    /// The shape of a vector a scan hands out: its codes are in row order, which lands them in
17265    /// every block of the dictionary in no order at all, so a read of the whole vector has to put
17266    /// them in block order itself to read each block once.
17267    fn scattered_rows(len: usize) -> (Vec<u32>, Vec<bool>) {
17268        let codes = (0..len)
17269            .map(|row| u32::try_from(row * 7_919 % len).expect("a small dictionary"))
17270            .collect::<Vec<_>>();
17271        let valid = (0..len).map(|row| row % 7 != 3).collect::<Vec<_>>();
17272        (codes, valid)
17273    }
17274
17275    /// `length` over a vector with nulls keeps the counts and not the blocks, the same as over one
17276    /// without.
17277    ///
17278    /// The whole vector count used to be taken only when no row was null, and every other vector
17279    /// went a row at a time through the bytes, which keeps every block it reads. A column with a
17280    /// null in each vector was held decoded after one `length` over it.
17281    #[test]
17282    fn character_lengths_with_nulls_are_counted_without_keeping_the_dictionary_blocks() {
17283        let spellings = (0..2_500)
17284            .map(|index| format!("héllo {index:05} {}", "ü".repeat(index % 30)))
17285            .collect::<Vec<_>>();
17286        let (path, reader) = stored_spellings("character-lengths-nulls", &spellings);
17287        let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
17288        let (codes, valid) = scattered_rows(spellings.len());
17289        let rows = Vector::dictionary_over(codes.clone(), Arc::clone(&dictionary))
17290            .expect("every code is inside")
17291            .with_validity(Validity::from_run(&valid));
17292
17293        let resting = dictionary.footprint();
17294        let lens = rudb_kernels::call("length", &[&rows], &LogicalType::BigInt, None)
17295            .expect("length reads");
17296        let counted = dictionary.footprint() - resting;
17297        let blocks = dictionary.len().div_ceil(TEXT_PAYLOAD_VALUES);
17298        assert!(
17299            counted <= blocks * TEXT_PAYLOAD_VALUES * size_of::<u32>(),
17300            "length over a vector with nulls kept {counted} bytes, more than a count a value"
17301        );
17302        let expected = (0..rows.len())
17303            .map(|row| match valid[row] {
17304                true => Value::BigInt(
17305                    i64::try_from(spellings[codes[row] as usize].chars().count()).expect("small"),
17306                ),
17307                false => Value::Null,
17308            })
17309            .collect::<Vec<_>>();
17310        let answers = (0..lens.len()).map(|row| lens.value_at(row)).collect::<Vec<_>>();
17311        assert_eq!(answers, expected, "a count of characters where a row has one, null elsewhere");
17312        fs::remove_file(path).expect("remove scratch file");
17313    }
17314
17315    /// `lower`, `upper` and `substring` read a stored dictionary a block at a time and keep none of
17316    /// it while the column is at its budget, until reading without keeping stops being cheap.
17317    ///
17318    /// The three used to read a row at a time through the bytes, which keeps every block a row lands
17319    /// in for as long as the table is open. They read the whole vector in one visit now, and the
17320    /// dictionary here is opened with a budget of zero so that what a visit would keep under the
17321    /// budget of a running database is what the test sees dropped. After a column's worth of blocks
17322    /// has been decoded and dropped the visit keeps what it reads, which is what bounds its cost on
17323    /// a scan whose codes keep coming back to every block, and the end of the test holds it to that.
17324    #[test]
17325    fn string_kernels_read_a_stored_dictionary_without_keeping_its_blocks() {
17326        let spellings = (0..2_500)
17327            .map(|index| format!("HéLLo {index:05} {}", "Üß".repeat(index % 30)))
17328            .collect::<Vec<_>>();
17329        let (path, reader) = stored_spellings("string-kernels", &spellings);
17330        let page = reader.table.dictionaries[0].expect("a string column has one");
17331        let starved =
17332            open_global_dictionary(Arc::clone(&reader.file), page, &LogicalType::Varchar, 0)
17333                .expect("a dictionary opens whatever it may keep");
17334        let starved = Arc::new(starved);
17335        let (codes, valid) = scattered_rows(spellings.len());
17336        let rows = Vector::dictionary_over(codes.clone(), Arc::clone(&starved))
17337            .expect("every code is inside")
17338            .with_validity(Validity::from_run(&valid));
17339        let expected = |each: &dyn Fn(&str) -> String| {
17340            (0..rows.len())
17341                .map(|row| match valid[row] {
17342                    true => Value::Varchar(each(&spellings[codes[row] as usize])),
17343                    false => Value::Null,
17344                })
17345                .collect::<Vec<_>>()
17346        };
17347        let answers =
17348            |vector: &Vector| (0..vector.len()).map(|row| vector.value_at(row)).collect::<Vec<_>>();
17349
17350        // What a visit may add is the table of where every value ends, four bytes a value, which
17351        // reading every value this often makes worth building. A block is tens of bytes a value.
17352        let resting = starved.footprint();
17353        let ends = spellings.len() * size_of::<u32>();
17354        let lowered = rudb_kernels::call("lower", &[&rows], &LogicalType::Varchar, None)
17355            .expect("lower reads");
17356        assert_eq!(answers(&lowered), expected(&|text| text.to_lowercase()), "lower");
17357        assert!(starved.footprint() <= resting + ends, "lower kept a block it read");
17358
17359        let start = Vector::constant(LogicalType::BigInt, Value::BigInt(3), rows.len());
17360        let length = Vector::constant(LogicalType::BigInt, Value::BigInt(9), rows.len());
17361        let cut =
17362            rudb_kernels::call("substring", &[&rows, &start, &length], &LogicalType::Varchar, None)
17363                .expect("substring reads");
17364        let cut_of = |text: &str| text.chars().skip(2).take(9).collect::<String>();
17365        assert_eq!(answers(&cut), expected(&cut_of), "substring");
17366        assert!(starved.footprint() <= resting + ends, "substring kept a block it read");
17367
17368        // Every block has been read twice now and dropped the second time as well, which is a
17369        // column's worth dropped for want of a budget, so the next visit keeps what it reads.
17370        let raised = rudb_kernels::call("upper", &[&rows], &LogicalType::Varchar, None)
17371            .expect("upper reads");
17372        assert_eq!(answers(&raised), expected(&|text| text.to_uppercase()), "upper");
17373        let payload = spellings.iter().map(String::len).sum::<usize>();
17374        assert!(
17375            starved.footprint() >= resting + payload,
17376            "a visit that has dropped a column's worth of blocks keeps what it reads"
17377        );
17378        let again = rudb_kernels::call("upper", &[&rows], &LogicalType::Varchar, None)
17379            .expect("upper reads kept blocks");
17380        assert_eq!(answers(&again), answers(&raised), "the kept blocks answer the same");
17381        fs::remove_file(path).expect("remove scratch file");
17382    }
17383
17384    /// A sweep of the dictionary reads every value, and the second sweep keeps what it read, up to
17385    /// the budget.
17386    ///
17387    /// The point of the sweep is the resident size rather than the answer, so both are checked
17388    /// here. The first sweep keeps nothing, because a process that runs one statement never reads
17389    /// a block twice. A dictionary this small is well under [`TEXT_KEEP_BUDGET`], so the second
17390    /// sweep keeps everything and a third decodes nothing, which is what makes a session asking the
17391    /// same question again cost what it should. The ceiling is the other half of it and it has its own
17392    /// test below, because a ceiling that never binds is not a ceiling anybody checked.
17393    #[test]
17394    fn a_dictionary_sweep_reads_every_value_and_keeps_it_under_the_budget() {
17395        let path = path("dictionary-sweep");
17396        // Two thousand five hundred distinct values is two whole payload blocks and a part of a
17397        // third, so the sweep has to be called more than once and the last call has to stop short.
17398        let spellings = (0..2_500)
17399            .map(|index| Value::Varchar(format!("value {index:08} {}", "x".repeat(index % 40))))
17400            .collect::<Vec<_>>();
17401        let mut writer =
17402            Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
17403                .expect("new file");
17404        // A chunk is a part and a part is at most 1,024 rows, so the values go in three of them.
17405        // The dictionary is table wide and does not care where a value was written.
17406        for part in spellings.chunks(1_024) {
17407            writer
17408                .append(
17409                    &Chunk::new(vec![
17410                        Vector::from_values(LogicalType::Varchar, part).expect("strings"),
17411                    ])
17412                    .expect("one column"),
17413                )
17414                .expect("stripe written");
17415        }
17416        writer.finish().expect("commit");
17417
17418        let reader = Reader::open(&path).expect("valid directory");
17419        let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
17420        assert_eq!(dictionary.len(), spellings.len(), "every value is distinct");
17421        for first in [0, TEXT_PAYLOAD_VALUES, TEXT_PAYLOAD_VALUES * 2] {
17422            assert!(dictionary.text_block_might_contain(first, b"value").expect("signature"));
17423            assert!(!dictionary.text_block_might_contain(first, b"google").expect("signature"));
17424        }
17425
17426        let resting = dictionary.footprint();
17427        let sweep = || {
17428            let mut swept: Vec<Vec<u8>> = Vec::new();
17429            let mut at = 0;
17430            let mut calls = 0;
17431            while at < dictionary.len() {
17432                let stopped = dictionary
17433                    .sweep_text(at, dictionary.len(), &mut |index: usize, text: &[u8]| {
17434                        assert_eq!(index, swept.len(), "a sweep hands its values over in order");
17435                        swept.push(text.to_vec());
17436                        Ok(())
17437                    })
17438                    .expect("a sweep reads");
17439                assert!(stopped > at, "a sweep moves");
17440                at = stopped;
17441                calls += 1;
17442            }
17443            assert_eq!(calls, 3, "a sweep hands over one block at a time");
17444            swept
17445        };
17446        let swept = sweep();
17447        assert_eq!(dictionary.footprint(), resting, "a first sweep keeps nothing it decoded");
17448        assert_eq!(sweep(), swept, "a second sweep reads what the first did");
17449        let after = dictionary.footprint();
17450        assert!(after > resting, "a second sweep under the budget keeps what it decoded");
17451
17452        let read = (0..dictionary.len())
17453            .map(|code| dictionary.try_bytes_at(code).expect("read").expect("a value").to_vec())
17454            .collect::<Vec<_>>();
17455        assert_eq!(swept, read, "a sweep answers what a point read answers");
17456        // A read per value is about what makes the unpacked ends worth building, so whether they
17457        // are built here depends on how many reads the sweep made on the way. They are the one thing
17458        // allowed to grow, by four bytes a value, and nothing of the payload is.
17459        let grown = dictionary.footprint() - after;
17460        assert!(
17461            grown == 0 || grown == dictionary.len() * size_of::<u32>(),
17462            "a point read of a kept block decodes nothing, and {grown} bytes grew"
17463        );
17464        fs::remove_file(path).expect("remove scratch file");
17465    }
17466
17467    #[test]
17468    fn a_narrow_signature_of_an_older_file_answers_by_its_own_width() {
17469        let path = path("narrow-substring-signature");
17470        let blocks = [&b"https://google.com/"[..], b"https://example.org/", b"mail.google.com"];
17471        let mut grams = Vec::new();
17472        for text in blocks {
17473            let mut bits = vec![0_u8; NARROW_GRAM_BYTES];
17474            for gram in text.windows(4) {
17475                for bit in gram_bits(gram, NARROW_GRAM_BYTES) {
17476                    bits[bit / 8] |= 1 << (bit % 8);
17477                }
17478            }
17479            grams.extend(bits);
17480        }
17481        fs::write(&path, &grams).expect("scratch file");
17482        let file = File::open(&path).expect("open scratch file");
17483        let signatures = NativeGrams {
17484            start: 0,
17485            length: grams.len(),
17486            width: NARROW_GRAM_BYTES,
17487            hash: checksum(&grams),
17488            verdicts: Mutex::new(Vec::new()),
17489        };
17490        let verdict = signatures.verdicts(&file, b"google").expect("signatures read");
17491        assert_eq!(&verdict[..], &[true, false, true], "one verdict a block, at the narrow width");
17492        assert!(signatures.footprint() > 0, "a verdict is remembered");
17493        let again = signatures.verdicts(&file, b"google").expect("remembered");
17494        assert!(Arc::ptr_eq(&verdict, &again), "a second question about a literal reads nothing");
17495
17496        let damaged = NativeGrams {
17497            hash: signatures.hash ^ 1,
17498            verdicts: Mutex::new(Vec::new()),
17499            ..signatures
17500        };
17501        let error = damaged.verdicts(&file, b"google").expect_err("a damaged region is refused");
17502        assert!(error.to_string().contains("substring signatures checksum differs"), "{error}");
17503        fs::remove_file(path).expect("remove scratch file");
17504    }
17505
17506    #[test]
17507    fn a_damaged_substring_signature_is_checked_only_when_used() {
17508        let path = path("damaged-substring-signature");
17509        let mut writer =
17510            Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
17511                .expect("new file");
17512        let rows = [Value::Varchar("google".into()), Value::Varchar("example".into())];
17513        writer
17514            .append(
17515                &Chunk::new(vec![
17516                    Vector::from_values(LogicalType::Varchar, &rows).expect("strings"),
17517                ])
17518                .expect("one column"),
17519            )
17520            .expect("stripe written");
17521        writer.finish().expect("commit");
17522
17523        let reader = Reader::open(&path).expect("valid directory");
17524        let page = reader.table.dictionaries[0].expect("string dictionary page");
17525        let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
17526        file.seek(SeekFrom::Start(page.offset + u64::from(page.length) - 1))
17527            .expect("last signature byte");
17528        file.write_all(&[255]).expect("damage signature");
17529        let reader = Reader::open(&path).expect("the directory is still valid");
17530        let dictionary = reader.dictionary(0).expect("index is still valid").expect("dictionary");
17531        let error = dictionary
17532            .text_block_might_contain(0, b"goog")
17533            .expect_err("a used signature checks its own checksum");
17534        assert!(error.message().contains("substring signatures checksum differs"), "{error}");
17535        fs::remove_file(path).expect("remove scratch file");
17536    }
17537
17538    /// A sweep over a block whose second run of offsets is short reads the same values as a point
17539    /// read does.
17540    ///
17541    /// The sweep decodes the offsets of a whole run at a time rather than a value at a time, and a
17542    /// run holds half a block, so the count it asks for is the run length everywhere but at the end
17543    /// of the dictionary. Two thousand five hundred values, which is what the test above writes,
17544    /// never puts a short run second in its block: the last block there begins on a run boundary and
17545    /// holds one run. Two thousand eight hundred does, so the last block is a whole run of five
17546    /// hundred and twelve followed by two hundred and forty, and an off by one in either the count
17547    /// asked for or the slice taken out of the answer shows up as a wrong value or a refusal.
17548    #[test]
17549    fn a_sweep_over_a_block_with_a_short_second_run_reads_what_a_point_read_reads() {
17550        let path = path("dictionary-sweep-short-run");
17551        let spellings = (0..2_800)
17552            .map(|index| Value::Varchar(format!("value {index:08} {}", "x".repeat(index % 40))))
17553            .collect::<Vec<_>>();
17554        let mut writer =
17555            Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
17556                .expect("new file");
17557        for part in spellings.chunks(1_024) {
17558            writer
17559                .append(
17560                    &Chunk::new(vec![
17561                        Vector::from_values(LogicalType::Varchar, part).expect("strings"),
17562                    ])
17563                    .expect("one column"),
17564                )
17565                .expect("stripe written");
17566        }
17567        writer.finish().expect("commit");
17568
17569        let reader = Reader::open(&path).expect("valid directory");
17570        let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
17571        assert_eq!(dictionary.len(), spellings.len(), "every value is distinct");
17572        let last = dictionary.len() % TEXT_PAYLOAD_VALUES;
17573        assert!(last > TEXT_OFFSET_RUN, "the last block has to reach into a second run of offsets");
17574        assert!(last < TEXT_PAYLOAD_VALUES, "and that second run has to be short of a whole one");
17575
17576        let mut swept: Vec<Vec<u8>> = Vec::new();
17577        let mut at = 0;
17578        while at < dictionary.len() {
17579            let stopped = dictionary
17580                .sweep_text(at, dictionary.len(), &mut |index: usize, text: &[u8]| {
17581                    assert_eq!(index, swept.len(), "a sweep hands its values over in order");
17582                    swept.push(text.to_vec());
17583                    Ok(())
17584                })
17585                .expect("a sweep reads");
17586            assert!(stopped > at, "a sweep moves");
17587            at = stopped;
17588        }
17589        let read = (0..dictionary.len())
17590            .map(|code| dictionary.try_bytes_at(code).expect("read").expect("a value").to_vec())
17591            .collect::<Vec<_>>();
17592        assert_eq!(swept, read, "a sweep answers what a point read answers");
17593        fs::remove_file(path).expect("remove scratch file");
17594    }
17595
17596    /// The unpacked ends answer what the packed ends answer, on both sides of the switch.
17597    ///
17598    /// A column asked for one offset at a time reads them out of the packed form until the reads
17599    /// are worth a table and out of the table after that, so every value here is read twice and the
17600    /// two passes are compared against the spellings and against each other. Two thousand eight
17601    /// hundred values is two payload blocks and a bit, which puts the switch in the middle of the
17602    /// first pass and means the pass straddles a block boundary, where the start of a value is zero
17603    /// rather than the end of the value before it.
17604    #[test]
17605    fn the_unpacked_ends_answer_what_the_packed_ends_answer() {
17606        let path = path("dictionary-unpacked-ends");
17607        let spellings = (0..2_800)
17608            .map(|index| Value::Varchar(format!("value {index:08} {}", "x".repeat(index % 40))))
17609            .collect::<Vec<_>>();
17610        let mut writer =
17611            Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
17612                .expect("new file");
17613        for part in spellings.chunks(1_024) {
17614            writer
17615                .append(
17616                    &Chunk::new(vec![
17617                        Vector::from_values(LogicalType::Varchar, part).expect("strings"),
17618                    ])
17619                    .expect("one column"),
17620                )
17621                .expect("stripe written");
17622        }
17623        writer.finish().expect("commit");
17624
17625        let reader = Reader::open(&path).expect("valid directory");
17626        let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
17627        assert_eq!(dictionary.len(), spellings.len(), "every value is distinct");
17628        let wanted = (0..spellings.len())
17629            .map(|index| format!("value {index:08} {}", "x".repeat(index % 40)).into_bytes())
17630            .collect::<Vec<_>>();
17631
17632        let pass = |what: &str| {
17633            for (index, value) in wanted.iter().enumerate() {
17634                let len = dictionary.try_bytes_len_at(index).expect("read").expect("a value");
17635                assert_eq!(len, value.len(), "{what} has the wrong length at {index}");
17636                let bytes = dictionary.try_bytes_at(index).expect("read").expect("a value");
17637                assert_eq!(bytes, value.as_slice(), "{what} has the wrong value at {index}");
17638            }
17639        };
17640        pass("the first pass");
17641        pass("the second pass");
17642
17643        // The whole vector in one call, over the text and through codes into it, which is how a
17644        // scan of a stored column hands it out. The codes run backwards and repeat so that they are
17645        // neither the positions nor in order.
17646        let lens = wanted.iter().map(|value| value.len() as i64).collect::<Vec<_>>();
17647        let mut whole = vec![0i64; wanted.len()];
17648        assert!(dictionary.try_bytes_lens(&mut whole).expect("read"), "the text answers whole");
17649        assert_eq!(whole, lens, "a vector of lengths answers what a length at a time answers");
17650        let codes = (0..4_000_u32).map(|row| (7 * (4_000 - row)) % 2_800).collect::<Vec<_>>();
17651        let coded = Vector::dictionary_over(codes.clone(), dictionary).expect("codes in range");
17652        let mut through = vec![0i64; codes.len()];
17653        assert!(coded.try_bytes_lens(&mut through).expect("read"), "the codes answer whole");
17654        for (row, &code) in codes.iter().enumerate() {
17655            assert_eq!(through[row], lens[code as usize], "row {row} reads code {code}");
17656            let one = coded.try_bytes_len_at(row).expect("read").expect("a value");
17657            assert_eq!(through[row], one as i64, "row {row} a row at a time");
17658        }
17659
17660        // A handful of codes over a column nobody has read yet is short of the table, so the same
17661        // call answers out of the packed ends instead, and has to answer the same.
17662        let fresh = Reader::open(&path).expect("valid directory");
17663        let untouched = fresh.dictionary(0).expect("read").expect("a string column has one");
17664        let few = vec![2_799_u32, 0, 1_024, 1_023, 511, 512];
17665        let coded = Vector::dictionary_over(few.clone(), untouched).expect("in range");
17666        let mut short = vec![0i64; few.len()];
17667        assert!(coded.try_bytes_lens(&mut short).expect("read"), "the codes answer whole");
17668        let expected = few.iter().map(|&code| lens[code as usize]).collect::<Vec<_>>();
17669        assert_eq!(short, expected, "the packed ends answer what the table answers");
17670        fs::remove_file(path).expect("remove scratch file");
17671    }
17672
17673    /// All three block layouts come back as the same values in the same order.
17674    ///
17675    /// Blocks outside the page are what every file this build writes holds. Blocks that say where
17676    /// they are but sit inside the page behind the order are format 26, and blocks behind one
17677    /// another with only their ends recorded are older still. Nothing in the writer produces the
17678    /// last two any more, so the only way to find out whether the reader still understands those
17679    /// files is to write them here. The
17680    /// bytes go straight into a file with no directory around them, because what is under test is
17681    /// [`open_global_dictionary`], which is handed a page and a file and asks the directory for
17682    /// nothing.
17683    ///
17684    /// Three thousand values so that there are three payload blocks and a partial fourth, which is
17685    /// what makes the last block the one place where a length and an end disagree about what they
17686    /// are counting.
17687    #[test]
17688    fn a_dictionary_reads_the_same_whether_its_blocks_say_where_they_are() {
17689        let spellings = (0..3_000)
17690            .map(|index| format!("value {index:08} {}", "y".repeat(index % 40)))
17691            .collect::<Vec<_>>();
17692        let mut read = Vec::new();
17693        for layout in ["outside", "inside", "behind"] {
17694            let mut dictionary = GlobalDictionary::new();
17695            for text in &spellings {
17696                dictionary.code(text).expect("a code for every spelling");
17697            }
17698            dictionary.finish_blocks().expect("the last block encodes");
17699            let order = dictionary.ranked(None).expect("a sorted order");
17700            // Where the blocks go if they start at `from` and follow one another.
17701            let laid = |from: u64| {
17702                let mut at = from;
17703                dictionary
17704                    .blocks
17705                    .iter()
17706                    .map(|block| {
17707                        let place =
17708                            Placed { start: at, length: block.len() as u64, hash: checksum(block) };
17709                        at += block.len() as u64;
17710                        place
17711                    })
17712                    .collect::<Vec<_>>()
17713            };
17714            let payload = dictionary.blocks.concat();
17715            let scattered = layout != "behind";
17716            let (bytes, encoded, offset, length) = if layout == "outside" {
17717                let mut bytes = vec![0; HEADER as usize];
17718                bytes.extend_from_slice(&payload);
17719                let encoded = encode_global_dictionary(&dictionary, &order, &laid(HEADER), true)
17720                    .expect("an encoding");
17721                let offset = bytes.len() as u64;
17722                bytes.extend_from_slice(&encoded.index);
17723                bytes.extend_from_slice(&encoded.ranks);
17724                bytes.extend_from_slice(&encoded.grams);
17725                let length = encoded.index.len() + encoded.ranks.len() + encoded.grams.len();
17726                (bytes, encoded, offset, length)
17727            } else {
17728                // The index is the same length wherever the blocks are, so a first pass says where
17729                // the page ends and the second writes the places that follow it.
17730                let first = encode_global_dictionary(&dictionary, &order, &laid(0), scattered)
17731                    .expect("an encoding");
17732                let body = (first.index.len() + first.ranks.len() + first.grams.len()) as u64;
17733                let encoded = encode_global_dictionary(&dictionary, &order, &laid(body), scattered)
17734                    .expect("an encoding");
17735                let mut bytes = encoded.index.clone();
17736                bytes.extend_from_slice(&encoded.ranks);
17737                bytes.extend_from_slice(&encoded.grams);
17738                bytes.extend_from_slice(&payload);
17739                let length = bytes.len();
17740                (bytes, encoded, 0, length)
17741            };
17742            let path = path(&format!("blocks-{layout}"));
17743            fs::write(&path, &bytes).expect("the dictionary is written on its own");
17744            let file = Arc::new(File::open(&path).expect("it opens again"));
17745            let page = Page {
17746                offset,
17747                length: u32::try_from(length).expect("a test dictionary is small"),
17748                hash: checksum(&encoded.index),
17749            };
17750            let opened =
17751                open_global_dictionary(file, page, &LogicalType::Varchar, TEXT_KEEP_BUDGET)
17752                    .expect("a dictionary laid out either way opens");
17753            let mut swept: Vec<Vec<u8>> = Vec::new();
17754            let mut at = 0;
17755            while at < opened.len() {
17756                at = opened
17757                    .sweep_text(at, opened.len(), &mut |_index: usize, text: &[u8]| {
17758                        swept.push(text.to_vec());
17759                        Ok(())
17760                    })
17761                    .expect("a sweep reads");
17762            }
17763            fs::remove_file(&path).expect("clean up");
17764            read.push(swept);
17765        }
17766        let wanted =
17767            spellings.iter().map(|text| text.as_bytes().to_vec()).collect::<Vec<Vec<u8>>>();
17768        assert_eq!(read[0], wanted, "the blocks outside the page hold the values");
17769        assert_eq!(read[1], read[0], "the blocks inside the page hold the same values");
17770        assert_eq!(read[2], read[0], "the blocks behind one another hold the same values");
17771    }
17772
17773    /// A dictionary at its budget sweeps without keeping, and still answers what it answered.
17774    ///
17775    /// The budget is a quarter of a gigabyte in a running database, which is a fine size for a real
17776    /// column and no size at all for a test, so this opens the same dictionary a second time with a
17777    /// budget of zero. That is the shape of the hundred million row case: `URL` fills the budget
17778    /// somewhere in the middle of itself and everything past that point is read and dropped, which
17779    /// costs the decode again and holds none of it.
17780    #[test]
17781    fn a_dictionary_at_its_budget_sweeps_without_keeping() {
17782        let path = path("dictionary-budget");
17783        let spellings = (0..2_500)
17784            .map(|index| Value::Varchar(format!("value {index:08} {}", "y".repeat(index % 40))))
17785            .collect::<Vec<_>>();
17786        let mut writer =
17787            Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
17788                .expect("new file");
17789        for part in spellings.chunks(1_024) {
17790            writer
17791                .append(
17792                    &Chunk::new(vec![
17793                        Vector::from_values(LogicalType::Varchar, part).expect("strings"),
17794                    ])
17795                    .expect("one column"),
17796                )
17797                .expect("stripe written");
17798        }
17799        writer.finish().expect("commit");
17800
17801        let reader = Reader::open(&path).expect("valid directory");
17802        let page = reader.table.dictionaries[0].expect("a string column has one");
17803        let file = Arc::clone(&reader.file);
17804        let starved = open_global_dictionary(file, page, &LogicalType::Varchar, 0)
17805            .expect("a dictionary opens whatever it may keep");
17806
17807        let resting = starved.footprint();
17808        let mut swept: Vec<Vec<u8>> = Vec::new();
17809        let mut at = 0;
17810        while at < starved.len() {
17811            at = starved
17812                .sweep_text(at, starved.len(), &mut |_index: usize, text: &[u8]| {
17813                    swept.push(text.to_vec());
17814                    Ok(())
17815                })
17816                .expect("a sweep reads");
17817        }
17818        assert_eq!(swept.len(), spellings.len(), "a starved sweep still reads every value");
17819        assert_eq!(starved.footprint(), resting, "and keeps no block it decoded");
17820
17821        let generous = reader.dictionary(0).expect("read").expect("a string column has one");
17822        let read = (0..generous.len())
17823            .map(|code| generous.try_bytes_at(code).expect("read").expect("a value").to_vec())
17824            .collect::<Vec<_>>();
17825        assert_eq!(swept, read, "a starved sweep answers what a point read answers");
17826        fs::remove_file(path).expect("remove scratch file");
17827    }
17828
17829    /// A part is hashed the first time a reader reads it and not after, and a reader opened after
17830    /// the part was damaged still refuses it.
17831    #[test]
17832    fn a_part_is_checked_once_per_open_reader() {
17833        let path = path("checked-once");
17834        let mut writer = Writer::create(
17835            &path,
17836            "items",
17837            vec![
17838                Field::required("id", LogicalType::Integer),
17839                Field::new("text", LogicalType::Varchar),
17840            ],
17841        )
17842        .expect("new file");
17843        writer.append(&sample()).expect("stripe written");
17844        writer.finish().expect("commit");
17845
17846        let reader = Reader::open(&path).expect("valid directory");
17847        let first = reader.read_rows(0, &[0], &[0, 1], false).expect("checked and read");
17848        assert!(reader.is_verified(0), "the part is remembered as checked");
17849        let page = reader.table.stripes[0].pages[0];
17850        let mut file = OpenOptions::new().write(true).open(&path).expect("open column page");
17851        file.seek(SeekFrom::Start(page.offset + u64::from(page.length) - 1)).expect("page end");
17852        file.write_all(&[0xa5]).expect("damage page");
17853        if let Err(error) = reader.read_rows(0, &[0], &[0, 1], false) {
17854            assert!(!error.message().contains("checksum differs"), "not hashed again: {error}");
17855        }
17856        let fresh = Reader::open(&path).expect("valid directory");
17857        let error = fresh.read_rows(0, &[0], &[0, 1], false).expect_err("a new reader checks");
17858        assert!(error.message().contains("column page checksum differs"), "{error}");
17859        assert_eq!(first.len(), 2);
17860        fs::remove_file(path).expect("remove scratch file");
17861    }
17862
17863    #[test]
17864    fn damaged_membership_cannot_skip_a_string_page() {
17865        let path = path("damaged-membership");
17866        let mut writer = Writer::create(
17867            &path,
17868            "items",
17869            vec![
17870                Field::required("id", LogicalType::Integer),
17871                Field::new("text", LogicalType::Varchar),
17872            ],
17873        )
17874        .expect("new file");
17875        writer.append(&sample()).expect("stripe written");
17876        writer.finish().expect("commit");
17877
17878        let reader = Reader::open(&path).expect("valid directory");
17879        let membership = reader.table.stripes[0].memberships.get(1).expect("string membership");
17880        let mut file = OpenOptions::new().write(true).open(&path).expect("open membership page");
17881        file.seek(SeekFrom::Start(membership.offset)).expect("membership start");
17882        file.write_all(&[255]).expect("damage membership");
17883        let error = reader.skips_codes(0, 1, &[3]).expect_err("corruption must not skip rows");
17884        assert!(error.message().contains("membership page checksum differs"), "{error}");
17885        fs::remove_file(path).expect("remove scratch file");
17886    }
17887
17888    #[test]
17889    fn membership_delta_stream_is_sorted_exact_and_bounded() {
17890        let unique = unique_codes(&[900, 4, 4, 72, 9, u32::MAX]);
17891        assert_eq!(unique, [4, 9, 72, 900, u32::MAX]);
17892        let encoded = encode_membership(&unique);
17893        assert_eq!(
17894            decode_membership(&encoded).expect("valid membership"),
17895            [4, 9, 72, 900, u32::MAX]
17896        );
17897        // A stripe's index is the union of its parts', so a code in two of them is in it once and
17898        // the result is still one ascending run of deltas.
17899        let merged = merged_codes(vec![vec![4, 900], vec![9, 900, u32::MAX], vec![72]]);
17900        assert_eq!(merged, [4, 9, 72, 900, u32::MAX]);
17901        assert_eq!(
17902            decode_membership(&encode_membership(&merged)).expect("valid membership"),
17903            unique
17904        );
17905        assert!(decode_membership(&[1, 0x80]).is_err(), "a truncated varint is invalid");
17906        assert!(
17907            decode_membership(&[1, 0xff, 0xff, 0xff, 0xff, 0x10]).is_err(),
17908            "a value past u32 is invalid"
17909        );
17910    }
17911
17912    #[test]
17913    fn a_global_dictionary_may_be_larger_than_one_column_page() {
17914        let dictionary = Page {
17915            offset: HEADER,
17916            length: u32::try_from(MAX_PAGE + 1).expect("the page bound fits on disk"),
17917            hash: 0,
17918        };
17919        let table = Table {
17920            name: "items".to_owned(),
17921            fields: vec![Field::new("text", LogicalType::Varchar)],
17922            stripes: Vec::new(),
17923            rows: 0,
17924            dictionaries: vec![Some(dictionary)],
17925            dictionary_payloads: Vec::new(),
17926            demoted: Vec::new(),
17927            distincts: vec![None],
17928            frequencies: vec![None],
17929            ordinal_bounds: Vec::new(),
17930            pair_frequencies: Vec::new(),
17931            frequency_texts: Vec::new(),
17932            host_groups: None,
17933            clustering: None,
17934            constraints: Constraints::default(),
17935            generation: 1,
17936            sections: Vec::new(),
17937        };
17938        let directory = encode_directory(&table).expect("directory");
17939        let file_size = dictionary.offset + u64::from(dictionary.length) + 1;
17940
17941        let decoded = decode_directory(&directory, file_size).expect("large lazy dictionary");
17942        assert_eq!(decoded.dictionaries[0].expect("dictionary").length, dictionary.length);
17943    }
17944
17945    #[test]
17946    fn a_column_with_one_value_everywhere_costs_almost_nothing_a_row() {
17947        let path = path("constant-codes");
17948        let mut writer =
17949            Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
17950                .expect("new file");
17951        let empty = vec![Value::Varchar(String::new()); 1024];
17952        for _ in 0..4 {
17953            let column = Vector::from_values(LogicalType::Varchar, &empty).expect("strings");
17954            writer.append(&Chunk::new(vec![column]).expect("one column")).expect("a part");
17955        }
17956        writer.finish().expect("commit");
17957
17958        let reader = Reader::open(&path).expect("valid directory");
17959        let pages = reader.layout().columns.first().expect("one column").pages;
17960        // This column used to cost four bytes a row, 16,384 of them, the same as a column of four
17961        // thousand distinct URLs would. The cascade calls each part a constant, so what is left is
17962        // a tag, a count and the value, and the row count stops being what drives the number.
17963        assert!(pages < 256, "{pages} bytes of pages for 4,096 rows of one value");
17964        let read = reader.read(3, &[0]).expect("the last part back");
17965        assert_eq!(read.value_at(0, 0), Value::Varchar(String::new()));
17966        assert_eq!(read.value_at(1023, 0), Value::Varchar(String::new()));
17967        fs::remove_file(path).expect("remove scratch file");
17968    }
17969
17970    #[test]
17971    fn a_cascade_value_too_wide_for_its_column_is_refused_rather_than_cut() {
17972        // What a damaged page looks like from here: the cascade decoded, so the bytes are not
17973        // truncated, but the values do not belong to the column the directory says they do.
17974        let over = integer::encode(&[i64::from(i32::MAX) + 1]).expect("a chunk");
17975        let error = cascade(&LogicalType::Integer, &over, 1).expect_err("a page that disagrees");
17976        assert!(format!("{error}").contains("not of its type"), "{error}");
17977        let low = integer::encode(&[i64::MIN]).expect("a chunk");
17978        assert!(cascade(&LogicalType::BigInt, &low, 1).is_ok(), "bigint holds all of i64");
17979        let zero = integer::encode(&[0]).expect("a chunk");
17980        assert!(cascade(&LogicalType::Varchar, &zero, 1).is_err(), "strings are not integers");
17981    }
17982
17983    #[test]
17984    fn a_code_stream_the_cascade_cannot_shrink_is_left_alone() {
17985        // A shift register rather than a run, because an arithmetic run is the one wide shape the
17986        // cascade does shrink. This is what a column with tens of millions of distinct values hands
17987        // over: full width codes with no order to them.
17988        let mut state: u32 = 0x9e37_79b9;
17989        let spread: Vec<u32> = (0..1024)
17990            .map(|_| {
17991                state ^= state << 13;
17992                state ^= state >> 17;
17993                state ^= state << 5;
17994                state
17995            })
17996            .collect();
17997        assert_eq!(encoded_codes(&spread).expect("no failure"), None);
17998        let near: Vec<u32> = (0..1024).collect();
17999        let coded = encoded_codes(&near).expect("no failure").expect("counting up is packable");
18000        assert!(coded.len() < near.len() * 4, "{} bytes for a run of 1,024", coded.len());
18001    }
18002
18003    /// The columns of a stripe are encoded on whichever thread got to them, so the one thing that
18004    /// must not depend on which thread that was is the file. Two writes of the same rows are
18005    /// compared byte for byte rather than value for value, because a dictionary that two columns
18006    /// somehow shared would still read back correctly and would hand out its codes in the order the
18007    /// threads happened to run in, which is exactly what this is here to catch.
18008    #[test]
18009    fn two_writes_of_the_same_rows_give_the_same_bytes() {
18010        fn written(path: &PathBuf) {
18011            let fields = (0..40)
18012                .map(|column| {
18013                    let ty =
18014                        if column % 4 == 0 { LogicalType::Varchar } else { LogicalType::BigInt };
18015                    Field::new(format!("c{column}"), ty)
18016                })
18017                .collect::<Vec<_>>();
18018            let mut writer = Writer::create(path, "wide", fields).expect("new file");
18019            for part in 0..70_u64 {
18020                let columns = (0..40)
18021                    .map(|column| {
18022                        let values = (0..64_u64)
18023                            .map(|row| {
18024                                let seed = part.wrapping_mul(31).wrapping_add(row);
18025                                if column % 4 == 0 {
18026                                    Value::Varchar(format!("v{}", seed % 17))
18027                                } else {
18028                                    Value::BigInt(i64::try_from(seed % 97).expect("small"))
18029                                }
18030                            })
18031                            .collect::<Vec<_>>();
18032                        let ty = if column % 4 == 0 {
18033                            LogicalType::Varchar
18034                        } else {
18035                            LogicalType::BigInt
18036                        };
18037                        Vector::from_values(ty, &values).expect("a column")
18038                    })
18039                    .collect::<Vec<_>>();
18040                writer.append(&Chunk::new(columns).expect("forty columns")).expect("a part");
18041            }
18042            writer.finish().expect("commit");
18043        }
18044
18045        let first = path("repeatable-one");
18046        let second = path("repeatable-two");
18047        written(&first);
18048        written(&second);
18049        let left = fs::read(&first).expect("the first file");
18050        let right = fs::read(&second).expect("the second file");
18051        assert_eq!(left.len(), right.len(), "two writes of the same rows differ in length");
18052        assert!(left == right, "two writes of the same rows differ in their bytes");
18053
18054        // And the rows are still there, since a pair of identically wrong files would pass the
18055        // comparison above on its own.
18056        let reader = Reader::open(&first).expect("valid directory");
18057        assert_eq!(reader.table().rows(), 70 * 64);
18058        let read = reader.read(0, &[0, 1]).expect("the first part back");
18059        assert_eq!(read.value_at(0, 0), Value::Varchar("v0".to_owned()));
18060        assert_eq!(read.value_at(0, 1), Value::BigInt(0));
18061        fs::remove_file(first).expect("remove scratch file");
18062        fs::remove_file(second).expect("remove scratch file");
18063    }
18064
18065    /// Three tables of different shapes in one file, read back by name.
18066    fn three_tables(path: &PathBuf) {
18067        let writer = Writer::create(
18068            path,
18069            "region",
18070            vec![
18071                Field::new("r_key", LogicalType::Integer),
18072                Field::new("r_name", LogicalType::Varchar),
18073            ],
18074        )
18075        .expect("new file");
18076        let mut writer = writer;
18077        writer
18078            .append(
18079                &Chunk::new(vec![
18080                    Vector::from_values(
18081                        LogicalType::Integer,
18082                        &[Value::Integer(0), Value::Integer(1)],
18083                    )
18084                    .expect("keys"),
18085                    Vector::from_values(
18086                        LogicalType::Varchar,
18087                        &[Value::Varchar("AFRICA".to_owned()), Value::Varchar("ASIA".to_owned())],
18088                    )
18089                    .expect("names"),
18090                ])
18091                .expect("two columns"),
18092            )
18093            .expect("a part");
18094        let mut writer = writer
18095            .next("empty", vec![Field::new("nothing", LogicalType::BigInt)])
18096            .expect("a second table");
18097        writer
18098            .append(
18099                &Chunk::new(vec![
18100                    Vector::from_values(LogicalType::BigInt, &[Value::BigInt(7)]).expect("a row"),
18101                ])
18102                .expect("one column"),
18103            )
18104            .expect("a part");
18105        let mut writer =
18106            writer.next("wide", vec![Field::new("n", LogicalType::BigInt)]).expect("a third table");
18107        for part in 0..70_i64 {
18108            let values = (0..64).map(|row| Value::BigInt(part * 64 + row)).collect::<Vec<_>>();
18109            writer
18110                .append(
18111                    &Chunk::new(vec![
18112                        Vector::from_values(LogicalType::BigInt, &values).expect("a column"),
18113                    ])
18114                    .expect("one column"),
18115                )
18116                .expect("a part");
18117        }
18118        writer.finish().expect("commit");
18119    }
18120
18121    #[test]
18122    fn three_tables_in_one_file_read_back_by_name() {
18123        let file = path("three-tables");
18124        three_tables(&file);
18125        let catalog = Catalog::open(&file).expect("a committed catalog");
18126        assert_eq!(catalog.names().collect::<Vec<_>>(), ["region", "empty", "wide"]);
18127
18128        let region = catalog.table("region").expect("the first table");
18129        assert_eq!(region.table().rows(), 2);
18130        assert_eq!(
18131            region.read(0, &[1]).expect("names").value_at(1, 0),
18132            Value::Varchar("ASIA".to_owned())
18133        );
18134
18135        let wide = catalog.table("wide").expect("the third table");
18136        assert_eq!(wide.table().rows(), 70 * 64);
18137        assert_eq!(wide.read(0, &[0]).expect("the first part").value_at(0, 0), Value::BigInt(0));
18138
18139        // The middle table is reached without the one after it having been touched, which is what
18140        // a directory per table buys over one directory of everything.
18141        let empty = catalog.table("empty").expect("the second table");
18142        assert_eq!(empty.table().rows(), 1);
18143        assert_eq!(empty.read(0, &[0]).expect("the row").value_at(0, 0), Value::BigInt(7));
18144
18145        fs::remove_file(file).expect("remove scratch file");
18146    }
18147
18148    #[test]
18149    fn a_name_the_file_does_not_hold_is_an_error_rather_than_the_first_table() {
18150        let file = path("three-tables-missing");
18151        three_tables(&file);
18152        let catalog = Catalog::open(&file).expect("a committed catalog");
18153        let error = catalog.table("nation").expect_err("no such table");
18154        assert!(error.message().contains("nation"), "{}", error.message());
18155        fs::remove_file(file).expect("remove scratch file");
18156    }
18157
18158    #[test]
18159    fn a_file_of_three_tables_will_not_open_as_one() {
18160        let file = path("three-tables-unnamed");
18161        three_tables(&file);
18162        let error = Reader::open(&file).expect_err("more than one table");
18163        assert!(error.message().contains("more than one table"), "{}", error.message());
18164        fs::remove_file(file).expect("remove scratch file");
18165    }
18166
18167    /// One column per storage width, because the width is what decides how many bytes a row costs.
18168    #[test]
18169    fn decimals_of_every_storage_width_round_trip() {
18170        let file = path("decimals");
18171        let widths = [(4_u8, 2_u8), (9, 2), (18, 4), (38, 6)];
18172        let fields = widths
18173            .iter()
18174            .enumerate()
18175            .map(|(index, (width, scale))| {
18176                Field::new(
18177                    format!("d{index}"),
18178                    LogicalType::decimal(*width, *scale).expect("a decimal type"),
18179                )
18180            })
18181            .collect::<Vec<_>>();
18182        let mut writer = Writer::create(&file, "money", fields).expect("new file");
18183        let rows: [i128; 3] = [-1234, 0, 999];
18184        let columns = widths
18185            .iter()
18186            .map(|(width, scale)| {
18187                let values = rows
18188                    .iter()
18189                    .map(|unscaled| Value::Decimal {
18190                        unscaled: *unscaled,
18191                        width: *width,
18192                        scale: *scale,
18193                    })
18194                    .collect::<Vec<_>>();
18195                Vector::from_values(
18196                    LogicalType::decimal(*width, *scale).expect("a decimal type"),
18197                    &values,
18198                )
18199                .expect("a decimal column")
18200            })
18201            .collect::<Vec<_>>();
18202        writer.append(&Chunk::new(columns).expect("four columns")).expect("a part");
18203        writer.finish().expect("commit");
18204
18205        let reader = Reader::open(&file).expect("a committed file");
18206        for (index, (width, scale)) in widths.iter().enumerate() {
18207            assert_eq!(
18208                reader.table().fields()[index].ty,
18209                LogicalType::decimal(*width, *scale).expect("a decimal type"),
18210                "column {index} came back as another type"
18211            );
18212            let column = reader.read(0, &[index]).expect("the column");
18213            for (row, unscaled) in rows.iter().enumerate() {
18214                assert_eq!(
18215                    column.value_at(row, 0),
18216                    Value::Decimal { unscaled: *unscaled, width: *width, scale: *scale },
18217                    "column {index} row {row}"
18218                );
18219            }
18220        }
18221        fs::remove_file(file).expect("remove scratch file");
18222    }
18223
18224    #[test]
18225    fn two_tables_of_one_name_are_refused_before_anything_is_committed() {
18226        let file = path("two-of-a-name");
18227        let writer = Writer::create(&file, "t", vec![Field::new("a", LogicalType::BigInt)])
18228            .expect("new file");
18229        let error = writer
18230            .next("t", vec![Field::new("a", LogicalType::BigInt)])
18231            .expect_err("the same name twice");
18232        assert!(error.message().contains("same name"), "{}", error.message());
18233        fs::remove_file(file).expect("remove scratch file");
18234    }
18235
18236    #[test]
18237    fn integer_tally_counts_encoded_rows_and_declines_null_parts() {
18238        let file = path("integer-tally");
18239        let mut writer =
18240            Writer::create(&file, "events", vec![Field::new("source", LogicalType::SmallInt)])
18241                .expect("new file");
18242        let mut values = vec![Value::SmallInt(0); 1024];
18243        values[7] = Value::SmallInt(3);
18244        values[99] = Value::SmallInt(-2);
18245        values[1001] = Value::SmallInt(3);
18246        let column = Vector::from_values(LogicalType::SmallInt, &values).expect("integer values");
18247        writer.append(&Chunk::new(vec![column]).expect("one column")).expect("first part");
18248        values[0] = Value::Null;
18249        let column = Vector::from_values(LogicalType::SmallInt, &values).expect("nullable values");
18250        writer.append(&Chunk::new(vec![column]).expect("one column")).expect("second part");
18251        writer.finish().expect("commit");
18252
18253        let reader = Reader::open(&file).expect("read file");
18254        assert_eq!(
18255            reader.integer_tally(0, 0).expect("valid part"),
18256            Some(vec![(-2, 1), (0, 1021), (3, 2)])
18257        );
18258        assert!(reader.integer_tally(1, 0).expect("valid null part").is_none());
18259        let catalog = Catalog::open(&file).expect("catalog");
18260        assert_eq!(
18261            catalog.integer_tally("events", 0).expect("nullable column"),
18262            Some(vec![(-2, 2), (0, 2041), (3, 4)])
18263        );
18264        fs::remove_file(file).expect("remove scratch file");
18265    }
18266
18267    #[test]
18268    fn catalog_tallies_one_integer_column_without_opening_the_whole_table() {
18269        let file = path("catalog-integer-tally");
18270        let mut writer = Writer::create(
18271            &file,
18272            "events",
18273            vec![
18274                Field::new("noise", LogicalType::SmallInt),
18275                Field::new("source", LogicalType::SmallInt),
18276            ],
18277        )
18278        .expect("new file");
18279        let noise = vec![Value::SmallInt(9); 1024];
18280        let mut source = vec![Value::SmallInt(0); 1024];
18281        source[7] = Value::SmallInt(3);
18282        source[99] = Value::SmallInt(-2);
18283        let chunk = Chunk::new(vec![
18284            Vector::from_values(LogicalType::SmallInt, &noise).expect("noise"),
18285            Vector::from_values(LogicalType::SmallInt, &source).expect("source"),
18286        ])
18287        .expect("two columns");
18288        writer.append(&chunk).expect("append");
18289        writer.finish().expect("commit");
18290
18291        let catalog = Catalog::open(&file).expect("catalog");
18292        assert_eq!(
18293            catalog.integer_tally("events", 1).expect("selected column"),
18294            Some(vec![(-2, 1), (0, 1022), (3, 1)])
18295        );
18296        assert_eq!(
18297            catalog.integer_tally("events", 0).expect("other column"),
18298            Some(vec![(9, 1024)])
18299        );
18300        fs::remove_file(file).expect("remove scratch file");
18301    }
18302
18303    #[test]
18304    fn opening_the_catalog_reads_no_table_directory() {
18305        let file = path("catalog-only");
18306        three_tables(&file);
18307        let catalog = Catalog::open(&file).expect("a committed catalog");
18308        // The header and one slot, and nothing under it. The third table's directory covers seventy
18309        // stripes and reading it here would be the whole point of the two levels thrown away.
18310        assert_eq!(catalog.opening.reads, 2, "opening the catalog read more than the slot");
18311        assert_eq!(catalog.names().len(), 3);
18312        fs::remove_file(file).expect("remove scratch file");
18313    }
18314
18315    /// The checksum answers what it has always answered, at every length its branches split on.
18316    ///
18317    /// This is a compatibility test rather than a correctness one. Nothing about the hash has to be
18318    /// any particular function, but a file already on disk carries the answers the version that
18319    /// wrote it gave, so a change here is a change that makes every stored file fail to verify. The
18320    /// lengths are the ones the code makes decisions about: nothing, under a block, a block exactly,
18321    /// a block and a word, a word and a half word, and a half word and a byte.
18322    ///
18323    /// The empty answer is the published xxHash64 vector for an empty input at seed zero, which is
18324    /// also a check that this is the function it says it is.
18325    #[test]
18326    fn the_checksum_answers_what_it_has_always_answered() {
18327        let bytes: Vec<u8> =
18328            (0..1000_u32).map(|at| (at.wrapping_mul(31).wrapping_add(7) % 251) as u8).collect();
18329        for (length, expected) in [
18330            (0, 0xef46_db37_51d8_e999),
18331            (1, 0xa96c_7f0c_e858_bbb7),
18332            (3, 0x56e6_9576_32a4_87f9),
18333            (4, 0xc60d_15b1_e3ff_8f04),
18334            (5, 0x8088_1585_8624_dd4e),
18335            (7, 0xafbe_fc3d_6c6f_9a8e),
18336            (8, 0x3da5_c7aa_2696_83e0),
18337            (9, 0x465e_c429_b13c_3892),
18338            (15, 0xdee8_9d8a_065a_6233),
18339            (16, 0x1330_489a_7767_9c80),
18340            (31, 0x3391_303d_485e_846e),
18341            (32, 0x40b7_aff7_5d45_bbc8),
18342            (33, 0x4997_cae4_951c_17a5),
18343            (39, 0x5807_28fd_5c14_5739),
18344            (40, 0xf95c_f6f5_c08a_3d3b),
18345            (63, 0x2944_b4da_fc69_b206),
18346            (64, 0xbb76_f6ef_19bd_5a1b),
18347            (65, 0x814e_0c65_4a9f_d640),
18348            (127, 0x00de_aab1_31cf_f89b),
18349            (1000, 0x9e33_00c1_cde3_c58d),
18350        ] {
18351            assert_eq!(checksum(&bytes[..length]), expected, "the checksum of {length} bytes");
18352        }
18353        assert_eq!(checksum(b"the quick brown fox jumps over the lazy dog"), 0xed71_4233_c5a9_a792);
18354    }
18355    /// A declared order survives the file, and a table that declared none stays as it was.
18356    ///
18357    /// The second half is the one worth a test. The clustering section is written only when there
18358    /// is a declaration, so a file of two tables where one is clustered exercises both the present
18359    /// and the absent branch of the decoder in one directory, which is where a length bug would
18360    /// show up as one table reading the other's bytes.
18361    #[test]
18362    fn a_declared_order_comes_back_out_of_the_file() {
18363        let path = path("clustered");
18364        let shipped = vec![
18365            Field::new("key", LogicalType::BigInt),
18366            Field::new("line", LogicalType::Integer),
18367            Field::new("shipdate", LogicalType::Date),
18368        ];
18369        let plain = vec![Field::new("a", LogicalType::Integer)];
18370        let stage_zero = Clustering::new(vec![2, 0, 1], Width::Month, &shipped).expect("valid");
18371
18372        let mut writer = Writer::create(&path, "lineitem", shipped)
18373            .expect("new file")
18374            .declare(stage_zero.clone())
18375            .expect("the columns are the table's");
18376        let column = |ty: LogicalType, values: &[Value]| {
18377            Vector::from_values(ty, values).expect("the values match the type")
18378        };
18379        writer
18380            .append(
18381                &Chunk::new(vec![
18382                    column(
18383                        LogicalType::BigInt,
18384                        &[Value::BigInt(0), Value::BigInt(1), Value::BigInt(2), Value::BigInt(3)],
18385                    ),
18386                    column(
18387                        LogicalType::Integer,
18388                        &[
18389                            Value::Integer(1),
18390                            Value::Integer(1),
18391                            Value::Integer(1),
18392                            Value::Integer(1),
18393                        ],
18394                    ),
18395                    column(
18396                        LogicalType::Date,
18397                        &[Value::Date(0), Value::Date(1), Value::Date(2), Value::Date(3)],
18398                    ),
18399                ])
18400                .expect("three columns"),
18401            )
18402            .expect("four rows");
18403        let mut writer = writer.next("nation", plain).expect("a second table");
18404        writer
18405            .append(
18406                &Chunk::new(vec![column(LogicalType::Integer, &[Value::Integer(7)])])
18407                    .expect("one column"),
18408            )
18409            .expect("one row");
18410        writer.finish().expect("commit");
18411
18412        let catalog = Catalog::open(&path).expect("reopen");
18413        let lineitem = catalog.table("lineitem").expect("the clustered table");
18414        assert_eq!(lineitem.table().clustering(), Some(&stage_zero));
18415        let nation = catalog.table("nation").expect("the plain table");
18416        assert_eq!(nation.table().clustering(), None, "nobody declared one here");
18417
18418        // And the rows are still the rows, because the section goes on the end of the directory
18419        // and the easy way to break that is to leave the cursor somewhere the next read trusts.
18420        assert_eq!(lineitem.table().rows(), 4);
18421        assert_eq!(nation.table().rows(), 1);
18422        fs::remove_file(&path).ok();
18423    }
18424
18425    /// A declaration naming a column the table does not have is refused where it is made.
18426    #[test]
18427    fn a_declaration_off_the_end_of_the_table_never_reaches_the_file() {
18428        let path = path("clustered-bad");
18429        let writer = Writer::create(&path, "items", vec![Field::new("a", LogicalType::Integer)])
18430            .expect("new file");
18431        let four =
18432            (0..4).map(|at| Field::new(format!("c{at}"), LogicalType::Integer)).collect::<Vec<_>>();
18433        let wrong = Clustering::new(vec![3], Width::Exact, &four).expect("valid against four");
18434        assert!(writer.declare(wrong).is_err(), "the table has one column, not four");
18435        fs::remove_file(&path).ok();
18436    }
18437
18438    /// The sorted order is the byte order, whatever the values do before they differ.
18439    ///
18440    /// A block handed out of the writer's lock to be encoded, and given back in whatever order the
18441    /// stripes happen to finish in, is the same block with the same signature as one encoded in
18442    /// place, and lands in the same position.
18443    #[test]
18444    fn blocks_handed_out_and_given_back_out_of_order_are_the_blocks_encoded_in_place() {
18445        let values = (0..PAYLOAD_SAMPLE_BLOCKS * TEXT_PAYLOAD_VALUES * 2 + 100)
18446            .map(|at| format!("http://example{}.test/page/{at:06}", at % 7))
18447            .collect::<Vec<_>>();
18448        let filled = || {
18449            let mut dictionary = GlobalDictionary::new();
18450            for value in &values {
18451                dictionary.code(value).expect("a code for every value");
18452            }
18453            dictionary.settle().expect("a shape");
18454            dictionary
18455        };
18456        let mut in_place = filled();
18457        in_place.finish_blocks().expect("every block encodes");
18458
18459        let mut handed = filled();
18460        let out = handed.hand_out(3);
18461        assert_eq!(out.len(), PAYLOAD_SAMPLE_BLOCKS * 2, "every sealed block goes out");
18462        assert!(handed.waiting.is_empty(), "and none is left to be encoded under the lock");
18463        for job in out.iter().rev() {
18464            assert_eq!(job.place().0, 3, "a block goes back to the column it came from");
18465            handed.take_back(job.place().1, job.encode().expect("encodes")).expect("taken back");
18466        }
18467        assert!(handed.early.is_empty(), "nothing is waiting on a gap");
18468        handed.finish_blocks().expect("the last block encodes");
18469
18470        assert_eq!(handed.blocks, in_place.blocks, "the same blocks in the same order");
18471        assert_eq!(handed.grams, in_place.grams, "with the same signatures");
18472    }
18473
18474    /// A block given back twice is a bug in whoever gave it, and is said rather than written twice.
18475    #[test]
18476    fn a_block_given_back_twice_is_refused() {
18477        let mut dictionary = GlobalDictionary::new();
18478        for at in 0..PAYLOAD_SAMPLE_BLOCKS * TEXT_PAYLOAD_VALUES {
18479            dictionary.code(&format!("value {at}")).expect("a code");
18480        }
18481        dictionary.settle().expect("a shape");
18482        let out = dictionary.hand_out(0);
18483        let last = out.last().expect("blocks went out");
18484        let at = last.place().1;
18485        dictionary.take_back(at, last.encode().expect("encodes")).expect("taken back once");
18486        assert!(dictionary.take_back(at, last.encode().expect("encodes")).is_err());
18487    }
18488
18489    /// The values here are the shape the sort is built for and the shape a comparison sort is worst
18490    /// at: a common scheme, a handful of hosts, and a path that only decides the pair thirty bytes
18491    /// in. They also cover what the bucketing has to get right at the edges, which is a value that
18492    /// has run out where another carries on, the empty value, and enough entries to take the range
18493    /// down through several passes and out the bottom into the comparison that finishes it.
18494    #[test]
18495    fn the_dictionary_order_is_the_byte_order_however_deep_the_values_agree() {
18496        let mut values = vec![String::new(), "http://".to_owned()];
18497        for host in 0..7 {
18498            for path in 0..30 {
18499                values.push(format!("http://example{host}.test/page/{path:04}/index.html"));
18500                values.push(format!("http://example{host}.test/page/{path:04}"));
18501            }
18502        }
18503        values.push("http://example0.test/page/0000/index.htmlx".to_owned());
18504
18505        let mut dictionary = GlobalDictionary::new();
18506        for value in &values {
18507            dictionary.code(value).expect("a code for every value");
18508        }
18509        dictionary.finish_blocks().expect("the last block encodes");
18510        let ranked = dictionary.ranked(None).expect("a sorted order");
18511        assert_eq!(ranked.len(), values.len(), "one entry a distinct value");
18512
18513        let spellings = dictionary_values(&dictionary);
18514        let seen = ranked
18515            .iter()
18516            .map(|&(_, code)| {
18517                String::from_utf8(spellings[code as usize].clone()).expect("text in, text out")
18518            })
18519            .collect::<Vec<_>>();
18520        let mut wanted = values.clone();
18521        wanted.sort_unstable();
18522        assert_eq!(seen, wanted, "the order is the order the bytes give");
18523
18524        for &(carried, code) in &ranked {
18525            let value = &spellings[code as usize];
18526            assert_eq!(carried, head(value), "the head belongs to the value it is filed with");
18527        }
18528    }
18529
18530    /// Picking the commonest entries leaves exactly what sorting all of them and cutting left.
18531    ///
18532    /// The counts here are deliberately full of ties, including a tie that straddles the cut, which
18533    /// is where a partition and a sort can disagree if the comparison they are given is not total.
18534    #[test]
18535    fn the_commonest_entries_are_the_ones_a_full_sort_would_have_kept() {
18536        let entry =
18537            |value: u32, count: u64| FrequencyEntry { value: FrequencyValue::Code(value), count };
18538        let mut all = (0..FREQUENCY_ENTRIES as u32 * 3)
18539            .map(|code| entry(code, u64::from(code % 7) + 1))
18540            .collect::<Vec<_>>();
18541        all.push(FrequencyEntry { value: FrequencyValue::Null, count: 4 });
18542
18543        let mut sorted = all.clone();
18544        sorted.sort_unstable_by(|left, right| {
18545            right.count.cmp(&left.count).then_with(|| frequency_order(left.value, right.value))
18546        });
18547        let wanted_omitted = sorted[FREQUENCY_ENTRIES].count;
18548        sorted.truncate(FREQUENCY_ENTRIES);
18549
18550        let mut picked = all.clone();
18551        let omitted = keep_most_frequent(&mut picked);
18552        assert_eq!(omitted, wanted_omitted, "the largest count that did not make the cut");
18553        assert_eq!(picked.len(), FREQUENCY_ENTRIES, "the cut is where it says it is");
18554        assert!(
18555            picked
18556                .iter()
18557                .zip(&sorted)
18558                .all(|(one, two)| one.value == two.value && one.count == two.count),
18559            "the same entries in the same order"
18560        );
18561
18562        let mut short = all[..FREQUENCY_ENTRIES - 1].to_vec();
18563        let omitted = keep_most_frequent(&mut short);
18564        assert_eq!(omitted, 0, "nothing is omitted when everything fits");
18565        assert!(short.windows(2).all(|pair| pair[0].count >= pair[1].count), "still in order");
18566    }
18567
18568    /// A dictionary too small to bucket, and one with nothing in it, come back in order too.
18569    #[test]
18570    fn a_short_dictionary_sorts_without_a_bucketing_pass() {
18571        let empty = GlobalDictionary::new();
18572        assert!(empty.ranked(None).expect("an empty order").is_empty(), "nothing in, nothing out");
18573
18574        let mut dictionary = GlobalDictionary::new();
18575        for value in ["pear", "apple", "", "apples", "app"] {
18576            dictionary.code(value).expect("a code for every value");
18577        }
18578        dictionary.finish_blocks().expect("the one block encodes");
18579        let spellings = dictionary_values(&dictionary);
18580        let seen = dictionary
18581            .ranked(None)
18582            .expect("a sorted order")
18583            .iter()
18584            .map(|&(_, code)| spellings[code as usize].clone())
18585            .collect::<Vec<_>>();
18586        let wanted: Vec<Vec<u8>> =
18587            [&b""[..], b"app", b"apple", b"apples", b"pear"].iter().map(|v| v.to_vec()).collect();
18588        assert_eq!(seen, wanted, "shorter first where one runs out inside another");
18589    }
18590
18591    /// A demoted dictionary gives back what it kept for looking values up, the load profile is told,
18592    /// and it refuses any value after that.
18593    #[test]
18594    fn a_demoted_dictionary_holds_less_and_takes_no_more_values() {
18595        let profile = LoadProfile::begin("demoted");
18596        let mut dictionary = GlobalDictionary::new();
18597        for value in 0..50_000 {
18598            dictionary.code(&format!("https://example.com/page/{value}")).expect("a code");
18599        }
18600        let (_, grown) = dictionary.recharge(Some(&profile));
18601        assert_eq!(profile.held(), grown, "the profile holds what the dictionary does");
18602
18603        dictionary.demote();
18604        let (before, after) = dictionary.recharge(Some(&profile));
18605        assert_eq!(before, grown);
18606        // What stays is the ends, the counts and the blocks not yet written, which a load writes
18607        // as it goes, so here the drop is the hash tables and the check hashes.
18608        assert!(after < grown - grown / 4, "the lookup is let go of: {after} of {grown}");
18609        assert_eq!(profile.held(), after, "the profile was told about the drop");
18610        assert!(dictionary.code("one more").is_err(), "a demoted dictionary takes no values");
18611
18612        dictionary.demote();
18613        assert_eq!(
18614            dictionary.recharge(Some(&profile)),
18615            (after, after),
18616            "demoting twice is a no-op"
18617        );
18618        assert_eq!(dictionary.values(), 50_000, "the values coded before stay");
18619    }
18620}