Skip to main content

rudb_native/
lib.rs

1//! Rudb's single-file columnar snapshot format.
2//!
3//! A committed directory names independently readable column pages. It has two levels: a catalog
4//! directory naming every table in the file, which is what a footer slot points at and what opening
5//! a database reads, and one directory per table under it holding that table's stripes, pages and
6//! statistics. One slot write publishes all of them, so a commit is atomic across tables.
7//!
8//! This version handles scalar columns; the file header has two generation slots so an unfinished
9//! replacement directory cannot hide the last complete one. See
10//! `spec/storage-v3/12-many-tables-in-one-file.md`.
11//!
12//! # Parts and stripes
13//!
14//! A part is one appended chunk, which is a thousand rows, and it is the unit a scan decodes and
15//! hands to the pipeline. A stripe is sixty four parts, and it is the unit the directory describes
16//! and the unit the file is laid out in: one page per column per stripe, holding that column's
17//! sixty four part payloads end to end.
18//!
19//! The two are separate because they are sized by different pressures. A part wants to be small
20//! because it is a vector and vectors live in cache. A stripe wants to be large because everything
21//! the directory holds is per stripe and the directory is one buffer that has to be read and
22//! decoded before a single row can be answered. A hundred million rows of the hundred and five
23//! column ClickBench table is ninety seven thousand parts, and a directory with a page entry and a
24//! pair of bounds per part per column is several hundred megabytes, which is what made that load
25//! fail before this split existed. Sixty four parts to a stripe divides that by sixty four.
26//!
27//! Where the parts of a page start is not in the directory either, for the same reason. Each
28//! stripe writes one index page holding a length and a checksum per part per column, and a reader
29//! preads the sixty four entries belonging to the column it wants. A scan reads the whole column
30//! page once and slices it; a sparse row fetch reads the index entries and then only the part it
31//! needs.
32
33#![forbid(unsafe_code)]
34
35use std::borrow::Cow;
36use std::cmp::Ordering;
37use std::collections::{BTreeMap, HashMap, VecDeque};
38use std::fs::File;
39use std::mem::{size_of, size_of_val};
40use std::path::Path;
41use std::slice;
42use std::sync::atomic::{AtomicBool, AtomicUsize, Ordering as Atomic};
43use std::sync::{Arc, Condvar, Mutex, OnceLock, PoisonError, Weak};
44
45use rudb_common::bounds::{self, Bound, Op, scaled_as};
46use rudb_common::{Clustering, Error, Field, LogicalType, PhysicalType, Result, Value, Width};
47use rudb_encoding::{bitpack, chooser, integer, string};
48use rudb_io::{Filesystem, OpenMode, RealFilesystem};
49use rudb_metrics::{LoadProfile, Stage};
50use rudb_storage::sieve::Sieve;
51use rudb_storage::{Probe, Range, Zone};
52use rudb_vector::string::StringColumn;
53use rudb_vector::validity::Validity;
54use rudb_vector::{Buffer, Chunk, Data, Packed, TextSource, Vector, search_below};
55
56mod distinct;
57pub mod graph;
58pub mod host;
59mod prepare;
60use prepare::Lent;
61pub mod section;
62pub mod stats;
63mod zones;
64
65pub use prepare::{DICTIONARY_CAP_BYTES, Merged, Merger, Paged, Prepared, Preparer};
66pub use section::Section;
67pub use zones::{Common, Stripes, ascending, distincts};
68
69const MAGIC: &[u8; 8] = b"RUDBNV10";
70const DIRECTORY: &[u8; 8] = b"RUDBDI10";
71const CATALOG: &[u8; 8] = b"RUDBCA10";
72const NONZERO_COUNTS: &[u8; 8] = b"RUDBNZ10";
73const AGGREGATE_SUMS: &[u8; 8] = b"RUDBAG10";
74const DISTINCT_COUNTS: &[u8; 8] = b"RUDBDC10";
75const INTEGER_EXTREMES: &[u8; 8] = b"RUDBEX10";
76const COMPLETE_FREQUENCIES: &[u8; 8] = b"RUDBFQ10";
77const MAX_CATALOG_FREQUENCIES: usize = 64;
78const FORMAT: u32 = 29;
79
80/// Formats this build can open.
81///
82/// More than one, for the first time, and the reason is spec/graph/10-milestones.md's G1 exit
83/// criterion: a build with the section table in it has to open a file written before the section
84/// table existed, unchanged and without a rewrite. Formats 22 and 23 are those files, and both read
85/// as a table with an empty section table, which is exactly what section 3.1 says a table with no
86/// graph sections is.
87///
88/// All three of the older ones are readable for the same reason. What took the format from 22 to 23
89/// was tags for fourteen more column types, and a file written before that has none of them in it,
90/// so nothing in an older file is a tag this build cannot read. What took it from 23 to 24 is the
91/// section table, which a file written before it simply does not have. What takes it from 24 to 25
92/// is the view section on the end of the catalog, which an older file does not have either, and a
93/// catalog that ends where the tables end reads as a catalog with no views in it. What takes it
94/// from 25 to 26 is that a global dictionary's payload blocks now say where they are, and a file
95/// written before that has them behind one another, which [`open_global_dictionary`] reads by
96/// turning the ends it finds into the same places the newer files name outright. What takes it
97/// from 26 to 27 is that those blocks are written into the file as the load goes, between the
98/// stripes, rather than behind the dictionary's index at the end, so the dictionary's page is the
99/// index and the sorted order and nothing else. A format 26 file has its blocks inside the page,
100/// and the reader tells the two apart by whether the page has room left over for them.
101///
102/// Format 28 adds per-payload-block substring signatures to global string dictionaries. Older
103/// files have no signatures and use the ordinary exact string filter. Format 29 makes each
104/// signature four times as wide, which a dictionary says with [`DICTIONARY_WIDE_GRAMS`], and a
105/// format 28 file is read with the narrow ones it has.
106///
107/// This is not a general compatibility promise. Seven formats are readable because there was a
108/// specific reason for each, and the list shrinks again the moment the older ones stop being worth
109/// carrying.
110const READABLE: &[u32] = &[22, 23, 24, 25, 26, 27, 28, FORMAT];
111
112const HEADER: u64 = 80;
113const SLOT_BYTES: usize = 28;
114const MAX_PAGE: usize = 256 * 1024 * 1024;
115const MAX_DIRECTORY: usize = 128 * 1024 * 1024;
116const FREQUENCIES_V2: &[u8; 8] = b"RUDBFQ2\0";
117const FREQUENCIES: &[u8; 8] = b"RUDBFQ3\0";
118/// Inline spellings for string entries in the bounded frequency synopsis.
119///
120/// A planner usually asks about one literal such as the empty string. Without this block it opens
121/// a multi-million-value global dictionary and visits the payload blocks of every retained entry
122/// merely to compare that literal with at most 512 heavy hitters. The spellings are already in
123/// memory while the writer sorts the dictionary, so storing this bounded copy makes planning a
124/// directory read and leaves the dictionary unopened.
125const FREQUENCY_TEXTS: &[u8; 8] = b"RUDBFT1\0";
126/// Certified host aggregate state for the version-one anchored replacement expression.
127const HOST_GROUPS: &[u8; 8] = b"RUDBHG1\0";
128/// Exact leading counts for a bounded pair of dictionary-backed grouping keys.
129///
130/// This is a separate optional directory block rather than another frequency format. Readers that
131/// predate it still understand every earlier directory, and a table without a pair worth keeping
132/// writes no block at all.
133const PAIR_FREQUENCIES: &[u8; 8] = b"RUDBPF1\0";
134/// The clustering declaration, written after the frequencies and only when there is one.
135///
136/// No format bump for this, which is the convention the frequency section set in #728: a new
137/// optional trailing section with its own magic leaves every file that does not use it byte for
138/// byte what it was, and the version is bumped for a change to a layout that already exists, as
139/// #1029 did. A file with no declaration is the same bytes this build wrote yesterday.
140///
141/// The width byte in this block gained a fifth value for #1285, for a declaration that leaves the
142/// bucket to the row count, and that did not bump the format either. It is the one case where the
143/// reasoning needs saying out loud, because it is a new value in a layout that already exists
144/// rather than a new section. A build without it reading one of these says `clustering width
145/// tag differs` and refuses the table, which is what that message was written for. Bumping the
146/// format instead would have made every file this build writes unreadable to an older one, whether
147/// it has a declaration in it or not, to warn about a case that only arises when it does.
148const CLUSTERING: &[u8; 8] = b"RUDBCL1\0";
149/// The string columns whose global dictionary stopped taking values partway through the load.
150///
151/// Section 5.5 of the encoding spec: a column whose stripes are nearly all new values, or the
152/// fastest growing one once the dictionaries together pass their cap, stops adding to its
153/// dictionary, and every stripe after that is written plainly. The stripes before keep their codes,
154/// so the dictionary is still written and still decodes them, but it no longer holds every value of
155/// the column, and nothing that reads it as if it did can be trusted: not the distinct count, not
156/// the frequencies, not the sorted order's first and last value, and not the codes as a group key
157/// or a membership index. A reader that finds a column named here decodes its coded pages to plain
158/// strings and answers everything else the way it answers a column with no dictionary.
159///
160/// Same convention as [`CLUSTERING`], written only when a column was demoted, so a file with none
161/// is the bytes it always was. A build that predates it refuses a file that has one with
162/// `directory extension magic differs`, which is the right answer, because that build would trust
163/// the dictionary.
164///
165/// A stripe written after the demotion has no membership index for the column. Its slot in the
166/// stripe is written as a page of no bytes, which no real membership index is, since the smallest
167/// one holds its code count.
168const DEMOTED: &[u8; 8] = b"RUDBDM1\0";
169/// The graph section table, written after the clustering declaration and written even when empty.
170///
171/// Same convention and the same reason as the block above it, with one difference: this one is
172/// always there, so a file written by this build says which sections it has rather than leaving a
173/// reader to infer it from where the bytes ran out. Section 3.1 of the graph spec is what makes
174/// that safe to add without a format bump, because a table with no sections answers every query
175/// the way it did before, only without the graph path.
176const SECTIONS: &[u8; 8] = b"RUDBSE1\0";
177/// How many bytes of each column's global dictionary live outside its page, written only when any do.
178///
179/// From format 27 a dictionary's payload blocks are written into the file while the load runs, so
180/// they sit between the stripes and the dictionary's page covers only its index and sorted order.
181/// Nothing needs the total to read the file, because the index names every block. It is here for
182/// what a file costs a column, which [`Reader::layout`] and the statistics budget both report, and
183/// which would otherwise lose most of the bytes of every large string column.
184const DICTIONARY_PAYLOADS: &[u8; 8] = b"RUDBDP1\0";
185
186/// The most sections one table's directory may name.
187///
188/// A relationship contributes at most three sections, so this bounds a table at a few thousand
189/// relationships, which is far past anything a schema has. The bound is here so that a torn
190/// directory naming four billion of them is refused at decode rather than turned into an
191/// allocation, the same reason the extent count has one.
192const MAX_SECTIONS: usize = 4096;
193const FREQUENCY_CANDIDATES: usize = 32_768;
194const FREQUENCY_ENTRIES: usize = 512;
195const FREQUENCY_BUILD_RANK: usize = 10;
196const FREQUENCY_ORDINALS: usize = 131_072;
197const MAX_PAIR_FREQUENCIES: usize = 1024;
198/// The most exact heavy-hitter text one column may copy into the directory.
199///
200/// A column with unusually large leading values keeps the old code-only synopsis instead. The
201/// optimization must never turn a valid load into a directory-size failure.
202const FREQUENCY_TEXT_BUDGET: usize = 1024 * 1024;
203/// The most threads the two per column passes at the end of a commit are spread over.
204///
205/// A table like `hits` has ninety numeric columns, so on a machine with more cores than this the
206/// cap is what decides how long the frequencies take rather than the columns are. It is here at all
207/// because each worker holds a candidate table and a decoded part, and a hundred of those at once
208/// on a narrow machine would be worse than waiting.
209const MAX_FREQUENCY_WORKERS: usize = 32;
210
211/// How many threads the passes at the end of a commit are spread over on this machine.
212fn close_workers() -> usize {
213    std::thread::available_parallelism().map_or(1, usize::from).min(MAX_FREQUENCY_WORKERS)
214}
215
216/// How many bytes the columns closing at the same time may hold between them.
217///
218/// Closing a global dictionary decodes every value it holds, sorts them and drops them, and #1356
219/// took the columns one at a time so that five of them decoded at once were not the peak of a load.
220/// A numeric column's frequencies hold a candidate table and, past it, an exact set of its distinct
221/// values that reaches 512 MiB. The two used to run side by side with only the dictionaries under a
222/// bound, and on the ClickBench `hits` 10M load the close took a load that had held 3.1 GB to 4.8
223/// GB. A column is taken while the ones already closing leave room for it under this, and always
224/// when nothing else is closing, so every dictionary of `hits` at 10M rows closes at once and `URL`
225/// at 100M, which is past this alone, still closes on its own.
226const CLOSE_BYTES: usize = 1 << 30;
227
228/// What a numeric column's frequencies hold before its exact distinct set, which is the candidate
229/// table, its recount and the page being read, with room to spare.
230const NUMERIC_CLOSE_BYTES: usize = 4 << 20;
231
232/// The most threads one stripe's encode is spread over.
233///
234/// Higher than the frequency cap because this is the load itself rather than a pass at the end of
235/// it, and the work is one column of sixty four parts, which is large enough that a thread that
236/// takes one is not a thread that was started for nothing. A machine with more cores than this has
237/// the rest of them on the Parquet read, which is still one thread and is the other half of #808.
238const MAX_ENCODE_WORKERS: usize = 32;
239
240/// How much a writer appends before it asks the kernel to start writing it to the device.
241///
242/// Without it every byte of a load waits in the page cache for the sync at the commit, and that
243/// sync was 1.3 to 1.7 s of a ClickBench `hits` 10M load of 8 to 9 s on the 32 core box. With it
244/// the device writes while the load is still encoding. Thirty two megabytes is a few stripes of
245/// `hits`, big enough that the call costs nothing next to the write, and small enough that what
246/// is left for the commit is one stretch.
247const WRITEBACK_STRETCH: u64 = 32 << 20;
248
249/// The most bytes one column of one part may spend on a membership sieve.
250///
251/// A part is a thousand rows, so a filter sized for every one of them being distinct is about
252/// thirteen hundred bytes and this never binds in practice. It is here so that a part that somehow
253/// arrives much wider than a vector cannot put an unbounded index in the file. What does bind is the
254/// rule in `Writer::encode_pages` that a sieve may not be as large as the part it indexes, which is a cap
255/// per column rather than one number for the whole file.
256const SIEVE_BUDGET: usize = 8 * 1024;
257
258/// The most bytes one end of a per part range may spend on a string.
259///
260/// A bound is allowed to be wider than the truth and never narrower, so a long string is cut down to
261/// this many bytes for the low end and cut down and then stepped up for the high end. The reason for
262/// a cap at all is that there are nine hundred and seventy four parts of a hundred and five columns
263/// in a million rows of ClickBench and `URL` runs to hundreds of bytes, so keeping every end whole
264/// would put more in the directory than the skipping is worth. Twenty four bytes is past the point
265/// where two URLs of the same site still look alike.
266const PART_BOUND_BYTES: usize = 24;
267
268fn io(error: std::io::Error) -> Error {
269    Error::io(error.to_string())
270}
271
272fn invalid(message: &str) -> Error {
273    Error::invalid_input(format!("invalid rudb native file: {message}"))
274}
275
276/// Adds a sequence of byte counts without an overflow the caller has to think about.
277fn sum(counts: impl Iterator<Item = u64>) -> u64 {
278    counts.fold(0, u64::saturating_add)
279}
280
281/// One column's span out of a per column list, or zero when the list is shorter than the column.
282fn span_bytes(spans: &[Span], at: usize) -> u64 {
283    spans.get(at).map_or(0, |span| u64::from(span.length))
284}
285
286/// One column's page out of a per column list, or zero when that column has no page at all.
287fn page_bytes(pages: &[Option<Page>], at: usize) -> u64 {
288    pages.get(at).and_then(Option::as_ref).map_or(0, Page::bytes)
289}
290
291/// Everything one column's global dictionary costs the file, its page and the blocks outside it.
292fn dictionary_bytes(table: &Table, at: usize) -> u64 {
293    page_bytes(&table.dictionaries, at)
294        .saturating_add(table.dictionary_payloads.get(at).copied().unwrap_or(0))
295}
296
297/// The xxHash64 of `bytes`, which is what every span this format stores is checked against.
298///
299/// It walks the input as chunks rather than as offsets into it, and that is the only thing about it
300/// worth a comment. The offset form reads `bytes[at..at + 8]`, and neither the slicing nor the
301/// `try_into` behind it can be proved in range by a compiler that does not know where `at` stopped,
302/// so each of the four lanes paid for a bounds check and a length check on every thirty two bytes.
303/// A chunk carries its own length, so both fold away and the loop is the multiplies and rotates it
304/// was meant to be. That loop runs over every byte of every span a query reads, which on ClickBench
305/// 8 is about five percent of the query.
306fn checksum(bytes: &[u8]) -> u64 {
307    seeded_checksum(bytes, 0)
308}
309
310/// A hundred and twenty eight bit name for `bytes`, as two xxHash64 walks under different seeds,
311/// with the format this build writes folded in so that a name made by one format is never taken
312/// for the name of a file in another.
313///
314/// For a caller outside this crate that has to name a file by what went into it, which is what a
315/// Parquet mirror's key is. See the global dictionary's use of the same pair for the arithmetic.
316#[must_use]
317pub fn content_name(bytes: &[u8]) -> u128 {
318    let seed = u64::from(FORMAT);
319    u128::from(seeded_checksum(bytes, seed)) << 64 | u128::from(seeded_checksum(bytes, !seed))
320}
321
322/// [`content_name`] of bytes that arrive in pieces, which gives the same name as the pieces joined.
323///
324/// A Parquet mirror is named for the file's footer, and that is 930 KB on the ten million row
325/// ClickBench file. Read whole to be hashed it is a freed megabyte in every process that opens the
326/// mirror, which the allocator keeps. Read a window at a time it is a window.
327#[derive(Debug, Clone)]
328pub struct ContentNamer {
329    seeds: [u64; 2],
330    lanes: [[u64; 4]; 2],
331    held: [u8; 32],
332    filled: usize,
333    length: u64,
334}
335
336impl Default for ContentNamer {
337    fn default() -> Self {
338        let seed = u64::from(FORMAT);
339        let seeds = [seed, !seed];
340        let lanes = seeds.map(|seed| {
341            [
342                seed.wrapping_add(XXH_P1).wrapping_add(XXH_P2),
343                seed.wrapping_add(XXH_P2),
344                seed,
345                seed.wrapping_sub(XXH_P1),
346            ]
347        });
348        Self { seeds, lanes, held: [0; 32], filled: 0, length: 0 }
349    }
350}
351
352impl ContentNamer {
353    /// Takes the next piece.
354    pub fn update(&mut self, mut bytes: &[u8]) {
355        self.length += bytes.len() as u64;
356        if self.filled > 0 {
357            let take = (32 - self.filled).min(bytes.len());
358            self.held[self.filled..self.filled + take].copy_from_slice(&bytes[..take]);
359            self.filled += take;
360            bytes = &bytes[take..];
361            if self.filled < 32 {
362                return;
363            }
364            let block = self.held;
365            self.lanes.iter_mut().for_each(|lanes| checksum_block(lanes, &block));
366            self.filled = 0;
367        }
368        let mut blocks = bytes.chunks_exact(32);
369        for block in blocks.by_ref() {
370            self.lanes.iter_mut().for_each(|lanes| checksum_block(lanes, block));
371        }
372        let rest = blocks.remainder();
373        self.held[..rest.len()].copy_from_slice(rest);
374        self.filled = rest.len();
375    }
376
377    /// The name of everything taken so far.
378    #[must_use]
379    pub fn finish(&self) -> u128 {
380        let rest = &self.held[..self.filled];
381        let [first, second] = [0, 1].map(|at| {
382            if self.length < 32 {
383                checksum_tail(self.seeds[at].wrapping_add(XXH_P5).wrapping_add(self.length), rest)
384            } else {
385                finish_checksum(self.lanes[at], rest, self.length)
386            }
387        });
388        u128::from(first) << 64 | u128::from(second)
389    }
390}
391
392/// The xxHash64 of `bytes` started from `seed`, which is the same walk with a different beginning.
393///
394/// A seed is here for one caller: a global dictionary decides whether two values are the same by
395/// their hashes rather than by their bytes, and one sixty four bit hash is not enough to do that
396/// with. Twenty million distinct values collide on sixty four bits about once in a hundred thousand
397/// loads, which for a wrong answer is far too often. Two hashes of the same value under different
398/// seeds are independent, so the pair is a hundred and twenty eight bits and the same arithmetic
399/// puts that at around one in 1e24.
400fn seeded_checksum(bytes: &[u8], seed: u64) -> u64 {
401    // Asked for before the loop rather than after it, because a `ChunksExact` settles what it
402    // cannot divide when it is built and hands back the same tail whether it has been walked or not.
403    let mut blocks = bytes.chunks_exact(32);
404    let rest = blocks.remainder();
405    if bytes.len() < 32 {
406        return checksum_tail(seed.wrapping_add(XXH_P5).wrapping_add(bytes.len() as u64), rest);
407    }
408    let mut lanes = [
409        seed.wrapping_add(XXH_P1).wrapping_add(XXH_P2),
410        seed.wrapping_add(XXH_P2),
411        seed,
412        seed.wrapping_sub(XXH_P1),
413    ];
414    for block in blocks.by_ref() {
415        checksum_block(&mut lanes, block);
416    }
417    finish_checksum(lanes, rest, bytes.len() as u64)
418}
419
420const XXH_P1: u64 = 11_400_714_785_074_694_791;
421const XXH_P2: u64 = 14_029_467_366_897_019_727;
422const XXH_P3: u64 = 1_609_587_929_392_839_161;
423const XXH_P4: u64 = 9_650_029_242_287_828_579;
424const XXH_P5: u64 = 2_870_177_450_012_600_261;
425
426fn checksum_round(state: u64, word: u64) -> u64 {
427    state.wrapping_add(word.wrapping_mul(XXH_P2)).rotate_left(31).wrapping_mul(XXH_P1)
428}
429
430fn checksum_word(chunk: &[u8]) -> u64 {
431    u64::from_le_bytes(chunk.try_into().expect("eight checksum bytes"))
432}
433
434/// One thirty two byte block into the four lanes.
435fn checksum_block(lanes: &mut [u64; 4], block: &[u8]) {
436    for (lane, chunk) in lanes.iter_mut().zip(block.chunks_exact(8)) {
437        *lane = checksum_round(*lane, checksum_word(chunk));
438    }
439}
440
441/// The lanes after every whole block, folded together with what was left over and the length.
442fn finish_checksum(lanes: [u64; 4], rest: &[u8], length: u64) -> u64 {
443    let merge = |state: u64, lane: u64| {
444        (state ^ checksum_round(0, lane)).wrapping_mul(XXH_P1).wrapping_add(XXH_P4)
445    };
446    let [one, two, three, four] = lanes;
447    let combined = one
448        .rotate_left(1)
449        .wrapping_add(two.rotate_left(7))
450        .wrapping_add(three.rotate_left(12))
451        .wrapping_add(four.rotate_left(18));
452    let hash = merge(merge(merge(merge(combined, one), two), three), four);
453    checksum_tail(hash.wrapping_add(length), rest)
454}
455
456/// The fewer than thirty two bytes after the last whole block, and the final mix.
457fn checksum_tail(mut hash: u64, mut rest: &[u8]) -> u64 {
458    let mut words = rest.chunks_exact(8);
459    for chunk in words.by_ref() {
460        hash ^= checksum_round(0, checksum_word(chunk));
461        hash = hash.rotate_left(27).wrapping_mul(XXH_P1).wrapping_add(XXH_P4);
462    }
463    rest = words.remainder();
464    if rest.len() >= 4 {
465        let (head, tail) = rest.split_at(4);
466        let quarter = u32::from_le_bytes(head.try_into().expect("four checksum bytes"));
467        hash ^= u64::from(quarter).wrapping_mul(XXH_P1);
468        hash = hash.rotate_left(23).wrapping_mul(XXH_P2).wrapping_add(XXH_P3);
469        rest = tail;
470    }
471    for &byte in rest {
472        hash ^= u64::from(byte).wrapping_mul(XXH_P5);
473        hash = hash.rotate_left(11).wrapping_mul(XXH_P1);
474    }
475    hash ^= hash >> 33;
476    hash = hash.wrapping_mul(XXH_P2);
477    hash ^= hash >> 29;
478    hash = hash.wrapping_mul(XXH_P3);
479    hash ^ (hash >> 32)
480}
481
482/// The checksum of `length` bytes of `file` from `offset`, read [`DIRECTORY_WINDOW`] at a time.
483///
484/// The same xxHash64 as [`checksum`], carried across reads rather than over one buffer, so that a
485/// directory can be checked without all of it being in memory at once. The four lanes take whole
486/// thirty two byte blocks, and a read that ends partway through one keeps the tail for the next.
487fn file_checksum(file: &File, offset: u64, length: usize) -> Result<u64> {
488    walk_checksummed(file, offset, length, DIRECTORY_WINDOW, |_| Ok(()))
489}
490
491/// Reads `length` bytes at `offset` a window at a time, hands each window to `each`, and answers
492/// the checksum of all of them.
493///
494/// `window` is a multiple of thirty two, so every window but the last is whole blocks of the hash
495/// and nothing has to be carried from one read to the next.
496fn walk_checksummed(
497    file: &File,
498    offset: u64,
499    length: usize,
500    window: usize,
501    mut each: impl FnMut(&[u8]) -> Result<()>,
502) -> Result<u64> {
503    debug_assert!(window % 32 == 0 && window > 0, "a window is whole blocks of the hash");
504    if length < 32 {
505        let mut bytes = vec![0; length];
506        read_at(file, offset, &mut bytes)?;
507        each(&bytes)?;
508        return Ok(checksum(&bytes));
509    }
510    let mut lanes = [XXH_P1.wrapping_add(XXH_P2), XXH_P2, 0, 0_u64.wrapping_sub(XXH_P1)];
511    let mut buffer = vec![0; window.min(length)];
512    let mut read = 0;
513    let (mut whole, mut filled) = (0, 0);
514    while read < length {
515        filled = buffer.len().min(length - read);
516        read_at(file, offset + read as u64, &mut buffer[..filled])?;
517        read += filled;
518        each(&buffer[..filled])?;
519        whole = filled / 32 * 32;
520        for block in buffer[..whole].chunks_exact(32) {
521            checksum_block(&mut lanes, block);
522        }
523    }
524    Ok(finish_checksum(lanes, &buffer[whole..filled], length as u64))
525}
526
527#[derive(Debug, Clone, Copy)]
528struct Slot {
529    offset: u64,
530    length: u32,
531    generation: u64,
532    hash: u64,
533}
534
535impl Slot {
536    fn bytes(self) -> [u8; SLOT_BYTES] {
537        let mut result = [0; SLOT_BYTES];
538        result[..8].copy_from_slice(&self.offset.to_le_bytes());
539        result[8..12].copy_from_slice(&self.length.to_le_bytes());
540        result[12..20].copy_from_slice(&self.generation.to_le_bytes());
541        result[20..28].copy_from_slice(&self.hash.to_le_bytes());
542        result
543    }
544
545    fn read(bytes: &[u8]) -> Self {
546        Self {
547            offset: u64::from_le_bytes(bytes[..8].try_into().expect("eight bytes")),
548            length: u32::from_le_bytes(bytes[8..12].try_into().expect("four bytes")),
549            generation: u64::from_le_bytes(bytes[12..20].try_into().expect("eight bytes")),
550            hash: u64::from_le_bytes(bytes[20..28].try_into().expect("eight bytes")),
551        }
552    }
553}
554
555#[derive(Debug, Clone, Copy)]
556struct Page {
557    offset: u64,
558    length: u32,
559    hash: u64,
560}
561
562impl Page {
563    /// How much of the file this page takes, for [`Reader::layout`].
564    fn bytes(&self) -> u64 {
565        u64::from(self.length)
566    }
567}
568
569#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
570enum FrequencyValue {
571    Null,
572    Integer(i128),
573    Code(u32),
574}
575
576/// A table keyed by the sixty four bits of the values the numeric frequency pass counts.
577///
578/// Every integer of every numeric column goes through one of these at least once when a table
579/// closes, and with the standard hasher that was a fifth of the close on its own, all of it SipHash
580/// guarding against an attacker who would have to choose the rows of the file being written.
581type FrequencyMap<V> = HashMap<u64, V, Spread>;
582
583/// The first pass of [`Writer::numeric_frequency`]: a Misra-Gries candidate table keyed by a value's
584/// sixty four bits, with the null counted beside it.
585///
586/// The table is an open addressed one of its own rather than a `HashMap`. On a column that is near
587/// unique, which `hits` has a dozen of, nearly every row is a value the table has not seen, and a
588/// `HashMap` spent a lookup and then a second hash and probe to insert it, and a `retain` over every
589/// bucket each time the table filled. Those were 6 percent of the CPU of loading the 10m ClickBench
590/// file, and the slowest of those columns decided how long the whole frequency step took. Here a
591/// value is found or given the empty slot it stopped at in one probe, and a decrement rebuilds the
592/// table from the few candidates that outlive it.
593///
594/// What the table holds after a stream of rows is the same set of counts either way, since that is
595/// fixed by the algorithm and not by where the counts live.
596#[derive(Debug)]
597struct Candidates {
598    /// A power of two number of slots, at most half of them in use. A count of zero is an empty
599    /// slot, which no candidate ever is, because one whose count reaches zero is dropped.
600    slots: Vec<Candidate>,
601    held: usize,
602    nulls: u32,
603    decrements: u64,
604    /// The candidates that outlive a decrement, kept so that each decrement is not an allocation.
605    survivors: Vec<Candidate>,
606}
607
608/// One slot of [`Candidates`], the value's bits beside its count so a probe reads one line.
609#[derive(Debug, Default, Clone, Copy)]
610struct Candidate {
611    bits: u64,
612    count: u32,
613}
614
615/// The slots a candidate table starts with, grown by doubling as it fills.
616const FIRST_CANDIDATE_SLOTS: usize = 64;
617
618impl Default for Candidates {
619    fn default() -> Self {
620        Self {
621            slots: vec![Candidate::default(); FIRST_CANDIDATE_SLOTS],
622            held: 0,
623            nulls: 0,
624            decrements: 0,
625            survivors: Vec::new(),
626        }
627    }
628}
629
630/// Starts the exact distinct count of a column whose candidate table is about to turn a value away.
631///
632/// Until then nothing was decremented and the table holds every value seen, so the set starts as
633/// those values and `ended`, the run about to be added. It is the caller's to insert every value
634/// after that. A column the close decided not to count gets a set that has already given up.
635fn count_from_full(
636    first: &Candidates,
637    distinct: &mut Option<distinct::ExactDistinct>,
638    ended: Option<u64>,
639    counted: bool,
640) {
641    if distinct.is_some() || !first.full() {
642        return;
643    }
644    if !counted {
645        *distinct = Some(distinct::ExactDistinct::declined());
646        return;
647    }
648    let mut set = distinct::ExactDistinct::new();
649    for held in first.held_bits() {
650        set.insert(held);
651    }
652    if let Some(ended) = ended {
653        set.insert(ended);
654    }
655    *distinct = Some(set);
656}
657
658impl Candidates {
659    /// Counts `times` rows of `bits` and ends in the state `times` rows counted one at a time would.
660    ///
661    /// A value already held, or one there is room to hold, takes the whole run at once, because
662    /// every row after the first would find it held. A value the full table turns away goes a row
663    /// at a time, because each of its rows decrements every candidate and one of those decrements
664    /// can free the place the next row takes.
665    fn add(&mut self, bits: Option<u64>, mut times: u32) {
666        while times > 0 {
667            let room = self.held + usize::from(self.nulls != 0) < FREQUENCY_CANDIDATES;
668            match bits {
669                Some(bits) => {
670                    let (at, found) = self.find(bits);
671                    if found {
672                        self.slots[at].count = self.slots[at].count.saturating_add(times);
673                        return;
674                    }
675                    if room {
676                        self.place(at, bits, times);
677                        return;
678                    }
679                }
680                None if self.nulls != 0 => {
681                    self.nulls = self.nulls.saturating_add(times);
682                    return;
683                }
684                None if room => {
685                    self.nulls = times;
686                    return;
687                }
688                None => {}
689            }
690            self.decrement();
691            times -= 1;
692        }
693    }
694
695    /// Whether a value not held yet would decrement the table rather than take a slot.
696    fn full(&self) -> bool {
697        self.held + usize::from(self.nulls != 0) >= FREQUENCY_CANDIDATES
698    }
699
700    /// The values held, in no particular order.
701    fn held_bits(&self) -> impl Iterator<Item = u64> + '_ {
702        self.slots.iter().filter(|slot| slot.count != 0).map(|slot| slot.bits)
703    }
704
705    /// The slot holding `bits` and `true`, or the empty slot a search for it stopped at and `false`.
706    fn find(&self, bits: u64) -> (usize, bool) {
707        let mask = self.slots.len() - 1;
708        let mut at = home(bits, self.slots.len());
709        loop {
710            let slot = self.slots[at];
711            if slot.count == 0 {
712                return (at, false);
713            }
714            if slot.bits == bits {
715                return (at, true);
716            }
717            at = (at + 1) & mask;
718        }
719    }
720
721    /// Where `bits` is held, for the recount, which reads the table without changing it.
722    fn position(&self, bits: u64) -> Option<usize> {
723        match self.find(bits) {
724            (at, true) => Some(at),
725            (_, false) => None,
726        }
727    }
728
729    /// Puts a new candidate in the empty slot `at`, which a search for it just stopped at, doubling
730    /// the table first when that would fill more than half of it.
731    fn place(&mut self, at: usize, bits: u64, count: u32) {
732        let at = if (self.held + 1) * 2 > self.slots.len() {
733            let wider = self.slots.len() * 2;
734            let old = std::mem::replace(&mut self.slots, vec![Candidate::default(); wider]);
735            for slot in old.into_iter().filter(|slot| slot.count != 0) {
736                let (to, _) = self.find(slot.bits);
737                self.slots[to] = slot;
738            }
739            self.find(bits).0
740        } else {
741            at
742        };
743        self.slots[at] = Candidate { bits, count };
744        self.held += 1;
745    }
746
747    /// Takes one from every candidate and the null, dropping the ones that reach zero.
748    fn decrement(&mut self) {
749        let mut survivors = std::mem::take(&mut self.survivors);
750        survivors.clear();
751        survivors.extend(
752            self.slots
753                .iter()
754                .filter(|slot| slot.count > 1)
755                .map(|slot| Candidate { bits: slot.bits, count: slot.count - 1 }),
756        );
757        self.slots.fill(Candidate::default());
758        self.held = survivors.len();
759        for &slot in &survivors {
760            let (at, _) = self.find(slot.bits);
761            self.slots[at] = slot;
762        }
763        self.survivors = survivors;
764        self.nulls = self.nulls.saturating_sub(1);
765        self.decrements = self.decrements.saturating_add(1);
766    }
767
768    /// Every candidate's bits and count, in no particular order.
769    fn pairs(&self) -> impl Iterator<Item = (u64, u32)> + '_ {
770        self.slots.iter().filter(|slot| slot.count != 0).map(|slot| (slot.bits, slot.count))
771    }
772}
773
774/// The slot a search for `bits` starts at in a table of `slots`, a power of two.
775///
776/// The top bits of a multiply by the golden ratio, which every bit of the value reaches, so a
777/// timestamp column whose values are all multiples of a million still spreads over the table.
778fn home(bits: u64, slots: usize) -> usize {
779    (bits.wrapping_mul(0x9E37_79B9_7F4A_7C15) >> (64 - slots.trailing_zeros())) as usize
780}
781
782/// Equal rows in a row, gathered so they are counted once.
783#[derive(Debug, Default)]
784struct Run {
785    bits: Option<u64>,
786    times: u32,
787}
788
789impl Run {
790    /// Adds one row, and hands back the run it ended if it was not the same value.
791    fn push(&mut self, bits: Option<u64>) -> Option<(Option<u64>, u32)> {
792        if self.times != 0 && self.bits == bits && self.times < u32::MAX {
793            self.times += 1;
794            return None;
795        }
796        let ended = self.take();
797        self.bits = bits;
798        self.times = 1;
799        ended
800    }
801
802    /// The run being gathered, if there is one, leaving none.
803    fn take(&mut self) -> Option<(Option<u64>, u32)> {
804        let times = std::mem::take(&mut self.times);
805        (times != 0).then_some((self.bits, times))
806    }
807}
808
809/// Builds the hasher for [`FrequencyMap`].
810#[derive(Debug, Default, Clone, Copy)]
811struct Spread;
812
813impl std::hash::BuildHasher for Spread {
814    type Hasher = SpreadHasher;
815
816    fn build_hasher(&self) -> SpreadHasher {
817        SpreadHasher(0)
818    }
819}
820
821/// Folds each word in with a full width multiply whose two halves are xored together.
822///
823/// A plain multiply leaves the low bits of the hash as poor as the low bits of the key, and the
824/// table picks its bucket from the low bits, so a timestamp column, whose values are all multiples
825/// of a million microseconds, would pile into a sixty fourth of the buckets. Folding the high half
826/// of the product back in is what gives the low bits the whole word.
827#[derive(Debug)]
828struct SpreadHasher(u64);
829
830impl SpreadHasher {
831    fn mix(&mut self, word: u64) {
832        let product = u128::from(self.0 ^ word) * 0x9E37_79B9_7F4A_7C15_u128;
833        self.0 = (product as u64) ^ ((product >> 64) as u64);
834    }
835}
836
837impl std::hash::Hasher for SpreadHasher {
838    fn write(&mut self, bytes: &[u8]) {
839        for part in bytes.chunks(8) {
840            let mut word = [0; 8];
841            word[..part.len()].copy_from_slice(part);
842            self.mix(u64::from_le_bytes(word));
843        }
844    }
845
846    fn write_u32(&mut self, value: u32) {
847        self.mix(u64::from(value));
848    }
849
850    fn write_u64(&mut self, value: u64) {
851        self.mix(value);
852    }
853
854    fn write_i128(&mut self, value: i128) {
855        self.mix(value as u64);
856        self.mix((value >> 64) as u64);
857    }
858
859    fn write_isize(&mut self, value: isize) {
860        self.mix(value as u64);
861    }
862
863    fn finish(&self) -> u64 {
864        self.0
865    }
866}
867
868#[derive(Debug, Clone)]
869struct FrequencyEntry {
870    value: FrequencyValue,
871    count: u64,
872}
873
874/// Exact leading frequencies for one column.
875///
876/// Values outside `entries` occur at most `omitted_max` times. This lets a count-descending TopN
877/// use the synopsis only when its last winner is strictly above every omitted value.
878#[derive(Debug, Clone)]
879struct FrequencySummary {
880    entries: Vec<FrequencyEntry>,
881    omitted_max: u64,
882    ordinals: Vec<u64>,
883    ordinal_entries: Vec<u16>,
884}
885
886#[derive(Debug, Clone)]
887struct PairFrequencyEntry {
888    first_entry: u16,
889    second: Option<u32>,
890    count: u64,
891}
892
893/// Exact leading counts for one numeric frequency anchor and one stable string code space.
894///
895/// `omitted_max` covers both first-key values outside the numeric synopsis and pairs below the
896/// retained prefix. A TopN may therefore use the entries only when its boundary strictly exceeds
897/// this number.
898#[derive(Debug, Clone)]
899struct PairFrequencySummary {
900    first: u16,
901    second: u16,
902    entries: Vec<PairFrequencyEntry>,
903    omitted_max: u64,
904}
905
906/// One column's frequency synopsis, in memory or left where it is in the file.
907///
908/// A writer holds what it counted. A reader leaves every synopsis in the file and reads one back
909/// when a query asks about its column, because they are the largest thing in a directory once they
910/// are decoded, forty eight bytes an entry and nearly twenty thousand entries over `hits`, and
911/// most queries ask about none of them. Where one sits is found at open, by reading it through and
912/// checking it, so a torn synopsis is still refused when the table is opened.
913#[derive(Debug, Clone)]
914enum Frequencies {
915    Held(FrequencySummary),
916    /// Where the synopsis sits, and whether it was written with the value of each ordinal, which
917    /// is what the directory's frequency magic says and the synopsis itself does not.
918    Stored {
919        span: Span,
920        values: bool,
921    },
922}
923
924/// The values one column's frequency synopsis lists, with a bound on everything it left out.
925///
926/// What [`Reader::frequency_prefix`] answers. The counts are exact, and `omitted_max` is how many
927/// rows any value not in the list can hold, which is zero when nothing was left out at all.
928#[derive(Debug, Clone)]
929pub struct FrequencyPrefix {
930    /// Every value the synopsis lists, with the number of rows holding it, count descending.
931    pub entries: Vec<(Value, u64)>,
932    /// How many rows the most common value outside the list holds, and zero for a complete list.
933    pub omitted_max: u64,
934}
935
936/// Sparse row ordinals covered by a numeric frequency candidate set.
937#[derive(Debug, Clone, PartialEq)]
938pub struct FrequencyOccurrences {
939    /// Upper bound for the frequency of every value absent from the fetched rows.
940    pub omitted_max: u64,
941    /// Table-wide row ordinals in ascending order.
942    pub ordinals: Vec<u64>,
943    /// The retained heavy-hitter values named by `anchor_indices`.
944    pub anchors: Vec<Value>,
945    /// The index in `anchors` at each ordinal, or empty for a legacy FQ2 directory.
946    pub anchor_indices: Vec<u16>,
947}
948
949/// Exact grouped counts for a pair of values, in descending count order.
950pub type PairFrequencyCounts = Vec<(Vec<Value>, u64)>;
951
952/// Where one column's page for one stripe sits in the file.
953///
954/// A column page has no checksum of its own because every part inside it carries one, and the
955/// stripe's index page holds those. Checking a part on the way out of the page covers exactly the
956/// bytes a reader is about to decode, and covers them once whether the reader took the whole page
957/// or pulled one part out of the middle of it.
958#[derive(Debug, Clone, Copy, Default)]
959struct Span {
960    offset: u64,
961    length: u32,
962}
963
964/// One optional page for each column of a stripe, holding only the pages that are there.
965///
966/// A stripe has three of these, the membership, sieve and part range pages. As a
967/// `Vec<Option<Page>>` each was thirty two bytes a column whether the page was there or not, and
968/// over the ten million rows of `hits` that is half a megabyte at open for 7171 pages out of 16380
969/// slots. Kept sparse and packed, a page that is there is twenty four bytes and one that is not is
970/// nothing.
971#[derive(Debug, Clone, Default)]
972struct Pages {
973    columns: usize,
974    held: Box<[StripePage]>,
975}
976
977/// A page and the column it is for, packed so that the column sits where the padding was.
978#[derive(Debug, Clone, Copy)]
979struct StripePage {
980    offset: u64,
981    hash: u64,
982    length: u32,
983    column: u32,
984}
985
986impl Pages {
987    /// The pages of `columns` columns, one slot each in column order.
988    fn from_slots(slots: Vec<Option<Page>>) -> Result<Self> {
989        let mut held = Vec::with_capacity(slots.iter().flatten().count());
990        for (column, page) in slots.iter().enumerate() {
991            if let Some(page) = page {
992                let column =
993                    u32::try_from(column).map_err(|_| invalid("too many columns for a page"))?;
994                held.push(StripePage {
995                    offset: page.offset,
996                    hash: page.hash,
997                    length: page.length,
998                    column,
999                });
1000            }
1001        }
1002        Ok(Self { columns: slots.len(), held: held.into_boxed_slice() })
1003    }
1004
1005    /// The page of one column, if it has one.
1006    fn get(&self, column: usize) -> Option<Page> {
1007        let at = self.held.binary_search_by_key(&column, |placed| placed.column as usize).ok()?;
1008        let placed = self.held[at];
1009        Some(Page { offset: placed.offset, length: placed.length, hash: placed.hash })
1010    }
1011
1012    /// One slot per column, in column order, the way the directory writes them.
1013    fn slots(&self) -> impl Iterator<Item = Option<Page>> + '_ {
1014        (0..self.columns).map(|column| self.get(column))
1015    }
1016
1017    /// How much of the file one column's page takes, or zero when it has none.
1018    fn bytes(&self, column: usize) -> u64 {
1019        self.get(column).map_or(0, |page| page.bytes())
1020    }
1021}
1022
1023/// One independently readable stripe of a table.
1024#[derive(Debug, Clone)]
1025pub struct Stripe {
1026    rows: usize,
1027    /// Rows in each part, in source order. Kept in the directory so that mapping a row ordinal to a
1028    /// part, which every sparse fetch does, never reads the file.
1029    parts: Vec<u32>,
1030    /// The index page: one section per column, holding a length and a checksum for every part and
1031    /// then a checksum of the section itself, so that a reader can pread one column's section and
1032    /// still know it is intact.
1033    index: Span,
1034    pages: Vec<Span>,
1035    memberships: Pages,
1036    /// One page per column holding the membership sieve of every part of the stripe, for the
1037    /// columns that have one. A column whose parts all declined a sieve has no page at all.
1038    sieves: Pages,
1039    /// One page per column holding the two ends and the null count of every part of the stripe.
1040    ///
1041    /// The stripe's own `zone` below covers sixty four times as many rows, and on a column that is
1042    /// not the one the rows are ordered by that is the difference between skipping half the file and
1043    /// skipping all but three percent of it. On ClickBench 24 the cutoff the answer settles at
1044    /// leaves eight stripes of sixteen alive and thirty parts of nine hundred and seventy four.
1045    ///
1046    /// A page per column rather than one page for the stripe, so that a query that compares one
1047    /// column reads the ends of that column and not of the hundred and four beside it. Read lazily
1048    /// for the same reason, like the sieves.
1049    part_ranges: Pages,
1050    zone: Zone,
1051}
1052
1053impl Stripe {
1054    /// Number of rows in this stripe.
1055    #[must_use]
1056    pub fn rows(&self) -> usize {
1057        self.rows
1058    }
1059
1060    /// Number of parts in this stripe.
1061    #[must_use]
1062    pub fn parts(&self) -> usize {
1063        self.parts.len()
1064    }
1065
1066    /// The two ends and the null count of every column over the whole stripe.
1067    ///
1068    /// In the directory and so in memory, which is what makes it the one a planner can ask. The
1069    /// finer ones are a page per column per stripe in the file, read by [`Reader::skips`] when a
1070    /// scan wants to know which parts to open.
1071    #[must_use]
1072    pub fn zone(&self) -> &Zone {
1073        &self.zone
1074    }
1075}
1076
1077/// The committed table directory.
1078#[derive(Debug, Clone)]
1079pub struct Table {
1080    name: String,
1081    fields: Vec<Field>,
1082    stripes: Vec<Stripe>,
1083    rows: usize,
1084    dictionaries: Vec<Option<Page>>,
1085    /// Bytes of each column's dictionary payload that are outside its page, which is all of them
1086    /// from format 27 and none of them before. See [`DICTIONARY_PAYLOADS`].
1087    ///
1088    /// Empty rather than a row of zeros on a table that has none, and read with `get` for that
1089    /// reason, so that a table built by hand in a test does not have to know about it.
1090    dictionary_payloads: Vec<u64>,
1091    /// The columns whose dictionary stopped taking values partway through the load, see
1092    /// [`DEMOTED`].
1093    ///
1094    /// Empty rather than a row of `false` on a table that has none, and read with `get`, for the
1095    /// same reason `dictionary_payloads` is.
1096    demoted: Vec<bool>,
1097    frequencies: Vec<Option<Frequencies>>,
1098    pair_frequencies: Vec<PairFrequencySummary>,
1099    /// String spellings aligned with each column's frequency entries.
1100    ///
1101    /// Empty for files written before `RUDBFT1`. A `None` entry is the null frequency entry; every
1102    /// code entry in a column named by the block has its exact bytes here.
1103    frequency_texts: Vec<Vec<Option<Vec<u8>>>>,
1104    /// Exact candidate host aggregates and an upper bound for every omitted host.
1105    host_groups: Option<host::HostSummary>,
1106    /// How many distinct values each column holds, for the columns that know.
1107    ///
1108    /// A dictionary entry is made the first time a value is seen and nothing ever removes one, so
1109    /// the size of the dictionary is the number of distinct values in the column. That is the whole
1110    /// story for a column with no null in it, and the wrong number by one for a column with a null
1111    /// in it, because a null row is written as the code for the empty string and makes an entry the
1112    /// dictionary would not otherwise have. The writer knows which case it is, since it counts the
1113    /// non-null rows that use each code while it builds the frequency summary, and the reader cannot
1114    /// work it out from the dictionary alone. So the writer settles it here.
1115    distincts: Vec<Option<u64>>,
1116    /// The order the rows of this table are meant to be stored in, if anybody declared one.
1117    ///
1118    /// A declaration and not a measurement. Nothing here checks that the stripes actually arrived
1119    /// in this order, and the reason it is worth storing anyway is that the order is the only thing
1120    /// about a table that a rewrite destroys without anybody noticing. The fragment ranges prune on
1121    /// whatever order the rows came in, so a table loaded sorted prunes and the same table after a
1122    /// checkpoint that did not know to keep the order quietly stops pruning and nothing says why.
1123    clustering: Option<Clustering>,
1124    /// The file generation of the commit that last wrote this table's column pages.
1125    ///
1126    /// This is what spec/graph/03-the-file-format.md section 3.2 calls the table generation, and
1127    /// the definition is deliberately about the pages rather than about the directory. A graph
1128    /// section is a restatement of a column in terms of row ids, so what invalidates one is the
1129    /// rows being renumbered, and nothing else. Adding a second table to the file, or attaching a
1130    /// section to this one, commits a new file generation without touching a single row of this
1131    /// table, and a definition that moved with those would declare every section in the file stale
1132    /// for no reason.
1133    ///
1134    /// Zero on a table written before format 23, where nothing recorded it. Real generations start
1135    /// at one, so zero can never match a section's stamp, and a table from format 22 has no
1136    /// sections for it to match anyway.
1137    generation: u64,
1138    /// The graph sections this table carries, per spec/graph/03-the-file-format.md section 3.2.
1139    ///
1140    /// Empty for every table written before the section table existed, and empty is not a
1141    /// degraded state: section 3.1 says deleting every graph section from a file changes no answer,
1142    /// only the time, so a table with none here answers every query the same way and slower. That
1143    /// is what lets this field arrive without a migration.
1144    sections: Vec<Section>,
1145}
1146
1147impl Table {
1148    /// The SQL table name held by this snapshot.
1149    #[must_use]
1150    pub fn name(&self) -> &str {
1151        &self.name
1152    }
1153
1154    /// Columns in their SQL order.
1155    #[must_use]
1156    pub fn fields(&self) -> &[Field] {
1157        &self.fields
1158    }
1159
1160    /// Committed row count.
1161    #[must_use]
1162    pub fn rows(&self) -> usize {
1163        self.rows
1164    }
1165
1166    /// Independently readable stripes.
1167    #[must_use]
1168    pub fn stripes(&self) -> &[Stripe] {
1169        &self.stripes
1170    }
1171
1172    /// The order the rows are meant to be stored in, if this table was declared with one.
1173    #[must_use]
1174    pub fn clustering(&self) -> Option<&Clustering> {
1175        self.clustering.as_ref()
1176    }
1177
1178    /// The generation every section of this table is judged against.
1179    ///
1180    /// See the field. A caller deciding whether to read a section asks [`Section::usable`] with
1181    /// this.
1182    #[must_use]
1183    pub fn generation(&self) -> u64 {
1184        self.generation
1185    }
1186
1187    /// Every graph section this table names, including the kinds this build does not know.
1188    ///
1189    /// Including them is the point. A caller that wants only the ones it can use asks
1190    /// [`Section::usable`], and a caller rewriting the directory carries the rest through, so a
1191    /// file opened by an older build and written again does not silently lose a section that build
1192    /// had no name for.
1193    #[must_use]
1194    pub fn sections(&self) -> &[Section] {
1195        &self.sections
1196    }
1197}
1198
1199/// One table's line in the catalog directory.
1200///
1201/// The small level of the two. It holds what opening a database needs and nothing else: the name to
1202/// bind, the shape to plan against, the row count, and where the table's own directory sits. A file
1203/// of eight tables is eight of these, and reading them costs the same whether the tables hold a
1204/// thousand rows or a billion.
1205///
1206/// The name, the fields and the row count are repeated here rather than pointed at inside the table
1207/// directory, which is the entire point of having two levels. A catalog that pointed at them would
1208/// have to read every table directory at open to answer what tables there are, which is the cost
1209/// this level exists to avoid.
1210#[derive(Debug, Clone)]
1211struct Entry {
1212    name: String,
1213    fields: Vec<Field>,
1214    rows: usize,
1215    /// Where this table's own directory sits, with the checksum it was committed under.
1216    directory: Page,
1217    /// Legacy nonzero counts. New files leave these empty and derive filtered counts from
1218    /// reusable column frequencies when a query needs them.
1219    nonzero: Vec<Option<u64>>,
1220    /// Exact sum and non-null count for signed integer columns.
1221    aggregates: Vec<Option<(i128, u64)>>,
1222    /// Exact non-null distinct values when the writer finished counting the column.
1223    distincts: Vec<Option<u64>>,
1224    /// Exact integer or date bounds; the inner `None` means every row is null.
1225    extremes: Vec<StoredIntegerExtremes>,
1226    /// Complete bounded numeric frequencies, including NULL when present.
1227    frequencies: Vec<StoredNumericFrequencies>,
1228}
1229
1230type StoredIntegerExtremes = Option<Option<(i128, i128)>>;
1231type StoredNumericFrequencies = Option<NumericFrequencies>;
1232
1233/// One view's line in the catalog directory.
1234///
1235/// A view has no pages, so unlike a table it is entirely here and there is no second level under it.
1236/// What it is made of is text: the body the binder binds again at every reference, and the whole
1237/// statement written back out, which is what `duckdb_views()` reports and nothing else reads.
1238///
1239/// The columns are a cache and they are written down anyway, which is worth saying out loud because
1240/// a cache in a file looks like a mistake. It is what the pin does. Create a view on a file, open
1241/// the file again in another process, and `duckdb_views()` answers `column_count` and `is_bound`
1242/// true without anything having bound the body, so the list survived the write. Not writing it
1243/// would answer null and false there, and the only way back would be to bind every view at open,
1244/// which is the thing the cache exists to avoid.
1245#[derive(Debug, Clone, PartialEq, Eq)]
1246pub struct ViewEntry {
1247    /// The view's own name, without the schema, the way a table entry holds its name.
1248    pub name: String,
1249    /// The query the view stands for, as the text that was written.
1250    pub sql: String,
1251    /// The whole `CREATE VIEW` written back out.
1252    pub statement: String,
1253    /// The column names the statement gave, which rename a prefix of what the body produces.
1254    pub aliases: Vec<String>,
1255    /// The columns the last bind of the body produced.
1256    pub columns: Vec<Field>,
1257}
1258
1259/// Where one column's bytes went, taken from the directory rather than by reading pages.
1260#[derive(Debug, Clone)]
1261pub struct ColumnLayout {
1262    /// The column's name, so a report does not have to carry the field list beside this.
1263    pub name: String,
1264    /// The type, spelled the way the catalog spells it.
1265    pub kind: String,
1266    /// Every stripe's page of this column added up, which is the encoded data itself.
1267    pub pages: u64,
1268    /// Every stripe's exact code membership page for this column.
1269    pub memberships: u64,
1270    /// Every stripe's membership sieve page for this column.
1271    pub sieves: u64,
1272    /// Every stripe's per part range page for this column.
1273    pub part_ranges: u64,
1274    /// The table wide dictionary of this column, if it has one.
1275    pub dictionary: u64,
1276}
1277
1278impl ColumnLayout {
1279    /// Everything this column costs, which is what the file would lose if the column went.
1280    #[must_use]
1281    pub fn total(&self) -> u64 {
1282        self.pages
1283            .saturating_add(self.memberships)
1284            .saturating_add(self.sieves)
1285            .saturating_add(self.part_ranges)
1286            .saturating_add(self.dictionary)
1287    }
1288}
1289
1290/// Where a whole file's bytes went.
1291///
1292/// Every number here comes out of the committed directory, so taking it costs one directory read
1293/// however large the file is. That is the point: a 45 GB table has to be able to say where it went
1294/// without being read, or nobody will ask.
1295///
1296/// The parts that are not a column are kept apart rather than shared out over the columns. The
1297/// stripe index page holds a section per column and could be split, and the directory and the
1298/// header cannot be, so splitting one of the three and not the others would read as if the columns
1299/// accounted for everything. They do not, and the gap is the thing worth looking at.
1300#[derive(Debug, Clone)]
1301pub struct Layout {
1302    /// The size of the file on disk.
1303    pub file: u64,
1304    /// Committed rows.
1305    pub rows: usize,
1306    /// Committed stripes.
1307    pub stripes: usize,
1308    /// Committed parts, which is how many chunks a scan reads.
1309    pub parts: usize,
1310    /// One entry per column, in the table's column order.
1311    pub columns: Vec<ColumnLayout>,
1312    /// Every stripe's index page, which carries a length and a checksum for every part of every
1313    /// column and is charged per stripe rather than per column.
1314    pub indexes: u64,
1315    /// The committed directory itself, the one that was read to build this.
1316    pub directory: u64,
1317    /// The fixed header, which holds the magic, the format and the two directory slots.
1318    pub header: u64,
1319}
1320
1321impl Layout {
1322    /// Everything the columns cost together.
1323    #[must_use]
1324    pub fn columns_total(&self) -> u64 {
1325        self.columns.iter().map(ColumnLayout::total).fold(0, u64::saturating_add)
1326    }
1327
1328    /// What the file holds that this does not account for.
1329    ///
1330    /// A committed file is written once and never rewritten in place, so an earlier directory and
1331    /// the pages of an earlier snapshot are still in it. That is the honest place for them: they
1332    /// are bytes on disk that no column owns.
1333    #[must_use]
1334    pub fn unaccounted(&self) -> u64 {
1335        self.file
1336            .saturating_sub(self.columns_total())
1337            .saturating_sub(self.indexes)
1338            .saturating_sub(self.directory)
1339            .saturating_sub(self.header)
1340    }
1341}
1342
1343/// How one part of one column is stored, which is one row of `pragma_storage_info`.
1344///
1345/// Everything here is read off the file rather than worked out from the schema, because the whole
1346/// question this answers is what the encoder chose, and the encoder chooses per part. Two files
1347/// holding the same rows in a different order give different answers and that difference is the
1348/// reason to ask.
1349///
1350/// The encoding costs a read of the column's page, so this is not free the way [`Layout`] is. It is
1351/// one read per column per stripe rather than one per part, because a part is a few kilobytes out
1352/// of a page that is a quarter of a megabyte.
1353#[derive(Debug, Clone)]
1354pub struct StoredPart {
1355    /// Which stripe the part belongs to.
1356    pub stripe: usize,
1357    /// Which part of that stripe it is, counting from zero inside the stripe.
1358    pub part: usize,
1359    /// The table wide row number the part starts at.
1360    pub row: usize,
1361    /// How many rows it holds.
1362    pub rows: usize,
1363    /// What the encoder made of it, as a line of text like `DICT(PACKED, PACKED)`.
1364    pub encoding: String,
1365    /// The stored bytes of the part, which is what it costs in the file.
1366    pub bytes: u64,
1367    /// Where in the file the column page holding this part starts.
1368    pub page: u64,
1369    /// Where in that page the part starts.
1370    pub offset: u64,
1371    /// The smallest value the part holds, when the stored ranges say.
1372    pub low: Option<Value>,
1373    /// The largest, same.
1374    pub high: Option<Value>,
1375    /// How many of its rows are null, when the stored ranges say.
1376    pub nulls: Option<usize>,
1377}
1378
1379/// Seeds the second hash a global dictionary tells its values apart by.
1380///
1381/// Any value that is not zero does, since zero is the seed [`checksum`] already uses and the point
1382/// is only that the two hashes of one value are not the same number. This one is the fractional part
1383/// of the golden ratio in sixty four bits, which is the constant everything else here is built out
1384/// of and is as good a nothing-up-my-sleeve number as any.
1385const DICTIONARY_CHECK_SEED: u64 = 11_400_714_819_323_198_485;
1386
1387/// One column's table wide dictionary while the load is running.
1388///
1389/// The thing to understand about this is what it does not hold. A dictionary of `URL` at a hundred
1390/// million ClickBench rows has about eighteen million distinct values and 1.3 GB of bytes in them,
1391/// and five columns like it are twelve of the seventeen gigabytes a load of `hits` peaks at. So the
1392/// bytes are not kept. A value's bytes go into [`GlobalDictionary::filling`], and when that reaches
1393/// [`TEXT_PAYLOAD_VALUES`] values the block is sealed, handed out at the end of the merge that
1394/// sealed it to be encoded with the stripe's pages, and never seen in that form again. What is left
1395/// is the encoded block, which is two to three times smaller, and that is the same bytes the file
1396/// is going to hold anyway.
1397///
1398/// Two things needed the raw bytes and neither needs them now. Deciding whether a value has been
1399/// seen before was a hash lookup and then a comparison of the bytes, and is now a hash lookup and a
1400/// comparison of a second hash under a different seed, which is [`DICTIONARY_CHECK_SEED`] and the
1401/// argument for why that is sound. Sorting the values at the end needed all of them at once, and
1402/// now reads the blocks back through [`GlobalDictionary::decoded`] one column at a time, which is
1403/// one column's bytes rather than every column's.
1404///
1405/// The offsets going block relative comes free with it, and takes the four gigabyte wall with it.
1406/// They were `u32` into a per column payload, so a column could not hold more than four gigabytes of
1407/// values however much memory the machine had, and `URL` and `Referer` are within a small factor of
1408/// that at a hundred million rows. A `u32` into a block of 1,024 values is not a bound anything real
1409/// reaches. The stored form is unchanged, because [`encode_offsets`] was already subtracting a per
1410/// block base before writing.
1411#[derive(Debug)]
1412struct GlobalDictionary {
1413    /// Keyed by the value's hash, which is already well spread, so the maps hash it once more
1414    /// with a multiply rather than with SipHash. SipHash here was one percent of a ClickBench load,
1415    /// and every stripe's merge of a column waits on the one before it.
1416    primary: HashMap<u64, u32, Spread>,
1417    collisions: HashMap<u64, Vec<u32>, Spread>,
1418    /// Every value's hash under [`DICTIONARY_CHECK_SEED`], in code order.
1419    checks: Vec<u64>,
1420    /// Where every value ends inside the payload block it is in, in code order.
1421    ends: Vec<u32>,
1422    counts: Vec<u64>,
1423    nulls: u64,
1424    /// The values of the block being filled, back to back.
1425    filling: Vec<u8>,
1426    /// One conservative four-byte substring signature per encoded payload block, in block order.
1427    ///
1428    /// Made where the block is encoded rather than where it is sealed, because sealing is under the
1429    /// writer's lock and every byte of every value going through [`gram_bits`] was 2.9 of the 14
1430    /// seconds the 10m ClickBench load spent on the 32 core box.
1431    grams: Vec<[u8; TEXT_GRAM_BYTES]>,
1432    /// Blocks that have filled and not been handed out to be encoded yet, each with its block number.
1433    ///
1434    /// Empty except inside the merge that filled them, and while the column is still too small to
1435    /// settle a shape on.
1436    waiting: Vec<(usize, Vec<u8>)>,
1437    /// Blocks kept raw to settle a shape on, spread across the column, each with its number.
1438    ///
1439    /// At most [`PAYLOAD_SAMPLE_BLOCKS`] of them and so at most a few megabytes. Spread rather than
1440    /// taken off the front for the reason [`settle_shape`] gives, and kept rather than read back
1441    /// because reading back is a decode and this is a sample of a column that is still growing.
1442    sample: Vec<(usize, Vec<u8>)>,
1443    /// How far apart the blocks in `sample` are, which doubles every time there are too many.
1444    stride: usize,
1445    /// What the blocks encoded so far were encoded with, once the column is big enough to settle it.
1446    shape: Option<chooser::Settled>,
1447    /// How many blocks had filled when that shape was settled.
1448    settled: usize,
1449    /// The blocks that are encoded and not yet in the file, in block order, following `placed`.
1450    ///
1451    /// Empty between stripes, because [`Writer::place_blocks`] writes them the moment they come
1452    /// back. Only a dictionary that never meets a writer, which is a test's, keeps them here.
1453    blocks: Vec<Vec<u8>>,
1454    /// Blocks that came back encoded ahead of a block before them, by block number.
1455    ///
1456    /// Two stripes merged one after the other can have their pages built in the other order, and a
1457    /// block cannot go into `blocks` until every block before it is there. They wait here until the
1458    /// gap closes, which is at most until the stripe merged just before this one is written.
1459    early: BTreeMap<usize, EncodedBlock>,
1460    /// Where every block already written to the file is, in block order.
1461    placed: Vec<Placed>,
1462    /// What the dictionary held the last time it was asked, see [`Self::recharge`], which is also
1463    /// what the load profile was told when there is one.
1464    charged: u64,
1465    /// Whether the dictionary stopped taking values, see [`Self::demote`].
1466    demoted: bool,
1467}
1468
1469/// Where one payload block of a global dictionary is in the file, and its checksum.
1470#[derive(Debug, Clone, Copy)]
1471struct Placed {
1472    start: u64,
1473    length: u64,
1474    hash: u64,
1475}
1476
1477/// Sorted `(head, code)` entries and the decoded bytes and block bases they were sorted over.
1478type RankedDictionary = (Vec<(u64, u32)>, Vec<u8>, Vec<u64>);
1479
1480impl GlobalDictionary {
1481    fn new() -> Self {
1482        Self {
1483            primary: HashMap::default(),
1484            collisions: HashMap::default(),
1485            checks: Vec::new(),
1486            ends: Vec::new(),
1487            counts: Vec::new(),
1488            nulls: 0,
1489            filling: Vec::new(),
1490            grams: Vec::new(),
1491            waiting: Vec::new(),
1492            sample: Vec::new(),
1493            stride: 1,
1494            shape: None,
1495            settled: 0,
1496            blocks: Vec::new(),
1497            early: BTreeMap::new(),
1498            placed: Vec::new(),
1499            charged: 0,
1500            demoted: false,
1501        }
1502    }
1503
1504    /// How many distinct values this dictionary holds, which is one past its largest code.
1505    fn values(&self) -> usize {
1506        self.ends.len()
1507    }
1508
1509    /// About how many bytes closing this dictionary holds at once: every value decoded, and a
1510    /// sort entry and a code for each.
1511    fn closing_bytes(&self) -> usize {
1512        let values = self.values();
1513        let decoded = (0..values.div_ceil(TEXT_PAYLOAD_VALUES))
1514            .map(|block| self.ends[((block + 1) * TEXT_PAYLOAD_VALUES).min(values) - 1] as usize)
1515            .sum::<usize>();
1516        decoded.saturating_add(values.saturating_mul(size_of::<(u64, u32)>() + size_of::<u32>()))
1517    }
1518
1519    /// About what the dictionary holds in memory, by capacity rather than by length.
1520    ///
1521    /// A hash table is charged its buckets, which is a power of two over eight sevenths of what it
1522    /// says it can hold, and a byte of control per bucket. The blocks waiting to be encoded and the
1523    /// ones kept to settle a shape on are counted one by one, and there are only ever a few.
1524    fn held_bytes(&self) -> u64 {
1525        fn table<K, V, S>(map: &HashMap<K, V, S>) -> usize {
1526            (map.capacity() * 8 / 7).next_power_of_two() * (size_of::<(K, V)>() + 1)
1527        }
1528        fn spilled<T>(values: &Vec<T>) -> usize {
1529            values.capacity() * size_of::<T>()
1530        }
1531        let raw = |blocks: &Vec<(usize, Vec<u8>)>| {
1532            spilled(blocks) + blocks.iter().map(|(_, block)| block.capacity()).sum::<usize>()
1533        };
1534        let bytes = table(&self.primary)
1535            + table(&self.collisions)
1536            + self.collisions.values().map(spilled).sum::<usize>()
1537            + spilled(&self.checks)
1538            + spilled(&self.ends)
1539            + spilled(&self.counts)
1540            + self.filling.capacity()
1541            + spilled(&self.grams)
1542            + raw(&self.waiting)
1543            + raw(&self.sample)
1544            + self.blocks.iter().map(Vec::capacity).sum::<usize>()
1545            + spilled(&self.placed);
1546        bytes as u64
1547    }
1548
1549    /// Tells `profile` what the dictionary has grown or shrunk by since the last time, and hands
1550    /// back what it held then and what it holds now.
1551    fn recharge(&mut self, profile: Option<&LoadProfile>) -> (u64, u64) {
1552        let before = self.charged;
1553        let now = self.held_bytes();
1554        if let Some(profile) = profile {
1555            if now >= before {
1556                profile.hold(now - before);
1557            } else {
1558                profile.release(before - now);
1559            }
1560        }
1561        self.charged = now;
1562        (before, now)
1563    }
1564
1565    /// Stops the dictionary taking values, for good.
1566    ///
1567    /// The block being filled is sealed so that it goes out with the others, and what the
1568    /// dictionary keeps for looking values up is let go of, which on a column of mostly new values
1569    /// is most of what it holds. What stays is what the close needs to write the dictionary's page:
1570    /// where every value ends, how often each was seen and where its blocks went. The stripes that
1571    /// were coded against it still need that page to be read. See [`DEMOTED`].
1572    fn demote(&mut self) {
1573        if self.demoted {
1574            return;
1575        }
1576        self.seal_rest();
1577        self.release_lookup();
1578        self.demoted = true;
1579    }
1580
1581    /// Frees what the dictionary keeps for coding new values, once none are coming.
1582    ///
1583    /// The hash tables, the check hash of every value and the blocks kept to settle a shape on are
1584    /// what a merge looks values up in. The close reads the counts, the ends and the written blocks
1585    /// and none of these, which are most of what the dictionary holds per value, so they go before
1586    /// the close takes memory of its own rather than after.
1587    fn release_lookup(&mut self) {
1588        self.primary = HashMap::default();
1589        self.collisions = HashMap::default();
1590        self.checks = Vec::new();
1591        self.sample = Vec::new();
1592        self.filling = Vec::new();
1593    }
1594
1595    /// How many blocks are encoded, written or not, which is the number the next one has to have.
1596    fn encoded(&self) -> usize {
1597        self.placed.len() + self.blocks.len()
1598    }
1599
1600    #[cfg(test)]
1601    fn code(&mut self, text: &str) -> Result<u32> {
1602        let bytes = text.as_bytes();
1603        self.code_hashed(bytes, checksum(bytes), seeded_checksum(bytes, DICTIONARY_CHECK_SEED))
1604    }
1605
1606    /// The code for a value whose two hashes the caller already has.
1607    ///
1608    /// A stripe prepared outside the writer's lock hashed every value it holds while it was coding
1609    /// them, and merging it into this dictionary is one of these a distinct value rather than two
1610    /// hashes of every row. See [`prepare`].
1611    fn code_hashed(&mut self, text: &[u8], hash: u64, check: u64) -> Result<u32> {
1612        if let Some(&code) = self.primary.get(&hash) {
1613            if self.checks.get(code as usize) == Some(&check) {
1614                return Ok(code);
1615            }
1616            if let Some(codes) = self.collisions.get(&hash) {
1617                if let Some(code) =
1618                    codes.iter().copied().find(|&code| self.checks[code as usize] == check)
1619                {
1620                    return Ok(code);
1621                }
1622            }
1623            let code = self.insert(text, check)?;
1624            self.collisions.entry(hash).or_default().push(code);
1625            return Ok(code);
1626        }
1627        let code = self.insert(text, check)?;
1628        self.primary.insert(hash, code);
1629        Ok(code)
1630    }
1631
1632    fn insert(&mut self, text: &[u8], check: u64) -> Result<u32> {
1633        if self.demoted {
1634            return Err(Error::internal("a value was coded against a demoted dictionary"));
1635        }
1636        let code = u32::try_from(self.ends.len())
1637            .map_err(|_| invalid("global dictionary has too many values"))?;
1638        self.filling.extend_from_slice(text);
1639        self.ends.push(
1640            u32::try_from(self.filling.len())
1641                .map_err(|_| invalid("a global dictionary value exceeds 4 GiB"))?,
1642        );
1643        self.checks.push(check);
1644        self.counts.push(0);
1645        if self.ends.len() % TEXT_PAYLOAD_VALUES == 0 {
1646            self.seal();
1647        }
1648        Ok(code)
1649    }
1650
1651    /// Closes the block being filled and puts it in the queue to be encoded.
1652    ///
1653    /// Also keeps a copy of it if it lands on the sample's stride, and halves the sample when that
1654    /// has left too many, which is what keeps the kept blocks spread evenly over however much of the
1655    /// column exists rather than bunched at whichever end was cheap to remember.
1656    fn seal(&mut self) {
1657        let at = self.ends.len().div_ceil(TEXT_PAYLOAD_VALUES) - 1;
1658        let bytes = std::mem::take(&mut self.filling);
1659        if at % self.stride == 0 {
1660            self.sample.push((at, bytes.clone()));
1661            if self.sample.len() > PAYLOAD_SAMPLE_BLOCKS {
1662                self.stride *= 2;
1663                let stride = self.stride;
1664                self.sample.retain(|(at, _)| at % stride == 0);
1665            }
1666        }
1667        self.waiting.push((at, bytes));
1668    }
1669
1670    /// The values of one block, as slices into the bytes the block was filled with.
1671    fn slices<'a>(&self, at: usize, bytes: &'a [u8]) -> Vec<&'a [u8]> {
1672        block_values(self.block_ends(at), bytes)
1673    }
1674
1675    /// Where every value of one block ends, relative to the block.
1676    fn block_ends(&self, at: usize) -> &[u32] {
1677        let first = (at * TEXT_PAYLOAD_VALUES).min(self.ends.len());
1678        let last = (first + TEXT_PAYLOAD_VALUES).min(self.ends.len());
1679        &self.ends[first..last]
1680    }
1681
1682    /// Takes every waiting block out to be encoded somewhere else, if the column has a shape to
1683    /// encode them with.
1684    ///
1685    /// This is what keeps the encoding out of the writer's lock. A block needs its bytes, where its
1686    /// values end and the shape, and nothing else of the dictionary, so it goes out with a copy of
1687    /// the four kilobytes of ends it has and comes back through [`GlobalDictionary::take_back`].
1688    fn hand_out(&mut self, column: usize) -> Vec<Unencoded> {
1689        let Some(shape) = &self.shape else { return Vec::new() };
1690        let waiting = std::mem::take(&mut self.waiting);
1691        waiting
1692            .into_iter()
1693            .map(|(at, bytes)| Unencoded {
1694                column,
1695                at,
1696                ends: self.block_ends(at).to_vec(),
1697                bytes,
1698                shape: shape.clone(),
1699            })
1700            .collect()
1701    }
1702
1703    /// Takes back one block that was handed out, and moves every block that is now next in line
1704    /// into `blocks`.
1705    fn take_back(&mut self, at: usize, block: EncodedBlock) -> Result<()> {
1706        if at < self.encoded() || self.early.insert(at, block).is_some() {
1707            return Err(Error::internal("a dictionary block came back twice"));
1708        }
1709        while let Some(block) = self.early.remove(&self.encoded()) {
1710            self.push_block(block);
1711        }
1712        Ok(())
1713    }
1714
1715    /// Appends the next encoded block and its signature.
1716    fn push_block(&mut self, (bytes, grams): EncodedBlock) {
1717        self.blocks.push(bytes);
1718        self.grams.push(*grams);
1719    }
1720
1721    /// Settles the shape the waiting blocks are about to be encoded with, if there is enough column
1722    /// to settle one on.
1723    ///
1724    /// Settled again once the column has grown fourfold, because the sample it was settled on then
1725    /// covered a quarter of what exists now and a dictionary in first seen order does not look the
1726    /// same at both ends. Blocks already encoded keep the shape they were encoded with. They can,
1727    /// because a block says what it is: nothing reading one asks the column what shape to expect.
1728    fn settle(&mut self) -> Result<()> {
1729        if self.sample.len() < PAYLOAD_SAMPLE_BLOCKS {
1730            return Ok(());
1731        }
1732        self.settle_on_sample()
1733    }
1734
1735    /// Settles a shape on whatever sample there is, for a column the load ended before it had
1736    /// enough of to settle one the usual way.
1737    ///
1738    /// Such a column has fewer than [`PAYLOAD_SAMPLE_BLOCKS`] blocks, so the sample is every block
1739    /// it has. Trying every candidate on each of them instead runs at two to six megabytes a second,
1740    /// and once `hits` stored its string columns with a dictionary, the forty or so small ones were
1741    /// more than half the CPU of a million row load, all of it in the close.
1742    fn settle_rest(&mut self) -> Result<()> {
1743        if self.shape.is_some() || self.sample.is_empty() {
1744            return Ok(());
1745        }
1746        self.settle_on_sample()
1747    }
1748
1749    fn settle_on_sample(&mut self) -> Result<()> {
1750        let complete = self.ends.len() / TEXT_PAYLOAD_VALUES;
1751        if self.shape.is_some() && complete < self.settled.saturating_mul(4) {
1752            return Ok(());
1753        }
1754        let sample =
1755            self.sample.iter().map(|(at, bytes)| self.slices(*at, bytes)).collect::<Vec<_>>();
1756        self.shape = Some(string::with_symbols(settle_shape(&sample)?, &sample));
1757        self.settled = complete;
1758        Ok(())
1759    }
1760
1761    /// Seals the part block at the end of the load, if there is one.
1762    fn seal_rest(&mut self) {
1763        // Asked of the values rather than of the bytes, because a block of empty strings has values
1764        // in it and no bytes, and a column of nulls is exactly that. A demoted dictionary sealed its
1765        // part block when it was demoted and has taken nothing since.
1766        if !self.demoted && self.ends.len() % TEXT_PAYLOAD_VALUES != 0 {
1767            self.seal();
1768        }
1769    }
1770
1771    /// Encodes the waiting block at `at`, with the settled shape when there is one and by trying
1772    /// everything when the column was too small to settle one.
1773    fn encode_waiting(&self, at: usize) -> Result<EncodedBlock> {
1774        let (block, bytes) = &self.waiting[at];
1775        let values = self.slices(*block, bytes);
1776        let encoded = match &self.shape {
1777            Some(shape) => string::encode_with(&values, shape)?,
1778            None => string::encode(&values)?,
1779        };
1780        Ok((encoded, block_grams(&values)))
1781    }
1782
1783    /// [`finish_dictionaries`] for one dictionary on this thread, for the tests that hold one.
1784    #[cfg(test)]
1785    fn finish_blocks(&mut self) -> Result<()> {
1786        self.seal_rest();
1787        let made = (0..self.waiting.len())
1788            .map(|at| self.encode_waiting(at))
1789            .collect::<Result<Vec<_>>>()?;
1790        for ((at, _), block) in std::mem::take(&mut self.waiting).into_iter().zip(made) {
1791            if self.encoded() != at {
1792                return Err(Error::internal("a dictionary block was encoded out of order"));
1793            }
1794            self.push_block(block);
1795        }
1796        Ok(())
1797    }
1798
1799    /// Every value of this dictionary read back out of its encoded blocks, as the bytes back to back
1800    /// and where each block starts in them.
1801    ///
1802    /// This is the one place the whole column is in memory at once and the reason [`Writer::close`]
1803    /// takes the columns one at a time rather than across threads. One column's values is 1.3 GB on
1804    /// the worst ClickBench column, and five columns of that at once is the peak this was all meant
1805    /// to remove.
1806    ///
1807    /// The blocks are spread over threads instead. Each block's decoded length is already known from
1808    /// the ends of its values, so the answer is laid out before anything is decoded and every thread
1809    /// decodes its own run of blocks straight into its own part of it. On the 10m ClickBench sample
1810    /// this was a second of the close for `URL` alone, on one core of thirty two, and the close is
1811    /// what a load waits on once its stripes are written.
1812    ///
1813    /// The blocks already written are read back out of `file`, so what a thread holds beyond the
1814    /// answer is one encoded block. They were written moments or minutes ago and are almost always
1815    /// still in the page cache, so this is a copy rather than a read of the disk.
1816    fn decoded(&self, file: Option<&dyn rudb_io::File>) -> Result<(Vec<u8>, Vec<u64>)> {
1817        let count = self.placed.len() + self.blocks.len();
1818        if count != self.values().div_ceil(TEXT_PAYLOAD_VALUES) {
1819            return Err(invalid("global dictionary blocks do not cover its values"));
1820        }
1821        let mut bases = Vec::with_capacity(count);
1822        let mut total = 0_usize;
1823        for block in 0..count {
1824            bases.push(total as u64);
1825            let last = ((block + 1) * TEXT_PAYLOAD_VALUES).min(self.values()) - 1;
1826            total = total
1827                .checked_add(self.ends[last] as usize)
1828                .ok_or_else(|| invalid("global dictionary does not fit in memory"))?;
1829        }
1830        let mut flat = vec![0_u8; total];
1831        let mut outs = Vec::with_capacity(count);
1832        let mut rest = flat.as_mut_slice();
1833        for block in 0..count {
1834            let end = bases.get(block + 1).map_or(total, |&base| base as usize);
1835            let (out, after) = rest.split_at_mut(end - bases[block] as usize);
1836            outs.push((block, out));
1837            rest = after;
1838        }
1839        let one = |run: &mut [(usize, &mut [u8])]| -> Result<()> {
1840            let mut stored = Vec::new();
1841            for (block, out) in run {
1842                let encoded = match self.placed.get(*block) {
1843                    Some(place) => {
1844                        let file = file.ok_or_else(|| {
1845                            Error::internal("a written dictionary block has no file")
1846                        })?;
1847                        let length = usize::try_from(place.length).map_err(|_| {
1848                            invalid("global dictionary block does not fit in memory")
1849                        })?;
1850                        stored.resize(length, 0);
1851                        read_at(file, place.start, &mut stored)?;
1852                        if checksum(&stored) != place.hash {
1853                            return Err(invalid(
1854                                "a global dictionary block did not read back as written",
1855                            ));
1856                        }
1857                        stored.as_slice()
1858                    }
1859                    None => &self.blocks[*block - self.placed.len()],
1860                };
1861                let decoded = string::decode_flat(encoded)?;
1862                if decoded.bytes().len() != out.len() {
1863                    return Err(invalid(
1864                        "a global dictionary block is not the length its ends say",
1865                    ));
1866                }
1867                out.copy_from_slice(decoded.bytes());
1868            }
1869            Ok(())
1870        };
1871        // Sixteen blocks a thread at the least, because a thread costs about what decoding a few
1872        // blocks does and most columns have one or two.
1873        let workers = close_workers().min(count / 16).max(1);
1874        if workers <= 1 {
1875            one(&mut outs)?;
1876        } else {
1877            let per = count.div_ceil(workers);
1878            std::thread::scope(|scope| {
1879                outs.chunks_mut(per)
1880                    .map(|run| scope.spawn(|| one(run)))
1881                    .collect::<Vec<_>>()
1882                    .into_iter()
1883                    .try_for_each(|handle| {
1884                        handle.join().map_err(|_| {
1885                            Error::internal("a global dictionary decode worker panicked")
1886                        })?
1887                    })
1888            })?;
1889        }
1890        drop(outs);
1891        Ok((flat, bases))
1892    }
1893
1894    /// Where the value at `code` sits in the bytes [`GlobalDictionary::decoded`] handed back.
1895    ///
1896    /// A block's first value starts at the block, and every other value starts where the one before
1897    /// it ended, which is what makes 1,024 values 1,024 numbers rather than 1,025.
1898    fn value_span(ends: &[u32], bases: &[u64], code: usize) -> (usize, usize) {
1899        let Some(&base) = bases.get(code / TEXT_PAYLOAD_VALUES) else { return (0, 0) };
1900        let Some(&end) = ends.get(code) else { return (0, 0) };
1901        let base = base as usize;
1902        let from = if code % TEXT_PAYLOAD_VALUES == 0 { 0 } else { ends[code - 1] as usize };
1903        (base + from, base + end as usize)
1904    }
1905
1906    /// This dictionary's values in sorted order, each as the first eight bytes of the value and the
1907    /// code that holds it, so entry `rank` describes the value that sits at `rank` when the values
1908    /// are sorted by their bytes.
1909    ///
1910    /// Codes themselves stay in first appearance order, which is what lets the writer hand one out
1911    /// the moment it sees a value rather than waiting for the last stripe, and which also keeps a
1912    /// stripe's codes close together because the data is clustered. This is what puts the values
1913    /// back in order for anything that needs it, and it is separate from the codes so that getting
1914    /// it costs a sort of the distinct values at the end rather than a rewrite of every code page.
1915    ///
1916    /// The order is the byte order of the values and nothing else. The heads are attached after the
1917    /// sort rather than sorted on, because padding with zero on the right is order preserving for
1918    /// byte strings and so sorting by head and then by bytes lands in the same place as sorting by
1919    /// bytes: a shorter value differs from a longer one that starts the same way at a position
1920    /// where the shorter one has run out, and zero is below every byte that could be there.
1921    ///
1922    /// The heads are kept because a reader searching this order wants a comparison it can make out
1923    /// of the index alone. What they buy there depends entirely on the column and is much less than
1924    /// it looks on the columns that cost the most, which [`sort_by_value`] measures.
1925    fn ranked_with_values(&self, file: Option<&dyn rudb_io::File>) -> Result<RankedDictionary> {
1926        let (flat, bases) = self.decoded(file)?;
1927        let value = |code: u32| {
1928            let (from, to) = Self::value_span(&self.ends, &bases, code as usize);
1929            flat.get(from..to).unwrap_or_default()
1930        };
1931        let mut codes = (0..self.values() as u32).collect::<Vec<_>>();
1932        sort_by_value_across(&mut codes, value, close_workers());
1933        let order = codes.into_iter().map(|code| (head(value(code)), code)).collect();
1934        Ok((order, flat, bases))
1935    }
1936
1937    #[cfg(test)]
1938    fn ranked(&self, file: Option<&dyn rudb_io::File>) -> Result<Vec<(u64, u32)>> {
1939        self.ranked_with_values(file).map(|(order, _, _)| order)
1940    }
1941}
1942
1943/// Appends pages and commits a new directory.
1944///
1945/// One writer covers a whole file rather than one table. [`Writer::next`] closes the table it is on
1946/// and opens another over the same file, and [`Writer::finish`] commits every table it has closed in
1947/// one generation. That is what makes a checkpoint atomic across tables: there is one slot write at
1948/// the end of it and a reader sees every table at the generation before it or every table at the
1949/// generation after it.
1950#[derive(Debug)]
1951pub struct Writer {
1952    /// The file, through `rudb-io` rather than `std::fs`, so that a test can hand the writer a
1953    /// simulated filesystem and crash a load at every call it makes.
1954    file: Box<dyn rudb_io::File>,
1955    /// Where the next write goes, counted here rather than asked of the file.
1956    ///
1957    /// The file's own cursor is not ours. Building the numeric frequencies reads pages back through
1958    /// [`read_at`], and a positional read is only positional about where it reads from: `pread`
1959    /// leaves the cursor alone, and the call Windows has for it moves the cursor to the end of what
1960    /// it read. A writer that asked the file where it was would then write the directory over a
1961    /// page it had already written, which is what it did.
1962    at: u64,
1963    /// How far into the file the kernel has been asked to start writing, see [`WRITEBACK_STRETCH`].
1964    written_back: u64,
1965    table: Table,
1966    generation: u64,
1967    /// The first and the last source position in every stripe, in the order the stripes were
1968    /// written.
1969    order: Vec<((u64, u64), (u64, u64))>,
1970    next_order: u64,
1971    dictionaries: Vec<Option<GlobalDictionary>>,
1972    /// Which columns still have a global dictionary, shared with every [`Preparer`] this writer
1973    /// hands out so that a stripe prepared after a column lost its dictionary is not coded for it.
1974    coded: Arc<prepare::Coding>,
1975    /// One per column, folding the rows into a summary and a sketch as they go past.
1976    ///
1977    /// `None` for a column with no hash rule, which is the interval and the nested types. See
1978    /// [`stats::Gather`] for why the statistics are built here rather than by reading the file back
1979    /// once it is committed.
1980    gathers: Vec<Option<stats::Gather>>,
1981    /// The dictionaries and the statistics while a [`Merger`] has them, which is from
1982    /// [`Writer::merger`] until the table is closed. `dictionaries` and `gathers` are empty then.
1983    lent: Option<Arc<Lent>>,
1984    pending: Vec<PendingChunk>,
1985    /// The tables already closed in this generation, in the order they were written.
1986    closed: Vec<Entry>,
1987    /// The views the next commit writes down, which [`Writer::with_views`] sets.
1988    ///
1989    /// Carried forward from the committed generation by [`Writer::open`], so a writer that was only
1990    /// opened to append a table does not have to know about views to avoid dropping them.
1991    views: Vec<ViewEntry>,
1992    /// Where the stages this writer runs are charged, which [`Writer::with_profile`] sets.
1993    ///
1994    /// The writer runs the page builder, the dictionary blocks, the writes and the publish, and it
1995    /// charges them once per stripe and once per worker, never per chunk. See
1996    /// `rudb_metrics::LoadProfile` for why that is the grain.
1997    profile: Option<Arc<LoadProfile>>,
1998}
1999
2000/// A chunk that has arrived and is waiting for the rest of its stripe.
2001///
2002/// The rows are kept rather than the pages they encode to, which is the whole of #808's first half.
2003/// Encoding on arrival put every column of every part on the thread that called `append_at`, and
2004/// that thread is the only one the load has. Encoding at the flush instead means a stripe's worth
2005/// of work is on the table at once, and a stripe splits by column into a hundred and five pieces
2006/// that share nothing.
2007#[derive(Debug)]
2008struct PendingChunk {
2009    order: (u64, u64),
2010    chunk: Chunk,
2011}
2012
2013/// What the writer still needs of a part once its columns are encoded: where in the source it came
2014/// from, how many rows it has and how large those rows were.
2015///
2016/// A stripe waiting for the writer's lock carries these rather than its chunks, so its rows are
2017/// freed as soon as they are encoded and not after the stripe is written. See [`prepare`].
2018#[derive(Debug, Clone, Copy)]
2019struct Part {
2020    order: (u64, u64),
2021    rows: usize,
2022    footprint: usize,
2023}
2024
2025impl Part {
2026    fn of(pending: &PendingChunk) -> Self {
2027        Self {
2028            order: pending.order,
2029            rows: pending.chunk.len(),
2030            footprint: pending.chunk.footprint(),
2031        }
2032    }
2033}
2034
2035/// One column's share of a stripe, which is what one encode worker produces.
2036///
2037/// Indexed by part, so a stripe is a column of these and the write loop reads down one of them.
2038/// That is also the order the loop wanted: `flush_pending` walks a column at a time and lays its
2039/// parts next to each other, and it used to reach across a row of parts to do it.
2040#[derive(Debug)]
2041struct ColumnStripe {
2042    pages: Vec<Vec<u8>>,
2043    codes: Vec<Option<Vec<u32>>>,
2044    sieves: Vec<Option<Sieve>>,
2045    ranges: Vec<Range>,
2046}
2047
2048/// Whether a column of this type is coded against a global dictionary.
2049///
2050/// A dictionary, its codes and the membership index beside them are about bytes and not about
2051/// text, so a blob gets one the same as a varchar does. ClickBench's `hits.parquet` stores every
2052/// string column as a plain byte array, which reads back as a blob, and those columns were being
2053/// written as a length and the bytes for every row: 533 MB for the first million rows where DuckDB
2054/// writes 142.
2055fn coded_type(ty: &LogicalType) -> bool {
2056    matches!(ty, LogicalType::Varchar | LogicalType::Blob)
2057}
2058
2059/// The tag a directory gives a column's global dictionary.
2060///
2061/// A varchar's is 1, as it always was. A blob's is 2, so that a reader from before blobs had
2062/// dictionaries meets a tag it does not know and refuses the file, rather than laying the rest of
2063/// the directory out as if the blob columns had no dictionary and reading everything after the
2064/// first one from the wrong place.
2065fn dictionary_tag(ty: &LogicalType) -> u8 {
2066    if ty == &LogicalType::Blob { 2 } else { 1 }
2067}
2068
2069/// Roughly what encoding a column of this type costs, for ordering the encode queue.
2070///
2071/// Only the order matters and only roughly. A string column hashes and copies every value into a
2072/// dictionary and is in a different class from everything else, and among the fixed widths the wide
2073/// ones carry more bytes through the cascade than the narrow ones. Anything finer than that would
2074/// be a cost model, and the queue already absorbs a wrong guess: it only has to avoid finishing on
2075/// a column nobody else can help with.
2076fn weight(ty: &LogicalType) -> usize {
2077    match ty {
2078        LogicalType::Varchar | LogicalType::Blob | LogicalType::Bit => 64,
2079        LogicalType::HugeInt
2080        | LogicalType::UHugeInt
2081        | LogicalType::Uuid
2082        | LogicalType::Interval => 16,
2083        LogicalType::BigInt
2084        | LogicalType::UBigInt
2085        | LogicalType::Timestamp
2086        | LogicalType::Time
2087        | LogicalType::TimeTz
2088        | LogicalType::TimestampTz
2089        | LogicalType::TimestampS
2090        | LogicalType::TimestampMs
2091        | LogicalType::TimestampNs
2092        | LogicalType::Double
2093        | LogicalType::Decimal { .. } => 8,
2094        LogicalType::Integer | LogicalType::UInteger | LogicalType::Date | LogicalType::Float => 4,
2095        LogicalType::SmallInt | LogicalType::USmallInt => 2,
2096        _ => 1,
2097    }
2098}
2099
2100/// Parts in one stripe.
2101///
2102/// Sixty four thousand rows is the smallest stripe that keeps the ClickBench directory in single
2103/// digit megabytes at a hundred million rows, and it puts a four byte column's page at a quarter of
2104/// a megabyte, which is the size a sequential read wants. Larger stripes buy a smaller directory
2105/// and cost a sparse fetch, which has to read a page index before it can reach one part.
2106pub const STRIPE_PARTS: usize = 64;
2107
2108/// How many rows the writer wants to see before it decides whether a varchar column gets to keep
2109/// its global dictionary.
2110///
2111/// See [`prepare::drops_dictionary`]. A stripe is up to [`STRIPE_PARTS`] parts, so most tables give it
2112/// far more than this and it binds only on a table that is smaller than one stripe. A handful of
2113/// rows says nothing about whether a column repeats itself, and the answer that costs nothing when
2114/// the sample is that small is the one the writer has always given, which is to keep the dictionary.
2115const DICTIONARY_DECIDE_ROWS: usize = 4_096;
2116
2117/// Out of ten. A varchar column loses its dictionary when more than this many rows in ten of the
2118/// first stripe held a value that stripe had not seen before.
2119///
2120/// See [`prepare::drops_dictionary`]. Nine and not five, because the properties a dictionary buys are
2121/// worth keeping everywhere they are real. On ClickBench the widest string column is `Referer` at
2122/// 0.131 of its first stripe and every other one is below that, so nothing there is near this and
2123/// every one of them keeps its dictionary, which is what a group by on codes wants. On TPC-H
2124/// `o_comment` and `c_comment` are at 0.97 and are what this catches.
2125///
2126/// `l_comment` sits at 0.883 and so keeps its dictionary. Eight was built and measured rather than
2127/// argued about, and it is not a clear win: it takes `select l_comment from lineitem` from 4.335 G
2128/// instructions to 3.473 G and the file from 280.2 MB to 260.4 MB, and it takes a `like` over the
2129/// same column from 3.29 G to 4.27 G, because a dictionary runs the predicate once a distinct value
2130/// and there are 3.6 M of those to 6.0 M rows. The 22 query suite came out 6.91 s against 7.07 s in
2131/// favour of nine. So nine stays until there is a reason to prefer one of those shapes. See #1137.
2132const DICTIONARY_DISTINCT_IN_TEN: usize = 9;
2133
2134/// Bytes one part takes in a stripe's index page: four for the length, eight for the checksum.
2135const INDEX_ENTRY: usize = size_of::<u32>() + size_of::<u64>();
2136
2137/// Bytes one column's section of a stripe's index page takes, including its own trailing checksum.
2138fn index_section(parts: usize) -> Result<usize> {
2139    parts
2140        .checked_mul(INDEX_ENTRY)
2141        .and_then(|bytes| bytes.checked_add(size_of::<u64>()))
2142        .ok_or_else(|| invalid("index page length overflow"))
2143}
2144
2145impl Writer {
2146    /// Opens a committed file and starts a table in the generation after the one it holds.
2147    ///
2148    /// The tables already in the file are carried forward by name and by directory pointer, and
2149    /// their pages are not read. Nothing in the file is overwritten: the new table's pages and the
2150    /// new catalog go on the end, past the catalog the committed generation points at, and the one
2151    /// write that is not an append is the slot in the header that [`Writer::finish`] does last.
2152    ///
2153    /// That slot is the other one. A file committed at generation 1 is named by the slot at 16 and
2154    /// generation 2 writes the one at 44, so until the last four bytes of the commit land the file
2155    /// still reads as the generation before it, and a slot torn across a write fails its checksum
2156    /// and the reader falls back to the one beside it. This is what the second slot has always been
2157    /// for.
2158    ///
2159    /// # Errors
2160    ///
2161    /// If the file has no valid committed directory, is not this build's format, repeats the name
2162    /// of a table already in it that holds rows, has a field with no scalar encoding, or cannot be
2163    /// written.
2164    pub fn open(
2165        path: impl AsRef<Path>,
2166        name: impl Into<String>,
2167        fields: Vec<Field>,
2168    ) -> Result<Self> {
2169        Self::open_in(&RealFilesystem::new(), path, name, fields)
2170    }
2171
2172    /// [`Writer::open`] on a file in `fs`, which is how a crash test runs an append against the
2173    /// simulated filesystem.
2174    ///
2175    /// # Errors
2176    ///
2177    /// The same as [`Writer::open`].
2178    pub fn open_in(
2179        fs: &dyn Filesystem,
2180        path: impl AsRef<Path>,
2181        name: impl Into<String>,
2182        fields: Vec<Field>,
2183    ) -> Result<Self> {
2184        for field in &fields {
2185            type_tag(&field.ty)?;
2186        }
2187        let name = name.into();
2188        let file = fs.open(path.as_ref(), OpenMode::ReadWrite)?;
2189        let size = file.len()?;
2190        let (slot, bytes, _) = committed_slot(&*file, size)?;
2191        let (mut closed, views) = decode_catalog(&bytes, size)?;
2192        // A table already in the file under this name is only in the way if it holds rows. One that
2193        // holds none has no pages for this generation to carry and no reader that could lose
2194        // anything, so the table being started here takes its place in the catalog rather than
2195        // colliding with it, and `finish` writes the new entry where the old one was.
2196        //
2197        // That is not a corner. It is the shape every loading script writes: the schema goes in one
2198        // statement and the rows go in the next, and a checkpoint between them commits the empty
2199        // table. Before this, the second statement had to build the whole table in memory because
2200        // the first had already put the name in the file, which is how a load of a table larger
2201        // than memory became a load that needed memory the size of the table.
2202        if let Some(at) = closed.iter().position(|held| held.name == name) {
2203            if closed[at].rows > 0 {
2204                return Err(invalid("two tables in one native file have the same name"));
2205            }
2206            closed.remove(at);
2207        }
2208        // The generation of the slot whose bytes checksummed, and not the highest number in the
2209        // header. A slot torn across a write can hold any number at all, and taking that one would
2210        // be choosing which slot to overwrite from a value nothing has vouched for, which is how a
2211        // half written commit gets to destroy the one good copy beside it.
2212        let generation = slot
2213            .generation
2214            .checked_add(1)
2215            .ok_or_else(|| invalid("native file generation overflow"))?;
2216        Ok(Self {
2217            file,
2218            // The end of the file, so that the committed generation's catalog stays where its slot
2219            // says it is and keeps naming a file a reader can still open.
2220            at: size,
2221            written_back: size,
2222            dictionaries: fields
2223                .iter()
2224                .map(|field| coded_type(&field.ty).then(GlobalDictionary::new))
2225                .collect(),
2226            coded: Arc::new(prepare::Coding::new(fields.iter().map(|field| coded_type(&field.ty)))),
2227            gathers: fields.iter().map(|field| stats::Gather::new(&field.ty, generation)).collect(),
2228            lent: None,
2229            table: Table {
2230                name,
2231                dictionaries: vec![None; fields.len()],
2232                dictionary_payloads: Vec::new(),
2233                demoted: Vec::new(),
2234                distincts: vec![None; fields.len()],
2235                fields,
2236                stripes: Vec::new(),
2237                rows: 0,
2238                frequencies: Vec::new(),
2239                pair_frequencies: Vec::new(),
2240                frequency_texts: Vec::new(),
2241                host_groups: None,
2242                clustering: None,
2243                generation,
2244                sections: Vec::new(),
2245            },
2246            generation,
2247            order: Vec::new(),
2248            next_order: 0,
2249            pending: Vec::with_capacity(STRIPE_PARTS),
2250            closed,
2251            views,
2252            profile: None,
2253        })
2254    }
2255
2256    /// Creates a new v10 file and its first table.
2257    ///
2258    /// # Errors
2259    ///
2260    /// If the file exists, a field has no scalar encoding, or the path cannot be written.
2261    pub fn create(
2262        path: impl AsRef<Path>,
2263        name: impl Into<String>,
2264        fields: Vec<Field>,
2265    ) -> Result<Self> {
2266        Self::create_in(&RealFilesystem::new(), path, name, fields)
2267    }
2268
2269    /// [`Writer::create`] with the file made in `fs` rather than on the real filesystem.
2270    ///
2271    /// Every call the writer makes on the file from here to [`Writer::finish`] goes to that
2272    /// filesystem, which is what lets a test built on `rudb_io::SimFilesystem` stop a load at any
2273    /// one of them and look at what a crash there would leave on the disk.
2274    ///
2275    /// # Errors
2276    ///
2277    /// The same as [`Writer::create`].
2278    pub fn create_in(
2279        fs: &dyn Filesystem,
2280        path: impl AsRef<Path>,
2281        name: impl Into<String>,
2282        fields: Vec<Field>,
2283    ) -> Result<Self> {
2284        for field in &fields {
2285            type_tag(&field.ty)?;
2286        }
2287        let file = fs.open(path.as_ref(), OpenMode::CreateNew)?;
2288        let mut header = [0; HEADER as usize];
2289        header[..8].copy_from_slice(MAGIC);
2290        header[8..12].copy_from_slice(&FORMAT.to_le_bytes());
2291        file.write_at(0, &header)?;
2292        Ok(Self {
2293            file,
2294            at: HEADER,
2295            written_back: HEADER,
2296            dictionaries: fields
2297                .iter()
2298                .map(|field| coded_type(&field.ty).then(GlobalDictionary::new))
2299                .collect(),
2300            coded: Arc::new(prepare::Coding::new(fields.iter().map(|field| coded_type(&field.ty)))),
2301            gathers: fields.iter().map(|field| stats::Gather::new(&field.ty, 1)).collect(),
2302            lent: None,
2303            table: Table {
2304                name: name.into(),
2305                dictionaries: vec![None; fields.len()],
2306                dictionary_payloads: Vec::new(),
2307                demoted: Vec::new(),
2308                distincts: vec![None; fields.len()],
2309                fields,
2310                stripes: Vec::new(),
2311                rows: 0,
2312                frequencies: Vec::new(),
2313                pair_frequencies: Vec::new(),
2314                frequency_texts: Vec::new(),
2315                host_groups: None,
2316                clustering: None,
2317                generation: 1,
2318                sections: Vec::new(),
2319            },
2320            generation: 1,
2321            order: Vec::new(),
2322            next_order: 0,
2323            pending: Vec::with_capacity(STRIPE_PARTS),
2324            closed: Vec::new(),
2325            views: Vec::new(),
2326            profile: None,
2327        })
2328    }
2329
2330    /// Creates a new file that holds no table at all, committed and ready to open.
2331    ///
2332    /// A database somebody dropped the last table out of is still a database, and until this there
2333    /// was no way to write one down. Every other way into this file goes through a table, because
2334    /// [`Writer::create`] takes the first one and [`Writer::finish`] commits the one it is on, so a
2335    /// catalog with nothing in it could be read and not written. The format already allowed it: the
2336    /// catalog is a count and that many entries, and a count of nought encodes and decodes the same
2337    /// way every other count does, which is why nothing here is a version change.
2338    ///
2339    /// It hands back nothing rather than a writer, because a writer with no table is a writer with
2340    /// nothing to append to. A file that is going to hold a table is [`Writer::create`], and one
2341    /// that is going to have a table added to it later is [`Writer::open`], which reads what this
2342    /// wrote the same way it reads any other generation.
2343    ///
2344    /// It takes the views anyway, because a database with no table can still have views in it. A
2345    /// view over `range` or over another view names no table, so dropping the last table out of a
2346    /// database does not have to leave the catalog with nothing worth writing down.
2347    ///
2348    /// # Errors
2349    ///
2350    /// If the file exists or the path cannot be written.
2351    pub fn empty(path: impl AsRef<Path>, views: &[ViewEntry]) -> Result<()> {
2352        let file = RealFilesystem::new().open(path.as_ref(), OpenMode::CreateNew)?;
2353        let mut header = [0; HEADER as usize];
2354        header[..8].copy_from_slice(MAGIC);
2355        header[8..12].copy_from_slice(&FORMAT.to_le_bytes());
2356        file.write_at(0, &header)?;
2357        let catalog = encode_catalog(&[], views)?;
2358        file.write_at(HEADER, &catalog)?;
2359        // The same two syncs in the same order as [`Writer::finish`], and for the same reason. The
2360        // catalog is on the disk before the slot names it, so a file this is interrupted in the
2361        // middle of is a header with no valid slot rather than a slot pointing at nothing.
2362        file.sync()?;
2363        let slot = Slot {
2364            offset: HEADER,
2365            length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
2366            generation: 1,
2367            hash: checksum(&catalog),
2368        };
2369        file.write_at(slot_offset(1), &slot.bytes())?;
2370        file.sync()?;
2371        Ok(())
2372    }
2373
2374    /// Closes the table this writer is on and starts another one in the same file.
2375    ///
2376    /// Nothing is published here. The closed table's directory is written so that the bytes are on
2377    /// disk and its span is known, and the catalog that names it is only written by
2378    /// [`Writer::finish`], so a crash between two tables leaves the previous generation intact.
2379    ///
2380    /// # Errors
2381    ///
2382    /// If the name repeats a table already closed, a field has no scalar encoding, or the table
2383    /// being closed cannot be written.
2384    pub fn next(mut self, name: impl Into<String>, fields: Vec<Field>) -> Result<Self> {
2385        for field in &fields {
2386            type_tag(&field.ty)?;
2387        }
2388        let name = name.into();
2389        let entry = self.close()?;
2390        if self.closed.iter().chain(std::iter::once(&entry)).any(|held| held.name == name) {
2391            return Err(invalid("two tables in one native file have the same name"));
2392        }
2393        let Self { file, at, generation, mut closed, views, .. } = self;
2394        closed.push(entry);
2395        Ok(Self {
2396            file,
2397            written_back: at,
2398            at,
2399            generation,
2400            closed,
2401            views,
2402            profile: None,
2403            dictionaries: fields
2404                .iter()
2405                .map(|field| coded_type(&field.ty).then(GlobalDictionary::new))
2406                .collect(),
2407            coded: Arc::new(prepare::Coding::new(fields.iter().map(|field| coded_type(&field.ty)))),
2408            gathers: fields.iter().map(|field| stats::Gather::new(&field.ty, generation)).collect(),
2409            lent: None,
2410            table: Table {
2411                name,
2412                dictionaries: vec![None; fields.len()],
2413                dictionary_payloads: Vec::new(),
2414                demoted: Vec::new(),
2415                distincts: vec![None; fields.len()],
2416                fields,
2417                stripes: Vec::new(),
2418                rows: 0,
2419                frequencies: Vec::new(),
2420                pair_frequencies: Vec::new(),
2421                frequency_texts: Vec::new(),
2422                host_groups: None,
2423                clustering: None,
2424                generation,
2425                sections: Vec::new(),
2426            },
2427            order: Vec::new(),
2428            next_order: 0,
2429            pending: Vec::with_capacity(STRIPE_PARTS),
2430        })
2431    }
2432
2433    /// Sets the views the next commit writes down, replacing whatever was carried forward.
2434    ///
2435    /// It replaces rather than adds because the caller has the whole catalog in front of it and the
2436    /// writer does not. A view that was dropped is a view that is not in the list any more, and
2437    /// there is no other way for the writer to hear about that, since nothing else it is told about
2438    /// mentions views at all.
2439    ///
2440    /// A writer that is never told anything writes back the views it read at [`Writer::open`], so a
2441    /// checkpoint that only had a table to append does not quietly drop them.
2442    #[must_use]
2443    pub fn with_views(mut self, views: Vec<ViewEntry>) -> Self {
2444        self.views = views;
2445        self
2446    }
2447
2448    /// Charges the stages this writer runs to `profile`.
2449    ///
2450    /// For the table being written now. [`Writer::next`] starts the next table without one,
2451    /// because a second table's stripes charged to the first table's load would be a profile of
2452    /// neither.
2453    #[must_use]
2454    pub fn with_profile(mut self, profile: Arc<LoadProfile>) -> Self {
2455        self.profile = Some(profile);
2456        self
2457    }
2458
2459    /// Sets what the table's global dictionaries may hold between them before the one growing
2460    /// fastest stops taking values, which is [`DICTIONARY_CAP_BYTES`] unless this says
2461    /// otherwise. It applies to every [`Preparer`] and [`Merger`] this writer has handed out too.
2462    #[must_use]
2463    pub fn with_dictionary_cap(self, bytes: u64) -> Self {
2464        self.coded.cap(bytes);
2465        self
2466    }
2467
2468    /// Records the order this table's rows are meant to be stored in.
2469    ///
2470    /// The declaration goes in the table directory and comes back out of
2471    /// [`Table::clustering`]. Nothing here sorts anything, and nothing here checks that the rows
2472    /// handed to [`Writer::append`] arrive in the order this claims. That is deliberate for now:
2473    /// the thing that was missing was a place to write the order down, and a loader that honours
2474    /// the declaration is the next piece rather than this one.
2475    ///
2476    /// The declaration applies to the table the writer is currently on, so it is set after
2477    /// [`Writer::next`] rather than once for the file.
2478    ///
2479    /// # Errors
2480    ///
2481    /// If the declaration names a column this table does not have.
2482    pub fn declare(mut self, clustering: Clustering) -> Result<Self> {
2483        // Rebuilt against this table's own column count rather than trusted, because the caller
2484        // built it against a catalog entry and the two could have drifted.
2485        self.table.clustering = Some(Clustering::new(
2486            clustering.columns().to_vec(),
2487            clustering.width(),
2488            &self.table.fields,
2489        )?);
2490        Ok(self)
2491    }
2492
2493    /// Appends bytes at the end of the file and moves the writer's own offset past them.
2494    ///
2495    /// Every write in here goes through this, so that [`Writer::at`] is the only answer to where
2496    /// anything is and the file's cursor is never consulted for it.
2497    fn put(&mut self, bytes: &[u8]) -> Result<()> {
2498        self.file.write_at(self.at, bytes)?;
2499        self.at = self
2500            .at
2501            .checked_add(bytes.len() as u64)
2502            .ok_or_else(|| invalid("native file length overflow"))?;
2503        if self.at - self.written_back >= WRITEBACK_STRETCH {
2504            self.file.start_writeback(self.written_back, self.at - self.written_back);
2505            self.written_back = self.at;
2506        }
2507        Ok(())
2508    }
2509
2510    /// Writes one chunk as independently readable column pages.
2511    ///
2512    /// # Errors
2513    ///
2514    /// If its width or types differ from the declared table, or a page exceeds its bound.
2515    pub fn append(&mut self, chunk: &Chunk) -> Result<()> {
2516        let order = (self.next_order, 0);
2517        self.next_order = self.next_order.saturating_add(1);
2518        self.append_at(order, chunk)
2519    }
2520
2521    /// Writes one chunk and records its source position for directory ordering.
2522    ///
2523    /// Pages may be encoded by parallel pipeline instances and reach the file in completion order.
2524    /// The stripe they land in is sorted by this key at commit, and [`Self::finish`] rejects a
2525    /// sequence whose parts do not come out in source order once the stripes are sorted, because a
2526    /// stripe groups whatever arrived together and cannot put a late part back where it belongs.
2527    ///
2528    /// # Errors
2529    ///
2530    /// The same as [`Self::append`].
2531    pub fn append_at(&mut self, order: (u64, u64), chunk: &Chunk) -> Result<()> {
2532        if chunk.is_empty() {
2533            return Ok(());
2534        }
2535        self.admit(chunk)?;
2536        if self.pending.last().is_some_and(|last| last.order > order) {
2537            self.flush_pending()?;
2538        }
2539        // Cloned rather than encoded, and a clone of a chunk that owns its buffers is a copy of
2540        // them. Sixty four parts of a hundred and five columns is tens of megabytes held for the
2541        // length of a stripe and a few seconds of memory traffic over a whole ClickBench load,
2542        // against the hundreds of seconds of encode this is what lets off one thread.
2543        self.pending.push(PendingChunk { order, chunk: chunk.clone() });
2544        if self.pending.len() == STRIPE_PARTS {
2545            self.flush_pending()?;
2546        }
2547        Ok(())
2548    }
2549
2550    /// Writes a run of chunks as one stripe of its own.
2551    ///
2552    /// [`Self::append_at`] decides where a stripe ends by watching the orders go past, which works
2553    /// when one caller hands over every chunk in source order and does not when several do. A
2554    /// writer being fed by more than one pipeline instance sees the orders interleave, and a stripe
2555    /// that ends every time two of them cross is a stripe of one or two parts.
2556    ///
2557    /// So the grouping moves to the caller. Whoever is buffering hands over a run it already knows
2558    /// is contiguous and in order, and gets a stripe holding exactly that run. The orders still
2559    /// have to come out in source order once the stripes are sorted, which [`Self::finish`] checks,
2560    /// so the runs from different callers may interleave with each other but may not overlap.
2561    ///
2562    /// # Errors
2563    ///
2564    /// The same as [`Self::append`], and if the run is longer than [`STRIPE_PARTS`].
2565    pub fn append_stripe(&mut self, parts: Vec<((u64, u64), Chunk)>) -> Result<()> {
2566        if parts.len() > STRIPE_PARTS {
2567            return Err(invalid("a stripe was handed more parts than it holds"));
2568        }
2569        // Whatever an earlier caller left behind is its own stripe rather than the front of this
2570        // one, because the two runs are from different places in the source and a stripe is a run.
2571        self.flush_pending()?;
2572        for (order, chunk) in parts {
2573            if chunk.is_empty() {
2574                continue;
2575            }
2576            self.admit(&chunk)?;
2577            self.pending.push(PendingChunk { order, chunk });
2578        }
2579        self.flush_pending()
2580    }
2581
2582    /// Checks a chunk against the declared table and counts its rows in.
2583    fn admit(&mut self, chunk: &Chunk) -> Result<()> {
2584        if chunk.width() != self.table.fields.len() {
2585            return Err(invalid("chunk width differs from table schema"));
2586        }
2587        for (index, field) in self.table.fields.iter().enumerate() {
2588            if chunk.column(index)?.logical_type() != &field.ty {
2589                return Err(invalid("chunk type differs from table schema"));
2590            }
2591        }
2592        self.table.rows = self
2593            .table
2594            .rows
2595            .checked_add(chunk.len())
2596            .ok_or_else(|| invalid("row count overflow"))?;
2597        Ok(())
2598    }
2599
2600    /// One column's parts of a stripe as pages, for a column with no global dictionary.
2601    fn encode_pages(columns: &[&Vector]) -> Result<ColumnStripe> {
2602        let mut stripe = ColumnStripe {
2603            pages: Vec::with_capacity(columns.len()),
2604            codes: Vec::with_capacity(columns.len()),
2605            sieves: Vec::with_capacity(columns.len()),
2606            ranges: Vec::with_capacity(columns.len()),
2607        };
2608        let mut settling = Settling::default();
2609        for &column in columns {
2610            let bytes = encode(column, &mut settling)?;
2611            if bytes.len() > MAX_PAGE {
2612                return Err(invalid("column page exceeds the configured bound"));
2613            }
2614            // The range is built first because the sieve reads it rather than walking the column a
2615            // second time to find out how wide it is.
2616            let range = Range::of(column);
2617            // A sieve at least as large as the part it indexes is not written. A reader reads the
2618            // sieve to decide whether to read the part, so when the sieve is the larger of the two
2619            // it has already spent more than the read it is trying to avoid, and that holds even if
2620            // it rejects every time. It is a necessary condition rather than the whole rule, which
2621            // is that a sieve pays when its bytes are under the rejection rate times the part's,
2622            // but the rejection rate depends on what a query probes for and the writer does not
2623            // know that. The necessary half needs two numbers that are both in hand here.
2624            //
2625            // A column with a global dictionary gets none, because it already has an exact
2626            // membership index per stripe. Those do not come through here. See [`prepare`].
2627            let sieve =
2628                Sieve::of(column, &range, SIEVE_BUDGET).filter(|sieve| sieve.len() < bytes.len());
2629            stripe.pages.push(bytes);
2630            stripe.codes.push(None);
2631            stripe.sieves.push(sieve);
2632            stripe.ranges.push(range);
2633        }
2634        Ok(stripe)
2635    }
2636
2637    /// Writes every encoded dictionary block that is not in the file yet and forgets its bytes.
2638    ///
2639    /// This is what keeps a load from holding its dictionaries' payload. The blocks land between
2640    /// stripes wherever the writer is, which is fine because the index says where each one is.
2641    fn place_blocks(&mut self) -> Result<()> {
2642        if let Some(lent) = self.lent.clone() {
2643            return self.place_lent_blocks(&lent);
2644        }
2645        let mut dictionaries = std::mem::take(&mut self.dictionaries);
2646        let placed = dictionaries.iter_mut().flatten().try_for_each(|dictionary| {
2647            for block in std::mem::take(&mut dictionary.blocks) {
2648                let start = self.at;
2649                self.put(&block)?;
2650                dictionary.placed.push(Placed {
2651                    start,
2652                    length: block.len() as u64,
2653                    hash: checksum(&block),
2654                });
2655            }
2656            Ok(())
2657        });
2658        self.dictionaries = dictionaries;
2659        placed
2660    }
2661
2662    /// [`Writer::place_blocks`] while a [`Merger`] has the dictionaries.
2663    ///
2664    /// A column whose merge is running is passed over rather than waited for, because the writer's
2665    /// lock is held here and a merge of `URL` can take tens of milliseconds. Its blocks go out with
2666    /// a later stripe, or at the close.
2667    fn place_lent_blocks(&mut self, lent: &Lent) -> Result<()> {
2668        for column in lent.columns() {
2669            let Ok(mut held) = column.try_lock() else { continue };
2670            let Some(dictionary) = held.dictionary.as_mut() else { continue };
2671            for block in std::mem::take(&mut dictionary.blocks) {
2672                let start = self.at;
2673                self.put(&block)?;
2674                dictionary.placed.push(Placed {
2675                    start,
2676                    length: block.len() as u64,
2677                    hash: checksum(&block),
2678                });
2679            }
2680        }
2681        Ok(())
2682    }
2683
2684    /// Takes the dictionaries and the statistics back from the [`Merger`] that has them.
2685    ///
2686    /// A merge that starts after this is refused, since whatever it merged would be lost.
2687    fn reclaim(&mut self) -> Result<()> {
2688        let Some(lent) = self.lent.take() else { return Ok(()) };
2689        let (dictionaries, gathers) = lent.reclaim()?;
2690        self.dictionaries = dictionaries;
2691        self.gathers = gathers;
2692        Ok(())
2693    }
2694
2695    /// Writes the buffered parts as one stripe, each column's parts contiguous on disk.
2696    ///
2697    /// The same four steps a caller holding this writer behind a lock takes, with nobody else
2698    /// waiting between them. See [`prepare`].
2699    fn flush_pending(&mut self) -> Result<()> {
2700        if self.pending.is_empty() {
2701            return Ok(());
2702        }
2703        let held = std::mem::take(&mut self.pending);
2704        let prepared = self.preparer().prepare_held(held)?;
2705        let merged = self.merge_held(prepared)?;
2706        let paged = merged.pages()?;
2707        self.write_paged(paged)
2708    }
2709
2710    /// Writes one stripe whose pages are built, each column's parts contiguous on disk.
2711    fn write_stripe(&mut self, held: &[Part], encoded: Vec<ColumnStripe>) -> Result<()> {
2712        let width = self.table.fields.len();
2713        let parts = held.len();
2714        if encoded.len() != width {
2715            return Err(Error::internal("a stripe came to the writer with the wrong columns"));
2716        }
2717        let profile = self.profile.clone();
2718        if let Some(profile) = &profile {
2719            let rows = held.iter().map(|part| part.rows as u64).sum();
2720            let raw = held.iter().map(|part| part.footprint as u64).sum();
2721            let pages =
2722                encoded.iter().flat_map(|stripe| &stripe.pages).map(|page| page.len() as u64).sum();
2723            profile.moved(Stage::Pages, raw, pages, rows);
2724        }
2725        // Before a byte of the stripe is written, so that the blocks the stripe's pages were built
2726        // with, and any that were waiting on them, are let go of now rather than a stripe later.
2727        let timing = profile.as_deref().map(|profile| profile.span(Stage::Dictionary));
2728        let before = self.at;
2729        self.place_blocks()?;
2730        drop(timing);
2731        if let Some(profile) = &profile {
2732            profile.moved(Stage::Dictionary, 0, self.at - before, 0);
2733        }
2734        let timing = profile.as_deref().map(|profile| profile.span(Stage::Write));
2735        let before = self.at;
2736        let mut pages = Vec::with_capacity(width);
2737        let mut memberships = vec![None; width];
2738        let mut ranges = Vec::with_capacity(width);
2739        let mut index = Vec::with_capacity(width.saturating_mul(index_section(parts)?));
2740        for stripe in &encoded {
2741            let offset = self.at;
2742            let section = index.len();
2743            let mut length = 0_usize;
2744            for bytes in &stripe.pages {
2745                self.file.write_at(self.at + length as u64, bytes)?;
2746                put_u32(
2747                    &mut index,
2748                    u32::try_from(bytes.len()).map_err(|_| invalid("part length overflow"))?,
2749                );
2750                put_u64(&mut index, checksum(bytes));
2751                length = length
2752                    .checked_add(bytes.len())
2753                    .ok_or_else(|| invalid("column page length overflow"))?;
2754            }
2755            let hash = checksum(&index[section..]);
2756            put_u64(&mut index, hash);
2757            if length > MAX_PAGE {
2758                return Err(invalid("column page exceeds the configured bound"));
2759            }
2760            self.at = self
2761                .at
2762                .checked_add(length as u64)
2763                .ok_or_else(|| invalid("native file length overflow"))?;
2764            pages.push(Span {
2765                offset,
2766                length: u32::try_from(length).map_err(|_| invalid("page length overflow"))?,
2767            });
2768            ranges.push(merged_range(stripe.ranges.iter().cloned()));
2769        }
2770        for (membership, stripe) in memberships.iter_mut().zip(&encoded) {
2771            if stripe.codes.iter().all(Option::is_none) {
2772                continue;
2773            }
2774            let lists = stripe
2775                .codes
2776                .iter()
2777                .map(|codes| codes.clone().unwrap_or_default())
2778                .collect::<Vec<_>>();
2779            let bytes = encode_membership(&merged_codes(lists));
2780            let offset = self.at;
2781            self.put(&bytes)?;
2782            *membership = Some(Page {
2783                offset,
2784                length: u32::try_from(bytes.len())
2785                    .map_err(|_| invalid("membership page length overflow"))?,
2786                hash: checksum(&bytes),
2787            });
2788        }
2789        let mut sieves = vec![None; width];
2790        for (page, stripe) in sieves.iter_mut().zip(&encoded) {
2791            if stripe.sieves.iter().all(Option::is_none) {
2792                continue;
2793            }
2794            let bytes = encode_sieves(stripe.sieves.iter())?;
2795            let offset = self.at;
2796            self.put(&bytes)?;
2797            *page = Some(Page {
2798                offset,
2799                length: u32::try_from(bytes.len())
2800                    .map_err(|_| invalid("sieve page length overflow"))?,
2801                hash: checksum(&bytes),
2802            });
2803        }
2804        // A stripe of one part has the same rows in it as that part, so its own bounds are already
2805        // the part's and a page here would say what the directory says. Everywhere else the page is
2806        // written unless it comes to more than the column it indexes, which is the rule the sieves
2807        // go by and for the same reason: a reader reads this to decide whether to read the column,
2808        // so a page larger than the column has spent more than the read it is avoiding.
2809        let mut part_ranges = vec![None; width];
2810        if parts > 1 {
2811            for ((page, stripe), span) in part_ranges.iter_mut().zip(&encoded).zip(&pages) {
2812                let bytes = encode_part_ranges(&stripe.ranges)?;
2813                if bytes.len() >= span.length as usize {
2814                    continue;
2815                }
2816                let offset = self.at;
2817                self.put(&bytes)?;
2818                *page = Some(Page {
2819                    offset,
2820                    length: u32::try_from(bytes.len())
2821                        .map_err(|_| invalid("part range page length overflow"))?,
2822                    hash: checksum(&bytes),
2823                });
2824            }
2825        }
2826        let offset = self.at;
2827        self.put(&index)?;
2828        let index = Span {
2829            offset,
2830            length: u32::try_from(index.len())
2831                .map_err(|_| invalid("index page length overflow"))?,
2832        };
2833        let mut rows = 0_usize;
2834        let mut lengths = Vec::with_capacity(parts);
2835        let mut span = None;
2836        for part in held {
2837            rows = rows.checked_add(part.rows).ok_or_else(|| invalid("row count overflow"))?;
2838            lengths.push(u32::try_from(part.rows).map_err(|_| invalid("part row count overflow"))?);
2839            span = Some(span.map_or((part.order, part.order), |(first, _)| (first, part.order)));
2840        }
2841        self.order.push(span.ok_or_else(|| invalid("a stripe was flushed with no parts"))?);
2842        self.table.stripes.push(Stripe {
2843            rows,
2844            parts: lengths,
2845            index,
2846            pages,
2847            memberships: Pages::from_slots(memberships)?,
2848            sieves: Pages::from_slots(sieves)?,
2849            part_ranges: Pages::from_slots(part_ranges)?,
2850            zone: Zone::from_ranges(ranges),
2851        });
2852        drop(timing);
2853        if let Some(profile) = &profile {
2854            profile.moved(Stage::Write, 0, self.at - before, rows as u64);
2855        }
2856        Ok(())
2857    }
2858
2859    /// Finds exact heavy hitters without keeping a hash table for every numeric column while the
2860    /// load is live. The pages are already in the target file, so one column at a time uses a
2861    /// bounded Misra-Gries candidate table and then recounts only those candidates.
2862    ///
2863    /// The first of those passes also counts the column's distinct values exactly, up to the cap in
2864    /// [`distinct`], which is the number a string column gets from its dictionary. It comes back
2865    /// beside the summary because a column whose heavy hitters cannot be proved can still have been
2866    /// counted.
2867    ///
2868    /// The tables are keyed by a value's sixty four bits rather than by [`FrequencyValue`], and a
2869    /// null is counted beside them. Every integer type the format stores fits in those bits, so
2870    /// within one column two values share bits only if they are the same value, and a sixteen byte
2871    /// entry keeps the whole candidate table in the second level cache where the forty eight byte
2872    /// one did not. The null takes part in the candidate table exactly as a key would: it holds a
2873    /// place while its count is above zero, and it is decremented with the rest.
2874    ///
2875    /// `counted` is false for a column whose sketch says its distinct values are far past what the
2876    /// exact set holds. It still gets its frequencies, and a count only if it turns out to have
2877    /// fewer values than the candidate table, which is the count that costs nothing.
2878    fn numeric_frequency(
2879        &self,
2880        column: usize,
2881        counted: bool,
2882    ) -> Result<(Option<FrequencySummary>, Option<u64>)> {
2883        let signed = match self.table.fields[column].ty {
2884            LogicalType::TinyInt
2885            | LogicalType::SmallInt
2886            | LogicalType::Integer
2887            | LogicalType::BigInt
2888            | LogicalType::Date
2889            | LogicalType::Timestamp => true,
2890            LogicalType::UTinyInt
2891            | LogicalType::USmallInt
2892            | LogicalType::UInteger
2893            | LogicalType::UBigInt => false,
2894            _ => return Ok((None, None)),
2895        };
2896        let value_of = |bits: Option<u64>| match bits {
2897            None => FrequencyValue::Null,
2898            Some(bits) if signed => FrequencyValue::Integer(i128::from(bits as i64)),
2899            Some(bits) => FrequencyValue::Integer(i128::from(bits)),
2900        };
2901        // A column the writer's tally held whole has its exact counts already, gathered as the rows
2902        // went past, so the pages are not read back to count them again. On `hits` that is most of
2903        // the flag and enum columns. The tally only speaks for the whole column when it saw every
2904        // row, which is the same check the statistics make before they are written.
2905        let tallied = self
2906            .gathers
2907            .get(column)
2908            .and_then(Option::as_ref)
2909            .filter(|gather| gather.rows() == self.table.rows as u64)
2910            .and_then(stats::Gather::frequencies)
2911            .and_then(|(values, nulls)| {
2912                let exact = values
2913                    .iter()
2914                    .map(|(value, count)| Some((frequency_bits(value)?, *count)))
2915                    .collect::<Option<FrequencyMap<_>>>()?;
2916                Some((exact, (nulls != 0).then_some(nulls), values.len() as u64))
2917            });
2918        let (exact, null_count, decrements, distinct_count) = match tallied {
2919            Some((exact, null_count, distinct)) => (exact, null_count, 0, Some(distinct)),
2920            None => {
2921                // Rows arrive a run of equal values at a time, because a sorted column is runs and
2922                // a flag column is mostly one value, so a run is counted and inserted once rather
2923                // than per row.
2924                //
2925                // The exact distinct count is left alone until the candidate table is full. Until
2926                // then no candidate has been decremented, so the table holds every value the column
2927                // has had and its size is the count. Most columns never fill it and so never build
2928                // the set. The one that fills it hands the set everything it holds at that moment,
2929                // plus the run it is about to add, and the set carries on from there as it always
2930                // did.
2931                let mut first = Candidates::default();
2932                let mut distinct: Option<distinct::ExactDistinct> = None;
2933                let mut run = Run::default();
2934                self.visit_numeric(column, signed, |_, bits| {
2935                    if let Some((ended, times)) = run.push(bits) {
2936                        count_from_full(&first, &mut distinct, ended, counted);
2937                        first.add(ended, times);
2938                    }
2939                    if run.times == 1 {
2940                        if let (Some(distinct), Some(bits)) = (distinct.as_mut(), bits) {
2941                            distinct.insert(bits);
2942                        }
2943                    }
2944                })?;
2945                if let Some((bits, times)) = run.take() {
2946                    count_from_full(&first, &mut distinct, bits, counted);
2947                    first.add(bits, times);
2948                }
2949                let distinct_count = match distinct.as_mut() {
2950                    Some(distinct) => distinct.count(),
2951                    None => Some(first.held as u64),
2952                };
2953                let (nulls, decrements) = (first.nulls, first.decrements);
2954                let (exact, null_count) = if decrements == 0 {
2955                    let exact = first
2956                        .pairs()
2957                        .map(|(bits, count)| (bits, u64::from(count)))
2958                        .collect::<FrequencyMap<_>>();
2959                    (exact, (nulls != 0).then_some(u64::from(nulls)))
2960                } else {
2961                    let mut lower = first.pairs().map(|(_, count)| count).collect::<Vec<_>>();
2962                    if nulls != 0 {
2963                        lower.push(nulls);
2964                    }
2965                    lower.sort_unstable_by(|left, right| right.cmp(left));
2966                    if lower.len() < FREQUENCY_BUILD_RANK
2967                        || u64::from(lower[FREQUENCY_BUILD_RANK - 1]) <= decrements
2968                    {
2969                        return Ok((None, distinct_count));
2970                    }
2971                    // Counted beside the slot each candidate sits in, since the table is not
2972                    // changed again and a lookup in it is the one probe the first pass made.
2973                    let mut recounts = vec![0_u64; first.slots.len()];
2974                    let mut null_count = (nulls != 0).then_some(0_u64);
2975                    let mut recount = |bits: Option<u64>, times: u32| {
2976                        let held = match bits {
2977                            Some(bits) => first.position(bits).map(|at| &mut recounts[at]),
2978                            None => null_count.as_mut(),
2979                        };
2980                        if let Some(count) = held {
2981                            *count = count.saturating_add(u64::from(times));
2982                        }
2983                    };
2984                    let mut run = Run::default();
2985                    self.visit_numeric(column, signed, |_, bits| {
2986                        if let Some((bits, times)) = run.push(bits) {
2987                            recount(bits, times);
2988                        }
2989                    })?;
2990                    if let Some((bits, times)) = run.take() {
2991                        recount(bits, times);
2992                    }
2993                    let exact = first
2994                        .slots
2995                        .iter()
2996                        .zip(&recounts)
2997                        .filter(|(slot, _)| slot.count != 0)
2998                        .map(|(slot, &count)| (slot.bits, count))
2999                        .collect::<FrequencyMap<_>>();
3000                    (exact, null_count)
3001                };
3002                (exact, null_count, decrements, distinct_count)
3003            }
3004        };
3005        let mut entries = exact
3006            .into_iter()
3007            .map(|(bits, count)| FrequencyEntry { value: value_of(Some(bits)), count })
3008            .chain(null_count.map(|count| FrequencyEntry { value: FrequencyValue::Null, count }))
3009            .collect::<Vec<_>>();
3010        let omitted_max = keep_most_frequent(&mut entries).max(decrements);
3011        let kept_rows = entries.iter().try_fold(0_u64, |total, entry| {
3012            total.checked_add(entry.count).filter(|&total| total <= FREQUENCY_ORDINALS as u64)
3013        });
3014        let mut ordinals = Vec::new();
3015        let mut ordinal_entries = Vec::new();
3016        if let Some(kept_rows) = kept_rows {
3017            let mut kept = FrequencyMap::default();
3018            let mut null_kept = None;
3019            for (at, entry) in entries.iter().enumerate() {
3020                let at = u16::try_from(at)
3021                    .map_err(|_| invalid("too many retained frequency entries"))?;
3022                match entry.value {
3023                    FrequencyValue::Integer(value) => {
3024                        kept.insert(value as u64, at);
3025                    }
3026                    FrequencyValue::Null => null_kept = Some(at),
3027                    FrequencyValue::Code(_) => {}
3028                }
3029            }
3030            ordinals.reserve(usize::try_from(kept_rows).unwrap_or(FREQUENCY_ORDINALS));
3031            ordinal_entries.reserve(usize::try_from(kept_rows).unwrap_or(FREQUENCY_ORDINALS));
3032            self.visit_numeric(column, signed, |ordinal, bits| {
3033                let held = match bits {
3034                    Some(bits) => kept.get(&bits).copied(),
3035                    None => null_kept,
3036                };
3037                if let Some(entry) = held {
3038                    ordinals.push(ordinal);
3039                    ordinal_entries.push(entry);
3040                }
3041            })?;
3042        }
3043        Ok((
3044            Some(FrequencySummary { entries, omitted_max, ordinals, ordinal_entries }),
3045            distinct_count,
3046        ))
3047    }
3048
3049    /// Hands every row of an integer column to `visit` as its ordinal and its sixty four bits, or
3050    /// `None` for a null.
3051    ///
3052    /// `signed` says which of the two readings the column has. A packed unsigned column would come
3053    /// back from `signed_block` as a base plus a code in `i64`, which wraps for a value past the top
3054    /// of `BIGINT`, so only a signed column takes the block path.
3055    fn visit_numeric(
3056        &self,
3057        column: usize,
3058        signed: bool,
3059        mut visit: impl FnMut(u64, Option<u64>),
3060    ) -> Result<()> {
3061        let ty = &self.table.fields[column].ty;
3062        let mut start = 0_u64;
3063        let mut block = Vec::new();
3064        for stripe in &self.table.stripes {
3065            let spans = read_index(&self.file, stripe, column)?;
3066            let page = stripe.pages[column];
3067            let mut bytes = vec![0; page.length as usize];
3068            read_at(&self.file, page.offset, &mut bytes)?;
3069            for (span, &rows) in spans.iter().zip(&stripe.parts) {
3070                let part = part_bytes(&bytes, *span)?;
3071                if checksum(part) != span.hash {
3072                    return Err(invalid("column page checksum differs while building frequencies"));
3073                }
3074                let rows = rows as usize;
3075                let vector = decode(ty, rows, part, None)?;
3076                // Every signed layout a numeric column decodes to, which is every column of `hits`,
3077                // comes out as one run of `i64` and is walked as a slice. The row path below is for
3078                // the unsigned types and anything else that cannot be handed over that way.
3079                if signed && vector.signed_block(&mut block) && block.len() == rows {
3080                    if vector.none_null() {
3081                        for (row, &value) in block.iter().enumerate() {
3082                            visit(start.saturating_add(row as u64), Some(value as u64));
3083                        }
3084                    } else {
3085                        for (row, &value) in block.iter().enumerate() {
3086                            let bits = (!vector.is_null_at(row)).then_some(value as u64);
3087                            visit(start.saturating_add(row as u64), bits);
3088                        }
3089                    }
3090                    start = start.saturating_add(rows as u64);
3091                    continue;
3092                }
3093                // row at a time: frequency construction visits decoded values to update bounded candidates.
3094                for row in 0..rows {
3095                    let bits = if vector.is_null_at(row) {
3096                        None
3097                    } else {
3098                        // An unsigned column has no signed reading, and the documented fallback is
3099                        // the value itself. Every width the format stores fits in sixty four bits,
3100                        // so nothing is lost on the way through.
3101                        let widened = match vector.signed_at(row) {
3102                            Some(value) => Some(value as u64),
3103                            None => match vector.value_at(row) {
3104                                Value::UTinyInt(value) => Some(u64::from(value)),
3105                                Value::USmallInt(value) => Some(u64::from(value)),
3106                                Value::UInteger(value) => Some(u64::from(value)),
3107                                Value::UBigInt(value) => Some(value),
3108                                _ => None,
3109                            },
3110                        };
3111                        Some(widened.ok_or_else(|| {
3112                            invalid("numeric frequency page did not contain an integer value")
3113                        })?)
3114                    };
3115                    visit(start.saturating_add(row as u64), bits);
3116                }
3117                start = start.saturating_add(rows as u64);
3118            }
3119        }
3120        Ok(())
3121    }
3122
3123    /// The columns that get numeric frequencies, which are the integer, date and timestamp ones.
3124    fn numeric_columns(&self) -> Vec<usize> {
3125        self.table
3126            .fields
3127            .iter()
3128            .enumerate()
3129            .filter_map(|(column, field)| {
3130                matches!(
3131                    field.ty,
3132                    LogicalType::TinyInt
3133                        | LogicalType::SmallInt
3134                        | LogicalType::Integer
3135                        | LogicalType::BigInt
3136                        | LogicalType::UTinyInt
3137                        | LogicalType::USmallInt
3138                        | LogicalType::UInteger
3139                        | LogicalType::UBigInt
3140                        | LogicalType::Date
3141                        | LogicalType::Timestamp
3142                )
3143                .then_some(column)
3144            })
3145            .collect()
3146    }
3147
3148    /// Reads one stable dictionary code column only at sorted table-wide row ordinals.
3149    #[allow(dead_code)]
3150    fn stable_codes_at(&self, column: usize, ordinals: &[u64]) -> Result<Option<Vec<Option<u32>>>> {
3151        if self.dictionaries.get(column).and_then(Option::as_ref).is_none() {
3152            return Ok(None);
3153        }
3154        if ordinals.windows(2).any(|pair| pair[0] >= pair[1]) {
3155            return Err(invalid("frequency ordinals are not sorted and unique"));
3156        }
3157        let mut out = Vec::with_capacity(ordinals.len());
3158        let mut wanted = 0;
3159        let mut stripe_start = 0_u64;
3160        for stripe in &self.table.stripes {
3161            let stripe_end = stripe_start.saturating_add(stripe.rows as u64);
3162            if wanted == ordinals.len() || ordinals[wanted] >= stripe_end {
3163                stripe_start = stripe_end;
3164                continue;
3165            }
3166            let spans = read_index(&self.file, stripe, column)?;
3167            let page = stripe.pages[column];
3168            let mut bytes = vec![0; page.length as usize];
3169            read_at(&self.file, page.offset, &mut bytes)?;
3170            let mut part_start = stripe_start;
3171            for (span, &rows) in spans.iter().zip(&stripe.parts) {
3172                let part_end = part_start.saturating_add(u64::from(rows));
3173                if wanted < ordinals.len() && ordinals[wanted] < part_end {
3174                    let part = part_bytes(&bytes, *span)?;
3175                    if checksum(part) != span.hash {
3176                        return Err(invalid(
3177                            "column page checksum differs while building pair frequencies",
3178                        ));
3179                    }
3180                    let upto = ordinals.partition_point(|&ordinal| ordinal < part_end);
3181                    let positions = ordinals[wanted..upto]
3182                        .iter()
3183                        .map(|&ordinal| {
3184                            usize::try_from(ordinal.saturating_sub(part_start))
3185                                .map_err(|_| invalid("frequency row offset does not fit in memory"))
3186                        })
3187                        .collect::<Result<Vec<_>>>()?;
3188                    if !decode_selected_stable_codes(rows as usize, part, &positions, &mut out)? {
3189                        return Ok(None);
3190                    }
3191                    wanted = upto;
3192                }
3193                part_start = part_end;
3194            }
3195            stripe_start = stripe_end;
3196        }
3197        if wanted != ordinals.len() {
3198            return Err(invalid("frequency ordinal is outside the table"));
3199        }
3200        Ok(Some(out))
3201    }
3202
3203    /// Derives bounded two-key leaders from numeric anchor ordinals and stable string codes.
3204    #[allow(dead_code)]
3205    fn pair_frequencies(
3206        &self,
3207        frequencies: &[Option<Frequencies>],
3208    ) -> Result<Vec<PairFrequencySummary>> {
3209        let anchors = frequencies
3210            .iter()
3211            .enumerate()
3212            .filter_map(|(column, summary)| {
3213                // A writer holds every synopsis it counted, so there is nothing stored to skip.
3214                match summary {
3215                    Some(Frequencies::Held(summary)) => Some(summary),
3216                    _ => None,
3217                }
3218                .filter(|summary| {
3219                    !summary.ordinals.is_empty()
3220                        && summary.ordinal_entries.len() == summary.ordinals.len()
3221                })
3222                .cloned()
3223                .map(|summary| (column, summary))
3224            })
3225            .collect::<Vec<_>>();
3226        let strings = self
3227            .dictionaries
3228            .iter()
3229            .enumerate()
3230            .filter_map(|(column, dictionary)| dictionary.as_ref().map(|_| column))
3231            .collect::<Vec<_>>();
3232        let mut summaries = Vec::new();
3233        for (first, anchors) in anchors {
3234            for &second in &strings {
3235                if summaries.len() == MAX_PAIR_FREQUENCIES {
3236                    return Ok(summaries);
3237                }
3238                let Some(codes) = self.stable_codes_at(second, &anchors.ordinals)? else {
3239                    continue;
3240                };
3241                if codes.len() != anchors.ordinal_entries.len() {
3242                    return Err(invalid("pair frequency columns have different lengths"));
3243                }
3244                let mut counts = HashMap::<(u16, Option<u32>), u64>::new();
3245                for (&anchor, code) in anchors.ordinal_entries.iter().zip(codes) {
3246                    *counts.entry((anchor, code)).or_default() += 1;
3247                }
3248                let mut entries = counts
3249                    .into_iter()
3250                    .map(|((first_entry, second), count)| PairFrequencyEntry {
3251                        first_entry,
3252                        second,
3253                        count,
3254                    })
3255                    .collect::<Vec<_>>();
3256                entries.sort_unstable_by(|left, right| {
3257                    right
3258                        .count
3259                        .cmp(&left.count)
3260                        .then_with(|| left.first_entry.cmp(&right.first_entry))
3261                        .then_with(|| left.second.cmp(&right.second))
3262                });
3263                let pair_omitted = entries.get(FREQUENCY_ENTRIES).map_or(0, |entry| entry.count);
3264                entries.truncate(FREQUENCY_ENTRIES);
3265                summaries.push(PairFrequencySummary {
3266                    first: u16::try_from(first)
3267                        .map_err(|_| invalid("pair frequency column index overflows"))?,
3268                    second: u16::try_from(second)
3269                        .map_err(|_| invalid("pair frequency column index overflows"))?,
3270                    entries,
3271                    omitted_max: anchors.omitted_max.max(pair_omitted),
3272                });
3273            }
3274        }
3275        Ok(summaries)
3276    }
3277
3278    /// Writes the directory of the table this writer is on and says where it went.
3279    ///
3280    /// Everything [`Writer::finish`] used to do except the two writes that publish. Pulling it out
3281    /// is what lets a second table follow a first: the bytes of a closed table are complete and
3282    /// addressable while nothing yet points at them, and the pointer is the last write of the
3283    /// commit.
3284    ///
3285    /// # Errors
3286    ///
3287    /// If directory encoding or writing fails.
3288    fn close(&mut self) -> Result<Entry> {
3289        self.reclaim()?;
3290        self.flush_pending()?;
3291        // The rest of a table is its statistics, its dictionaries and its directory. The dictionary
3292        // work is charged as its own stage, because ranking a global dictionary can be most of what
3293        // this costs, and the rest as publish.
3294        let profile = self.profile.clone();
3295        let timing = profile.as_deref().map(|profile| profile.span(Stage::Publish));
3296        let before = self.at;
3297        let mut stripes = std::mem::take(&mut self.order)
3298            .into_iter()
3299            .zip(std::mem::take(&mut self.table.stripes))
3300            .collect::<Vec<_>>();
3301        stripes.sort_by_key(|(order, _)| order.0);
3302        let mut previous: Option<(u64, u64)> = None;
3303        for ((first, last), _) in &stripes {
3304            if previous.is_some_and(|previous| previous >= *first) {
3305                return Err(invalid("chunks did not arrive in source order"));
3306            }
3307            previous = Some(*last);
3308        }
3309        self.table.stripes = stripes.into_iter().map(|(_, stripe)| stripe).collect();
3310        drop(timing);
3311        let timing = profile.as_deref().map(|profile| profile.span(Stage::Dictionary));
3312        let placing = self.at;
3313        finish_dictionaries(&mut self.dictionaries)?;
3314        self.place_blocks()?;
3315        for dictionary in self.dictionaries.iter_mut().flatten() {
3316            dictionary.release_lookup();
3317            dictionary.recharge(profile.as_deref());
3318        }
3319        let (numeric, closed) = self.close_columns()?;
3320        let (frequencies, distincts): (Vec<Option<FrequencySummary>>, Vec<_>) =
3321            numeric.into_iter().unzip();
3322        let frequencies =
3323            frequencies.into_iter().map(|held| held.map(Frequencies::Held)).collect::<Vec<_>>();
3324        // Pair leaders are query results, not reusable column statistics.
3325        let pairs = Vec::new();
3326        self.table.frequencies = frequencies;
3327        self.table.distincts = distincts;
3328        self.table.pair_frequencies = pairs;
3329        if let Some(profile) = &profile {
3330            profile.release(self.dictionaries.iter().flatten().map(|held| held.charged).sum());
3331        }
3332        self.table.demoted = self
3333            .dictionaries
3334            .iter()
3335            .map(|dictionary| dictionary.as_ref().is_some_and(|held| held.demoted))
3336            .collect();
3337        if !self.table.demoted.contains(&true) {
3338            self.table.demoted = Vec::new();
3339        }
3340        self.dictionaries = Vec::new();
3341        self.table.dictionary_payloads = vec![0; self.table.fields.len()];
3342        self.table.frequency_texts = vec![Vec::new(); self.table.fields.len()];
3343        self.table.host_groups = None;
3344        for (index, closed) in closed.into_iter().enumerate() {
3345            let Some(closed) = closed else { continue };
3346            let ClosedDictionary { distinct, frequencies, texts, hosts, encoded, payload } = closed;
3347            self.table.distincts[index] = distinct;
3348            self.table.frequencies[index] = frequencies.map(Frequencies::Held);
3349            self.table.frequency_texts[index] = texts;
3350            if hosts.is_some() {
3351                self.table.host_groups = hosts;
3352            }
3353            let offset = self.at;
3354            self.put(&encoded.index)?;
3355            self.put(&encoded.ranks)?;
3356            self.put(&encoded.grams)?;
3357            self.table.dictionary_payloads[index] = payload;
3358            let length = encoded
3359                .index
3360                .len()
3361                .checked_add(encoded.ranks.len())
3362                .and_then(|len| len.checked_add(encoded.grams.len()))
3363                .ok_or_else(|| invalid("dictionary page length overflow"))?;
3364            self.table.dictionaries[index] = Some(Page {
3365                offset,
3366                length: u32::try_from(length)
3367                    .map_err(|_| invalid("dictionary page length overflow"))?,
3368                hash: checksum(&encoded.index),
3369            });
3370        }
3371        drop(timing);
3372        let timing = profile.as_deref().map(|profile| profile.span(Stage::Publish));
3373        let placed = self.at - placing;
3374        self.write_stats()?;
3375        let directory = encode_directory(&self.table)?;
3376        if directory.len() > MAX_DIRECTORY {
3377            return Err(invalid("directory exceeds the configured bound"));
3378        }
3379        let offset = self.at;
3380        self.put(&directory)?;
3381        drop(timing);
3382        if let Some(profile) = &profile {
3383            profile.moved(Stage::Dictionary, 0, placed, 0);
3384            profile.moved(Stage::Publish, 0, self.at - before - placed, 0);
3385        }
3386        Ok(Entry {
3387            name: self.table.name.clone(),
3388            fields: self.table.fields.clone(),
3389            rows: self.table.rows,
3390            nonzero: vec![None; self.table.fields.len()],
3391            aggregates: table_aggregate_sums(&self.table),
3392            distincts: self.table.distincts.clone(),
3393            extremes: table_integer_extremes(&self.table),
3394            frequencies: table_complete_numeric_frequencies(&self.table),
3395            directory: Page {
3396                offset,
3397                length: u32::try_from(directory.len())
3398                    .map_err(|_| invalid("directory length overflow"))?,
3399                hash: checksum(&directory),
3400            },
3401        })
3402    }
3403
3404    /// Every numeric column's frequencies and every global dictionary's page and statistics, by
3405    /// column, as many columns at a time as [`CLOSE_BYTES`] allows.
3406    ///
3407    /// The two kinds read what is already written and write nothing, so they share one set of
3408    /// threads. Each was most of a second on `hits` with the other waiting for it, and neither keeps
3409    /// every core busy on its own. The most expensive column that fits is the one taken next, so
3410    /// the long ones start first and the short ones fill in behind them. A column that does not fit
3411    /// waits for one that is closing to finish, unless nothing is closing, in which case it goes
3412    /// alone.
3413    ///
3414    /// A numeric column is charged the exact distinct set its sketch says it will need, and one the
3415    /// sketch puts far past what that set can hold does not build it, because the set would fill,
3416    /// give up and have held 512 MiB for nothing. A column with no sketch is charged the whole set.
3417    /// Each job charges itself as its own span, publish for the numeric ones and dictionary for the
3418    /// rest, because it runs on a thread of its own and a span on this one would see the wall time
3419    /// and none of the CPU.
3420    #[allow(clippy::type_complexity)]
3421    fn close_columns(
3422        &self,
3423    ) -> Result<(Vec<(Option<FrequencySummary>, Option<u64>)>, Vec<Option<ClosedDictionary>>)> {
3424        let numeric = self.numeric_columns().into_iter().map(|column| {
3425            let estimate =
3426                self.gathers.get(column).and_then(Option::as_ref).and_then(stats::Gather::distinct);
3427            let counted = !estimate.is_some_and(distinct::beyond);
3428            let set =
3429                if counted { distinct::bytes_for(estimate.unwrap_or(f64::INFINITY)) } else { 0 };
3430            let cost = self.table.rows.saturating_mul(weight(&self.table.fields[column].ty));
3431            (Closing::Numeric { column, counted }, NUMERIC_CLOSE_BYTES + set, cost)
3432        });
3433        let dictionaries =
3434            self.dictionaries.iter().enumerate().filter_map(|(index, dictionary)| {
3435                let dictionary = dictionary.as_ref()?;
3436                let bytes = dictionary.closing_bytes();
3437                Some((Closing::Dictionary { index, dictionary }, bytes, bytes))
3438            });
3439        let mut jobs = numeric.chain(dictionaries).collect::<Vec<_>>();
3440        jobs.sort_by_key(|&(_, _, cost)| cost);
3441        let columns = self.table.fields.len();
3442        let mut frequencies = vec![(None, None); columns];
3443        let mut closed = (0..columns).map(|_| None).collect::<Vec<_>>();
3444        let profile = self.profile.as_deref();
3445        let run = |job: Closing<'_>, bytes: usize| -> Result<Closed> {
3446            let _holding = profile.map(|profile| profile.holding(bytes as u64));
3447            match job {
3448                Closing::Numeric { column, counted } => {
3449                    let _timing = profile.map(|profile| profile.span(Stage::Publish));
3450                    Ok(Closed::Numeric(column, self.numeric_frequency(column, counted)?))
3451                }
3452                Closing::Dictionary { index, dictionary } => {
3453                    let _timing = profile.map(|profile| profile.span(Stage::Dictionary));
3454                    Ok(Closed::Dictionary(index, self.close_dictionary(index, dictionary)?))
3455                }
3456            }
3457        };
3458        let workers = close_workers().min(jobs.len());
3459        let pieces = if workers <= 1 {
3460            jobs.into_iter().map(|(job, bytes, _)| run(job, bytes)).collect::<Result<Vec<_>>>()?
3461        } else {
3462            // The columns not taken yet, cheapest first, and the bytes the ones closing now hold.
3463            let state = Mutex::new((jobs, 0_usize));
3464            let finished = Condvar::new();
3465            std::thread::scope(|scope| {
3466                (0..workers)
3467                    .map(|_| {
3468                        scope.spawn(|| {
3469                            let mut mine = Vec::new();
3470                            loop {
3471                                let mut held = state.lock().map_err(|_| {
3472                                    Error::internal("a native close worker panicked")
3473                                })?;
3474                                let (job, bytes) = loop {
3475                                    let (jobs, busy) = &mut *held;
3476                                    if jobs.is_empty() {
3477                                        return Ok(mine);
3478                                    }
3479                                    let fits = jobs.iter().rposition(|&(_, bytes, _)| {
3480                                        *busy == 0 || busy.saturating_add(bytes) <= CLOSE_BYTES
3481                                    });
3482                                    if let Some(at) = fits {
3483                                        let (job, bytes, _) = jobs.remove(at);
3484                                        *busy += bytes;
3485                                        break (job, bytes);
3486                                    }
3487                                    held = finished.wait(held).map_err(|_| {
3488                                        Error::internal("a native close worker panicked")
3489                                    })?;
3490                                };
3491                                drop(held);
3492                                // Given back on the way out whether the close worked, failed or
3493                                // panicked, so that a worker waiting for room is never left waiting.
3494                                let _room = Room { state: &state, finished: &finished, bytes };
3495                                mine.push(run(job, bytes)?);
3496                            }
3497                        })
3498                    })
3499                    .collect::<Vec<_>>()
3500                    .into_iter()
3501                    .map(|handle| {
3502                        handle
3503                            .join()
3504                            .map_err(|_| Error::internal("a native close worker panicked"))?
3505                    })
3506                    .collect::<Result<Vec<_>>>()
3507            })?
3508            .into_iter()
3509            .flatten()
3510            .collect()
3511        };
3512        for piece in pieces {
3513            match piece {
3514                Closed::Numeric(column, summary) => frequencies[column] = summary,
3515                Closed::Dictionary(index, one) => closed[index] = Some(one),
3516            }
3517        }
3518        Ok((frequencies, closed))
3519    }
3520
3521    /// One global dictionary's page and statistics, built from what is already in the file.
3522    ///
3523    /// Nothing is written here, so that [`Self::close`] can run this beside the numeric frequencies
3524    /// and put the pages down afterwards in column order, which is where they always went. The
3525    /// column's values are decoded in here and dropped before it returns, and
3526    /// [`Self::close_columns`] decides how many columns are in here at once.
3527    fn close_dictionary(
3528        &self,
3529        _index: usize,
3530        dictionary: &GlobalDictionary,
3531    ) -> Result<ClosedDictionary> {
3532        let (order, flat, bases) = dictionary.ranked_with_values(Some(&*self.file))?;
3533        // A code nothing counted is a code no non-null row of this column holds, which is the
3534        // empty string a null was written as and nothing else, because a code is only ever made by
3535        // a row asking for one. A demoted dictionary counted the stripes before its demotion and
3536        // none after, so it has no count or frequency of the column to give.
3537        let (distinct, frequencies, texts) = if dictionary.demoted {
3538            (None, None, Vec::new())
3539        } else {
3540            let distinct = dictionary.counts.iter().filter(|count| **count != 0).count() as u64;
3541            let (frequencies, texts) = code_frequency(dictionary, &flat, &bases)?;
3542            (Some(distinct), Some(frequencies), texts)
3543        };
3544        // Deriving a fixed SQL host expression at load time materializes its answer.
3545        let hosts = None;
3546        drop(flat);
3547        drop(bases);
3548        let encoded = encode_global_dictionary(dictionary, &order, &dictionary.placed, true)?;
3549        let payload = dictionary
3550            .placed
3551            .iter()
3552            .try_fold(0_u64, |sum, place| sum.checked_add(place.length))
3553            .ok_or_else(|| invalid("global dictionary payload overflow"))?;
3554        Ok(ClosedDictionary { distinct, frequencies, texts, hosts, encoded, payload })
3555    }
3556
3557    /// Writes the statistics sections for the table being closed, as far as the budget reaches.
3558    ///
3559    /// Called from [`Self::close`] after the last stripe and after the dictionaries, which is the
3560    /// first moment the table's column bytes are final and the last moment before the directory is
3561    /// encoded. Both halves matter: the budget is a share of the column bytes, and a section that
3562    /// went in after the directory would be a section the directory does not name.
3563    ///
3564    /// Nothing here can fail the write. A column whose gather came back blind gets no sections, a
3565    /// column the budget could not reach gets none, and section 3.1 says both of those plan the way
3566    /// they planned before statistics existed. The two errors that are returned are an encode
3567    /// failure and a section count past the bound, and neither is a thing a column can cause.
3568    fn write_stats(&mut self) -> Result<()> {
3569        let gathers = std::mem::take(&mut self.gathers);
3570        let rows = self.table.rows as u64;
3571        let mut payloads = Vec::new();
3572        for (column, gather) in gathers.into_iter().enumerate() {
3573            let Some(gather) = gather else { continue };
3574            // A gather that saw a different number of rows than the table committed is a gather
3575            // that missed some, and a distinct count over some of a column is the one error an
3576            // estimator cannot see coming. This has no way of happening today, since a table is
3577            // written once and every chunk goes through `flush_pending`, and that is exactly why it
3578            // is worth a line: it stays true only while that stays true.
3579            if gather.rows() != rows {
3580                continue;
3581            }
3582            let Some(stats) = gather.finish() else { continue };
3583            let mut summary = Vec::new();
3584            stats.summary.encode(&mut summary)?;
3585            let mut sketches = Vec::new();
3586            stats.sketches.encode(&mut sketches)?;
3587            payloads.push((column, summary, sketches));
3588        }
3589        if payloads.is_empty() {
3590            return Ok(());
3591        }
3592        let costs = payloads
3593            .iter()
3594            .map(|(_, summary, sketches)| summary.len() + sketches.len())
3595            .collect::<Vec<_>>();
3596        let allowance = stats::allowance(stats::column_bytes(&self.table), stats::BUDGET_SHARE);
3597        // Nothing is spent yet. A table this writer is closing is one it wrote from nothing, so the
3598        // only statistics sections it can have are the ones about to go in.
3599        let keep = stats::within(&costs, allowance, 0);
3600        for ((column, summary, sketches), _) in
3601            payloads.iter().zip(&keep).filter(|&(_, &keep)| keep)
3602        {
3603            let id = u64::try_from(*column).map_err(|_| invalid("column index overflow"))?;
3604            for (kind, bytes, header_bytes) in [
3605                // A summary is a header the whole way down: there is nothing behind it a reader
3606                // could decide not to read.
3607                (*section::SUMMARY, summary, summary.len() as u32),
3608                (*section::SKETCHES, sketches, rudb_stats::sketches::HEADER_BYTES),
3609            ] {
3610                let written = write_section(
3611                    &*self.file,
3612                    &mut self.at,
3613                    &section::Attachment { kind, id, flags: 0, header_bytes, bytes },
3614                    self.generation,
3615                )?;
3616                self.table.sections.push(written);
3617            }
3618        }
3619        if self.table.sections.len() > MAX_SECTIONS {
3620            return Err(invalid("the table would name more sections than the bound allows"));
3621        }
3622        Ok(())
3623    }
3624
3625    /// Commits every table this writer has written and syncs the file before publishing its header
3626    /// slot.
3627    ///
3628    /// The table handed back is the one the writer was on, which is the last of them. Callers that
3629    /// wrote several already know the others, since they named them.
3630    ///
3631    /// # Errors
3632    ///
3633    /// If directory encoding, writing, or syncing fails.
3634    pub fn finish(mut self) -> Result<Table> {
3635        let entry = self.close()?;
3636        let profile = self.profile.take();
3637        let _timing = profile.as_deref().map(|profile| profile.span(Stage::Publish));
3638        let mut tables = std::mem::take(&mut self.closed);
3639        tables.push(entry);
3640        let catalog = encode_catalog(&tables, &self.views)?;
3641        if catalog.len() > MAX_DIRECTORY {
3642            return Err(invalid("catalog exceeds the configured bound"));
3643        }
3644        let offset = self.at;
3645        self.put(&catalog)?;
3646        if let Some(profile) = &profile {
3647            profile.moved(Stage::Publish, 0, catalog.len() as u64, 0);
3648        }
3649        // Every page and every table directory is on the disk before anything points at them. The
3650        // slot write below is what makes this generation the one a reader picks, so the order of
3651        // these two syncs is the whole of the commit.
3652        synced(&*self.file, profile.as_deref())?;
3653        let slot = Slot {
3654            offset,
3655            length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
3656            generation: self.generation,
3657            hash: checksum(&catalog),
3658        };
3659        // The one write that is not an append, and the last one. It goes back over the slot in the
3660        // header, so it names its offset rather than going through `put`, and `at` does not move.
3661        // Which of the two slots it is alternates with the generation, so the one naming the
3662        // generation before this is still intact and still valid until this write lands.
3663        self.file.write_at(slot_offset(self.generation), &slot.bytes())?;
3664        synced(&*self.file, profile.as_deref())?;
3665        Ok(self.table)
3666    }
3667
3668    /// Commits a generation that changes the views and leaves every table exactly where it is.
3669    ///
3670    /// There was no way to do this before views existed, because everything that could change the
3671    /// catalog also wrote a table, so the only way to say something new about a file was to go
3672    /// through a table. A view is the first thing that can change on its own. Without this, adding
3673    /// a view to a database with eight tables in it would rewrite all eight, since the append path
3674    /// needs a table to append and the fallback is the whole file.
3675    ///
3676    /// It is the same commit as [`Writer::finish`] with nothing appended before it. The table
3677    /// entries are carried forward by directory pointer the way an append carries them, the new
3678    /// catalog goes on the end, and the slot write at the end is what publishes it.
3679    ///
3680    /// # Errors
3681    ///
3682    /// If the file has no valid committed directory, is not this build's format, or cannot be
3683    /// written.
3684    pub fn restate(path: impl AsRef<Path>, views: &[ViewEntry]) -> Result<()> {
3685        let file = RealFilesystem::new().open(path.as_ref(), OpenMode::ReadWrite)?;
3686        let size = file.len()?;
3687        let (slot, bytes, _) = committed_slot(&*file, size)?;
3688        let (closed, _) = decode_catalog(&bytes, size)?;
3689        let generation = slot
3690            .generation
3691            .checked_add(1)
3692            .ok_or_else(|| invalid("native file generation overflow"))?;
3693        let catalog = encode_catalog(&closed, views)?;
3694        if catalog.len() > MAX_DIRECTORY {
3695            return Err(invalid("catalog exceeds the configured bound"));
3696        }
3697        file.write_at(size, &catalog)?;
3698        file.sync()?;
3699        let slot = Slot {
3700            offset: size,
3701            length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
3702            generation,
3703            hash: checksum(&catalog),
3704        };
3705        file.write_at(slot_offset(generation), &slot.bytes())?;
3706        file.sync()?;
3707        Ok(())
3708    }
3709
3710    /// Adds exact count, sum, distinct, bound, and bounded frequency certificates to an older file without
3711    /// rewriting table pages. The old slot remains readable until the new catalog is synced.
3712    pub fn certify_summaries(path: impl AsRef<Path>) -> Result<()> {
3713        let path = path.as_ref();
3714        let (_, size, slot, bytes, _) = slot_bytes(path)?;
3715        let (mut entries, views) = decode_catalog(&bytes, size)?;
3716        let native = Catalog::open(path)?;
3717        for entry in &mut entries {
3718            let reader = native.table(&entry.name)?;
3719            entry.nonzero.fill(None);
3720            entry.aggregates = reader_aggregate_sums(&reader)?;
3721            entry.distincts = (0..entry.fields.len())
3722                .map(|column| reader.distinct_values(column))
3723                .collect::<Result<Vec<_>>>()?;
3724            entry.extremes = reader_integer_extremes(&reader)?;
3725            entry.frequencies = reader_complete_numeric_frequencies(&reader)?;
3726        }
3727        let generation = slot
3728            .generation
3729            .checked_add(1)
3730            .ok_or_else(|| invalid("native file generation overflow"))?;
3731        let catalog = encode_catalog(&entries, &views)?;
3732        if catalog.len() > MAX_DIRECTORY {
3733            return Err(invalid("catalog exceeds the configured bound"));
3734        }
3735        let file = RealFilesystem::new().open(path, OpenMode::ReadWrite)?;
3736        file.write_at(size, &catalog)?;
3737        file.sync()?;
3738        let slot = Slot {
3739            offset: size,
3740            length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
3741            generation,
3742            hash: checksum(&catalog),
3743        };
3744        file.write_at(slot_offset(generation), &slot.bytes())?;
3745        file.sync()?;
3746        Ok(())
3747    }
3748
3749    /// The earlier name for [`Self::certify_summaries`].
3750    pub fn certify_counts(path: impl AsRef<Path>) -> Result<()> {
3751        Self::certify_summaries(path)
3752    }
3753}
3754
3755/// Appends one run of bytes at `at` and moves it past them, answering where they went.
3756///
3757/// The append half of [`attach`], which cannot use [`Writer::put`] because it is not writing a
3758/// table. Every byte a section costs goes through here, so the offsets in an extent table come
3759/// from one place.
3760fn append(file: &dyn rudb_io::File, at: &mut u64, bytes: &[u8]) -> Result<u64> {
3761    let offset = *at;
3762    file.write_at(offset, bytes)?;
3763    *at =
3764        at.checked_add(bytes.len() as u64).ok_or_else(|| invalid("native file length overflow"))?;
3765    Ok(offset)
3766}
3767
3768/// Writes one attachment's payload as extents and returns the entry that names it.
3769///
3770/// The split is by bytes, and the `first` of each extent is therefore a byte count. A section kind
3771/// whose extents should break on a row boundary instead will want to hand its extents over already
3772/// split; nothing needs that yet, and guessing at the shape of it now would be guessing.
3773fn write_section(
3774    file: &dyn rudb_io::File,
3775    at: &mut u64,
3776    one: &section::Attachment<'_>,
3777    generation: u64,
3778) -> Result<Section> {
3779    // A payload of nothing is the exception, and it is not a special case so much as a different
3780    // reading of the same field: an entry with no bytes has no header to be longer than them, and
3781    // `header_bytes` is what the structure would have cost. See `Section::refused`.
3782    if !one.bytes.is_empty() && one.header_bytes as usize > one.bytes.len() {
3783        return Err(invalid("a section's header is longer than its payload"));
3784    }
3785    let mut extents = Vec::new();
3786    let mut first = 0_u64;
3787    for chunk in one.bytes.chunks(section::MAX_EXTENT as usize) {
3788        let offset = append(file, at, chunk)?;
3789        extents.push(section::Extent {
3790            offset,
3791            length: u32::try_from(chunk.len()).map_err(|_| invalid("extent length overflow"))?,
3792            hash: checksum(chunk),
3793            first,
3794        });
3795        first += chunk.len() as u64;
3796    }
3797    let mut table = Vec::with_capacity(extents.len() * section::EXTENT_BYTES);
3798    section::encode_extents(&extents, &mut table)?;
3799    // A payload of nothing is a section of no extents and no extent table, and its `extent_page`
3800    // is zero rather than the end of the file. Section 3.7 wants that entry to exist: it is how a
3801    // relationship that did not fit the budget is recorded as not built rather than forgotten.
3802    let extent_page = if table.is_empty() { 0 } else { append(file, at, &table)? };
3803    Ok(Section {
3804        kind: one.kind,
3805        id: one.id,
3806        generation,
3807        extents: u32::try_from(extents.len()).map_err(|_| invalid("too many extents"))?,
3808        extent_page,
3809        extent_bytes: u32::try_from(table.len()).map_err(|_| invalid("extent table overflow"))?,
3810        hash: checksum(&table),
3811        flags: one.flags,
3812        header_bytes: one.header_bytes,
3813    })
3814}
3815
3816/// Attaches graph sections to a table already committed in a file, without rewriting a page.
3817///
3818/// This is the second pass spec/graph/03-the-file-format.md section 3.8 asks for. A key map has to
3819/// exist before the link that uses it can be built, and it is built by reading the key column back,
3820/// so the structures of a table cannot be written during the load that wrote the table. They are
3821/// written afterwards, by this, and the file in between the two is a correct file that answers
3822/// every query more slowly.
3823///
3824/// Nothing is overwritten. The payloads, the extent tables, the new directory for this table and
3825/// the new catalog all go on the end of the file past the committed generation, and the last write
3826/// is the header slot, exactly as [`Writer::finish`] does it. So a crash anywhere in here leaves
3827/// the generation before it intact, and leaves unreferenced trailing bytes that the next commit
3828/// writes past.
3829///
3830/// An attachment replaces any section of the same kind and id, and every other section is carried
3831/// through untouched, including one whose kind this build does not know. The table's own generation
3832/// is carried through too, because attaching a section moves no row: see [`Table::generation`].
3833///
3834/// # Errors
3835///
3836/// If the file has no valid committed directory, is an older format than this build writes, holds
3837/// no table of that name, names a section whose payload cannot be written, or would end up naming
3838/// more sections than the format allows.
3839pub fn attach(
3840    path: impl AsRef<Path>,
3841    table: &str,
3842    attachments: &[section::Attachment<'_>],
3843) -> Result<Table> {
3844    let file = RealFilesystem::new().open(path.as_ref(), OpenMode::ReadWrite)?;
3845    let file = &*file;
3846    let size = file.len()?;
3847    let (slot, bytes, _) = committed_slot(file, size)?;
3848    let (mut entries, views) = decode_catalog(&bytes, size)?;
3849    let at = entries
3850        .iter()
3851        .position(|entry| entry.name == table)
3852        .ok_or_else(|| invalid(&format!("the file holds no table called {table}")))?;
3853    let mut version = [0; 4];
3854    read_at(file, 8, &mut version)?;
3855    let version = u32::from_le_bytes(version);
3856    // Readable is not the same as writable. A format 22 file has no section table, and giving its
3857    // directory one without moving the number in its header would leave a file that claims to be
3858    // format 22 and is not, which is worse than refusing. Rewriting it with this build is the
3859    // answer, and the format is at 0.3.x, so nobody has one of these that this project did not
3860    // just make.
3861    if version != FORMAT {
3862        return Err(invalid(&format!(
3863            "the file is format {version} and a graph section needs format {FORMAT}, so it has \
3864             to be written again"
3865        )));
3866    }
3867    let mut directory = vec![0; entries[at].directory.length as usize];
3868    read_at(file, entries[at].directory.offset, &mut directory)?;
3869    if checksum(&directory) != entries[at].directory.hash {
3870        return Err(invalid(&format!("the directory of table {table} does not checksum")));
3871    }
3872    let mut held = decode_directory(&directory, size)?;
3873    let mut cursor = size;
3874    for one in attachments {
3875        let written = write_section(file, &mut cursor, one, held.generation)?;
3876        held.sections.retain(|old| !(old.kind == one.kind && old.id == one.id));
3877        held.sections.push(written);
3878    }
3879    if held.sections.len() > MAX_SECTIONS {
3880        return Err(invalid("the table would name more sections than the bound allows"));
3881    }
3882    let encoded = encode_directory(&held)?;
3883    if encoded.len() > MAX_DIRECTORY {
3884        return Err(invalid("directory exceeds the configured bound"));
3885    }
3886    let offset = append(file, &mut cursor, &encoded)?;
3887    entries[at].directory = Page {
3888        offset,
3889        length: u32::try_from(encoded.len()).map_err(|_| invalid("directory length overflow"))?,
3890        hash: checksum(&encoded),
3891    };
3892    // The views the file already had, written back unchanged. Attaching a section to a table says
3893    // nothing about a view and must not drop one.
3894    let catalog = encode_catalog(&entries, &views)?;
3895    if catalog.len() > MAX_DIRECTORY {
3896        return Err(invalid("catalog exceeds the configured bound"));
3897    }
3898    let offset = append(file, &mut cursor, &catalog)?;
3899    file.sync()?;
3900    let generation =
3901        slot.generation.checked_add(1).ok_or_else(|| invalid("native file generation overflow"))?;
3902    let committed = Slot {
3903        offset,
3904        length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
3905        generation,
3906        hash: checksum(&catalog),
3907    };
3908    file.write_at(slot_offset(generation), &committed.bytes())?;
3909    file.sync()?;
3910    Ok(held)
3911}
3912
3913/// One column's frequency synopsis as values with their row counts, shared by every clone of a
3914/// reader.
3915type Synopsis = Arc<Vec<(Value, u64)>>;
3916
3917/// Reads committed native column pages without holding the table in memory.
3918#[derive(Debug, Clone)]
3919pub struct Reader {
3920    file: Arc<File>,
3921    table: Arc<Table>,
3922    dictionaries: Arc<Vec<OnceLock<Arc<Vector>>>>,
3923    /// Held while a global dictionary is being opened, one per column.
3924    ///
3925    /// The [`OnceLock`] above says whether one has been opened, which is the question a reader that
3926    /// already has it needs answered and is free. It does not say whether one is being opened, and
3927    /// the difference matters because every worker of a scan wants the same dictionary at the same
3928    /// moment. Without this they all miss, all read the page, all verify it and all decode it, and
3929    /// all but one throw the answer away. ClickBench 38 reads the URL dictionary, which is 515,958
3930    /// entries, and was paying for it twice.
3931    loading: Arc<Vec<Mutex<()>>>,
3932    /// Each column's frequency synopsis as values, the first time anything asks for it. See
3933    /// [`Reader::decode_frequencies`].
3934    frequency_values: Arc<Vec<OnceLock<Synopsis>>>,
3935    /// Stored frequency sections are decoded once per open table. A small directory can hold the
3936    /// summary inline, but a larger one otherwise rereads and decodes the same section on every
3937    /// plan and every summary-backed aggregate.
3938    frequency_summaries: Arc<Vec<OnceLock<Arc<FrequencySummary>>>>,
3939    /// How many global dictionaries have been opened. A scan of a dictionary column should open its
3940    /// dictionary once however many workers it has, and the test that says so is the only thing
3941    /// keeping it that way.
3942    opened: Arc<AtomicUsize>,
3943    /// The membership sieves of one stripe of one column, by column and then by stripe, read the
3944    /// first time a probe asks about them. A query filters on one or two columns and never looks at
3945    /// the rest, so reading these at open would be the whole index for the sake of a fraction of it.
3946    sieves: Arc<Vec<Vec<SieveSlot>>>,
3947    /// The per part ranges of one stripe of one column, by column and then by stripe, read the
3948    /// first time something compares that column and kept after that.
3949    part_ranges: Arc<Vec<Vec<RangeSlot>>>,
3950    /// Which stripe and which part of it every part of the table is, by table wide part number.
3951    places: Arc<Vec<Place>>,
3952    cache: Arc<Shelf>,
3953    /// Where the pages above are counted against the database's budget. See [`PagePool`].
3954    pool: PagePool,
3955    /// How many whole stripe pages have been read, which is what the sharing above is judged on. A
3956    /// scan of a column should read each of its stripes once however many workers it has.
3957    pages: Arc<AtomicUsize>,
3958    /// How many index sections have been read. A scan of a column should read each of its stripes
3959    /// once here too, and the test that says so is the only thing keeping it that way.
3960    indexes: Arc<AtomicUsize>,
3961    /// The file's size when it was opened, for [`Reader::layout`].
3962    size: u64,
3963    /// The committed directory's size, for [`Reader::layout`].
3964    directory: u64,
3965    /// What opening the file cost, which is a number rather than a claim.
3966    opening: Opening,
3967}
3968
3969/// What [`Reader::open`] read before it returned.
3970///
3971/// `spec/stats/04-in-memory.md` section 4.2 says opening a table reads the header and the directory
3972/// and nothing else, and once that document's statistics are in the file the tempting change is to
3973/// load a column summary or two on the way past, because they are small and the next query will
3974/// want them. A hundred milliseconds of that is a hundred milliseconds nobody asked for, and an
3975/// embedded database is opened by processes that are about to run one trivial query.
3976///
3977/// So the claim gets a number. Both of these are fixed by the schema and the stripe count and are
3978/// independent of how many rows the file holds, and the test that says so is what stops the
3979/// tempting change from landing quietly.
3980#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
3981pub struct Opening {
3982    /// How many times the file was read. The header, then each directory slot that looked valid
3983    /// enough to check, so three at the most.
3984    pub reads: u32,
3985    /// How many bytes those reads asked for.
3986    pub bytes: u64,
3987}
3988
3989/// What a reader has read, while it was being opened and since.
3990#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
3991pub struct Reads {
3992    /// What opening cost, before any query had been planned.
3993    pub opening: Opening,
3994    /// Whole stripe pages read since.
3995    pub pages: usize,
3996    /// Index sections read since.
3997    pub indexes: usize,
3998    /// Global dictionaries opened since. One per dictionary column that a query touched, however
3999    /// many workers touched it, which is a claim only a test can keep true.
4000    pub dictionaries: usize,
4001}
4002
4003/// Where one table wide part number lands.
4004#[derive(Debug, Clone, Copy)]
4005struct Place {
4006    stripe: u32,
4007    part: u32,
4008    rows: u32,
4009}
4010
4011/// One part's bytes inside one column page.
4012#[derive(Debug, Clone, Copy)]
4013struct PartSpan {
4014    start: usize,
4015    length: usize,
4016    hash: u64,
4017}
4018
4019/// What a reader holds for one stripe of one column.
4020///
4021/// The index is small and is loaded whether the caller wants the whole page or one part of it. The
4022/// page is loaded only by a scan, because a sparse fetch that wants a thousand rows out of sixty
4023/// four thousand would be reading sixty four times what it uses.
4024#[derive(Debug, Clone)]
4025struct CachedColumn {
4026    stripe: usize,
4027    index: Arc<Vec<PartSpan>>,
4028    page: Option<Arc<Vec<u8>>>,
4029}
4030
4031/// One column's stripes a reader holds, and which of them somebody is reading right now.
4032///
4033/// The pages are one slot per stripe of the table rather than a list of the ones being kept, so
4034/// finding a page is an index and not a walk. That matters because the walk happened under the
4035/// lock, once per part per column, and a scan that gives a whole stripe to each of thirty two
4036/// workers keeps enough pages that walking them was the longest thing the lock was held for. The
4037/// slots cost a pointer per stripe per column, which on the ClickBench file is eight kilobytes
4038/// against the forty megabytes of pages they point at. `order` is which of them are filled, oldest
4039/// first, because that is the one thing the slots cannot say by themselves.
4040///
4041/// `loading` is what keeps a scan from reading the same page once per worker. It is a list and not
4042/// a set because it holds at most one stripe per worker on the column and is walked far less often
4043/// than a hash of it would be built.
4044///
4045/// `index` is every index this reader has ever read for the column, one slot per stripe, and it is
4046/// never evicted. An index is a few hundred bytes and a page is a quarter of a megabyte, so the two
4047/// do not belong under the same budget. Riding in the page cache meant a worker that came back to a
4048/// stripe after its page had been evicted read the index again with it, which on the full
4049/// ClickBench file was about thirteen hundred reads out of a hundred and fourteen thousand.
4050#[derive(Debug, Default)]
4051struct Cached {
4052    pages: Vec<Option<Resident>>,
4053    loading: Vec<usize>,
4054    index: Vec<Option<Arc<Vec<PartSpan>>>>,
4055}
4056
4057/// One page a reader holds, and whether anyone has read it since the pool last looked.
4058#[derive(Debug, Clone)]
4059struct Resident {
4060    page: Arc<Vec<u8>>,
4061    used: Arc<AtomicBool>,
4062}
4063
4064/// Every column's pages of one reader, with how many each column holds and the floor under that.
4065#[derive(Debug)]
4066struct Shelf {
4067    columns: Vec<Mutex<Cached>>,
4068    /// How many pages each column holds right now. Counted outside the column locks so that the
4069    /// pool can tell whether a column is at its floor without taking a lock it might be under.
4070    held: Vec<AtomicUsize>,
4071    /// How many stripes of one column are kept whatever the budget says. See
4072    /// [`CACHED_STRIPES_PER_COLUMN`] for what sets it and [`Reader::keep_stripes`] for who raises it.
4073    kept: AtomicUsize,
4074}
4075
4076/// The pages every reader of one database keeps, under one budget in bytes.
4077///
4078/// A reader lives as long as the database does, so the pages it holds are what the next query finds
4079/// already in memory. They used to be four stripes a column, oldest out first, which on TPC-H SF1
4080/// meant every query read every page of lineitem off the file again and paid the system call for
4081/// it. Keeping every page there costs 38 MB and took a third of the system time off the suite.
4082///
4083/// So the question is no longer how many stripes a column keeps but how many bytes the database
4084/// does, and one budget answers it for every reader at once. A table nobody queries gives its pages
4085/// up to one that is being queried, which a count per column cannot do.
4086///
4087/// Pages leave by the clock. Each has a bit a read sets, and when the pool is over budget it walks
4088/// from the oldest: a page with the bit set loses the bit and goes round again, and a page without
4089/// it goes. That keeps what is read over and over and lets a page one scan read once go first.
4090///
4091/// The old count is still a floor. A column never gives up a page while it holds four or fewer,
4092/// because a scan whose workers evict each other's pages reads a quarter of a megabyte for every
4093/// part it takes, and a budget of zero is the cache as it was before the pool existed.
4094#[derive(Debug, Clone, Default)]
4095pub struct PagePool {
4096    ring: Arc<Mutex<Ring>>,
4097    budget: Arc<AtomicUsize>,
4098}
4099
4100#[derive(Debug, Default)]
4101struct Ring {
4102    held: VecDeque<Held>,
4103    bytes: usize,
4104}
4105
4106/// One page in the pool, pointing back at the reader that holds it.
4107///
4108/// Weak, because a reader that has gone, which every reader does at a checkpoint, should take its
4109/// pages with it and not have them kept alive by the pool.
4110#[derive(Debug)]
4111struct Held {
4112    shelf: Weak<Shelf>,
4113    column: usize,
4114    stripe: usize,
4115    bytes: usize,
4116    used: Arc<AtomicBool>,
4117}
4118
4119impl PagePool {
4120    /// A pool that keeps up to `budget` bytes of pages beyond each column's floor.
4121    #[must_use]
4122    pub fn new(budget: usize) -> Self {
4123        let pool = Self::default();
4124        pool.budget.store(budget, Atomic::Relaxed);
4125        pool
4126    }
4127
4128    /// The bytes of pages the pool is counting now.
4129    ///
4130    /// # Panics
4131    ///
4132    /// If the pool's lock is poisoned, which takes a panic while it was held.
4133    #[must_use]
4134    pub fn bytes(&self) -> usize {
4135        self.ring.lock().map_or(0, |ring| ring.bytes)
4136    }
4137
4138    /// Counts a page a reader has just taken in, and lets pages go until the pool is back under its
4139    /// budget or it has looked at every page once.
4140    ///
4141    /// Called with no column lock held. The pages that go are chosen under the pool's lock and
4142    /// dropped under their column's lock afterwards, so no thread ever holds both.
4143    fn admit(&self, held: Held) {
4144        let budget = self.budget.load(Atomic::Relaxed);
4145        let mut gone = Vec::new();
4146        {
4147            let Ok(mut ring) = self.ring.lock() else { return };
4148            ring.bytes += held.bytes;
4149            ring.held.push_back(held);
4150            // One lap and no more. A page read since the last pass loses its bit on this one and
4151            // can only go on a later one, which is the second chance the clock is named for.
4152            let mut looked = 0;
4153            let limit = ring.held.len();
4154            while ring.bytes > budget && looked < limit {
4155                looked += 1;
4156                let Some(entry) = ring.held.pop_front() else { break };
4157                let Some(shelf) = entry.shelf.upgrade() else {
4158                    ring.bytes -= entry.bytes;
4159                    continue;
4160                };
4161                if entry.used.swap(false, Atomic::Relaxed) {
4162                    ring.held.push_back(entry);
4163                    continue;
4164                }
4165                let count = &shelf.held[entry.column];
4166                if count.load(Atomic::Relaxed) <= shelf.kept.load(Atomic::Relaxed).max(1) {
4167                    ring.held.push_back(entry);
4168                    continue;
4169                }
4170                count.fetch_sub(1, Atomic::Relaxed);
4171                ring.bytes -= entry.bytes;
4172                gone.push((shelf, entry));
4173            }
4174            // A reader that has gone leaves its entries behind, and with a budget nobody reaches
4175            // they would pile up one checkpoint after another. The front is where the oldest are.
4176            while ring.held.front().is_some_and(|entry| entry.shelf.strong_count() == 0) {
4177                if let Some(entry) = ring.held.pop_front() {
4178                    ring.bytes -= entry.bytes;
4179                }
4180            }
4181        }
4182        for (shelf, entry) in gone {
4183            let Ok(mut cached) = shelf.columns[entry.column].lock() else { continue };
4184            if let Some(slot) = cached.pages.get_mut(entry.stripe) {
4185                if slot.as_ref().is_some_and(|slot| Arc::ptr_eq(&slot.used, &entry.used)) {
4186                    *slot = None;
4187                }
4188            }
4189        }
4190    }
4191}
4192
4193/// Stripes of one column a reader keeps the bytes of, when nobody has asked for more.
4194///
4195/// This has to hold at least as many stripes as a column has workers in it at once, or the workers
4196/// evict each other's pages and read them again. Four is what a scan that hands parts out in order
4197/// needs, because then every worker is within a few parts of every other and at most a couple of
4198/// stripes are open at a time. A scan that hands a whole stripe to each worker has one stripe open
4199/// per worker for the length of that stripe, and it says so with [`Reader::keep_stripes`] rather
4200/// than paying for sixteen slots on every table that is read one part at a time.
4201///
4202/// It multiplies by the page size, which is a quarter of a megabyte for a four byte column, and by
4203/// the number of columns a query touches.
4204const CACHED_STRIPES_PER_COLUMN: usize = 4;
4205
4206/// The sieves of one stripe of one column, once somebody has asked for them.
4207type SieveSlot = OnceLock<Arc<Vec<Option<Sieve>>>>;
4208
4209type RangeSlot = OnceLock<Arc<Vec<Range>>>;
4210
4211#[derive(Debug)]
4212struct NativeText {
4213    file: Arc<File>,
4214    /// How many values the dictionary holds.
4215    values: usize,
4216    /// Where each value ends inside its payload block, packed at `offset_bits` in runs of
4217    /// [`TEXT_OFFSET_RUN`].
4218    ///
4219    /// Ends rather than starts, because then a block of 1,024 values is 1,024 numbers rather than
4220    /// 1,025: the start of a value is the end of the one before it, and the first value of a block
4221    /// starts at zero by construction. Relative to the block rather than to the payload, because a
4222    /// reader decodes a whole block and slices it, so an offset into the payload is a number it
4223    /// would have to subtract a base from anyway.
4224    ///
4225    /// The vector is the index as it was read, so the offsets start after the header, and
4226    /// [`Self::packed`] is where they are read from.
4227    offsets: Vec<u8>,
4228    /// Bits one offset is packed at, which is what the largest block of this column spans and is the
4229    /// same for every block of it.
4230    offset_bits: usize,
4231    /// The same ends unpacked, built once enough readers have asked for one at a time.
4232    ///
4233    /// Reading one offset out of the packed form costs about fifty instructions: a division to find
4234    /// the run, a bounds check to slice it, a shift to reach the bit the value starts at and a
4235    /// narrowing on the way out. That is the right price for a reader that wants a handful. It is
4236    /// the wrong price for `STRLEN` over a column, which asks for one per row and nothing else, and
4237    /// where a million of them was a third of ClickBench 28.
4238    ///
4239    /// With the ends unpacked every read is a load, and a vector of lengths is one loop over them.
4240    /// The table is built only once the reads say it will be used, which is what
4241    /// [`Self::ends_worth_unpacking`] decides and [`Self::ends_asked`] counts towards, because a
4242    /// table built for a reader that wanted three values is four bytes a value spent on nothing.
4243    value_ends: OnceLock<Option<Vec<u32>>>,
4244    /// The length of every value, worked out of [`Self::value_ends`] the first time a vector of
4245    /// lengths is asked for.
4246    ///
4247    /// A length out of the ends is two loads, a test for whether the value opens its block and a
4248    /// check that it does not end before it starts, which came to thirteen instructions a row on
4249    /// ClickBench 28. Out of this it is one load. The order is checked once for the whole table
4250    /// while it is built, and a column that fails it gets no table and goes on reading the ends,
4251    /// which is where the error is reported. Two bytes a value where every value is short enough,
4252    /// four otherwise, and only for a column something has asked the length of a vector at a time.
4253    value_lens: OnceLock<Option<Lengths>>,
4254    /// How many single offset reads have come in while the table is not built.
4255    ///
4256    /// Relaxed, and read only against a threshold, so two threads racing here means the table is
4257    /// built one read early or one read late. Counting stops the moment the table exists, because
4258    /// [`OnceLock::get`] settles it before this is touched.
4259    ends_asked: AtomicUsize,
4260    /// How many entries the sorted order has, which is the value count.
4261    ranks: usize,
4262    /// Where the sorted order starts in the file. It is read a block at a time and only when
4263    /// something searches it, so a query that never compares this column against a literal never
4264    /// touches it at all.
4265    rank_at: u64,
4266    /// Where each block of the sorted order ends, as a byte offset from `rank_at`. A block is packed
4267    /// at whatever width its own heads need, so unlike the entries it replaced its length is not
4268    /// arithmetic on the block number.
4269    rank_ends: Vec<u64>,
4270    rank_hashes: Vec<u64>,
4271    rank_blocks: Vec<OnceLock<Result<Vec<u8>>>>,
4272    /// Bits one code is packed at, which is what the value count needs and is the same for every
4273    /// block of the column.
4274    code_bits: usize,
4275    /// The sorted order turned round, built the first time a reader asks for it.
4276    ///
4277    /// Four bytes per value against the four the offsets already hold, so a column that has this is
4278    /// carrying half again what it carried before rather than something of a new order. It is built
4279    /// only when something asks, which is a grouped min or max over this column and nothing else,
4280    /// and that reader was going to read the payload of this column once per row otherwise.
4281    code_ranks: OnceLock<Option<Vec<u32>>>,
4282    /// Where each block of the payload starts in the file, and how many stored bytes it is.
4283    ///
4284    /// Absolute rather than an offset from a base the blocks share, because a block is written the
4285    /// moment it fills and what comes after it in the file is whatever the load wrote next. A file
4286    /// old enough to have them back to back is read into these same two lists by adding the base to
4287    /// the ends it carries, so nothing below here knows which kind of file it came from.
4288    starts: Vec<u64>,
4289    lengths: Vec<u64>,
4290    hashes: Vec<u64>,
4291    /// Conservative four-byte substring signatures, read only by a compatible LIKE filter.
4292    grams: Option<NativeGrams>,
4293    /// The payload, read and decoded a block at a time and kept after that.
4294    blocks: Vec<OnceLock<Result<Vec<u8>>>>,
4295    /// The length in characters of every value of a block, worked out the first time `length` asks
4296    /// for a value in that block.
4297    ///
4298    /// Kept instead of the block it was counted out of. `length` reads every row of a column, and
4299    /// reading the bytes through [`Self::payload_block`] kept every block it touched, which is every
4300    /// distinct value of the column decoded: seven string columns of ClickBench held 13.9 GB to
4301    /// answer seven `max(length(...))`. The counts are four bytes a value, so the same scan keeps
4302    /// the counts and decodes each block once, the same number of times it did before.
4303    char_lens: Vec<OnceLock<Box<[u32]>>>,
4304    /// How many decoded payload bytes this column keeps before a sweep stops keeping what it reads.
4305    /// [`TEXT_KEEP_BUDGET`] everywhere but in the test of the ceiling.
4306    keep_budget: usize,
4307    /// Roughly how many decoded payload bytes are being kept, which is what [`TEXT_KEEP_BUDGET`]
4308    /// is measured against.
4309    ///
4310    /// Roughly, because two threads that keep the same block at the same time both add its length
4311    /// while [`OnceLock`] keeps one of the two. That makes the count read high and the budget bind
4312    /// a little early, which is the harmless direction, and it costs one relaxed add a block rather
4313    /// than a lock on the path every scan of a string column goes through.
4314    payload_kept: AtomicUsize,
4315    /// Which payload blocks a sweep has decoded before, one flag a block.
4316    ///
4317    /// A sweep keeps a block the second time it decodes it and not the first. A process that runs
4318    /// one statement, which is how a benchmark or a script uses the engine, sweeps each block once
4319    /// and so keeps nothing: on ten million rows a `URL LIKE` held 396 MB with every block kept and
4320    /// 97 MB with none, for the same processor time. A session that asks again pays the decode one
4321    /// more time and reads kept blocks from then on, under the same [`TEXT_KEEP_BUDGET`].
4322    swept: Vec<AtomicBool>,
4323    /// How many blocks [`TextSource::visit_at`] has decoded and dropped because the column was
4324    /// already holding its [`TEXT_KEEP_BUDGET`].
4325    ///
4326    /// A sweep reads the dictionary in order and touches a block once, so dropping what it reads
4327    /// past the budget costs one decode a block and bounds the column. A visit reads a vector of
4328    /// codes, and the codes of a scan land all over the dictionary: on ten million rows of
4329    /// ClickBench each vector of two thousand `URL`s touches about a hundred and forty of its two
4330    /// and a half thousand blocks, and so does the next one. A cache holding a tenth of the column
4331    /// still misses half of those, and dropping every block past the budget would decode the
4332    /// column hundreds of times over to answer one `lower(URL)`. So a visit drops past the budget
4333    /// only until it has dropped as many blocks as the column has, which is what a read whose codes
4334    /// are few or clustered never reaches, and keeps what it reads after that, the way a row at a
4335    /// time read always did. That bounds what a visit can cost over the old read at one more decode
4336    /// of the column.
4337    visit_dropped: AtomicUsize,
4338    /// The boundaries this dictionary has already been searched for, by the value searched for.
4339    ///
4340    /// A search is the expensive thing this type does. It settles a probe on the stored head where
4341    /// it can and reads a value where it cannot, and reading a value decodes the payload block it
4342    /// sits in, so one search can cost several blocks. The thing that makes remembering worth it is
4343    /// that the same search comes back: a top N asks once a chunk whether anything left can beat its
4344    /// worst candidate, and the worst candidate settles long before the chunks run out.
4345    ///
4346    /// Shared across the instances of a scan rather than kept per instance, because each of them has
4347    /// its own worst candidate and all of them are searching the same dictionary. One lock per chunk
4348    /// is nothing next to a probe of a file.
4349    ///
4350    /// Bounded by [`TEXT_SEARCH_MEMO`] and emptied rather than evicted when it is full. What fills
4351    /// it is a top N improving its bound, which happens a few dozen times and then stops, so the
4352    /// bound is there for the filter that searches for a different literal every chunk rather than
4353    /// for anything this is meant to help.
4354    searched: Mutex<HashMap<Vec<u8>, (usize, bool)>>,
4355}
4356
4357#[derive(Debug)]
4358struct NativeGrams {
4359    start: u64,
4360    length: usize,
4361    /// How long one block's signature is.
4362    width: usize,
4363    hash: u64,
4364    /// For each literal asked about lately, whether each block might hold it.
4365    ///
4366    /// The answer for every block at once, worked out by one pass over the signatures a window at a
4367    /// time, rather than the signatures read in and kept. On ClickBench `URL` they are 21 MB for
4368    /// ten million rows and a verdict is 2,650 flags, and a filter asks the same question of every
4369    /// block, so the pass is paid once and what stays resident is the flags.
4370    verdicts: Mutex<Vec<Verdict>>,
4371}
4372
4373/// A literal and whether each block might hold it.
4374type Verdict = (Vec<u8>, Arc<[bool]>);
4375
4376/// How many literals a column remembers the verdicts of.
4377const GRAM_VERDICTS: usize = 8;
4378
4379impl NativeGrams {
4380    /// Whether each block might hold `literal`, remembered or worked out now.
4381    ///
4382    /// The lock is held over the pass so that the threads of one scan, which all ask about the
4383    /// same literal at the start, read the signatures once between them.
4384    fn verdicts(&self, file: &File, literal: &[u8]) -> Result<Arc<[bool]>> {
4385        let mut held = self.verdicts.lock().map_err(|_| invalid("a poisoned signature verdict"))?;
4386        if let Some((_, verdict)) = held.iter().find(|(asked, _)| asked == literal) {
4387            return Ok(Arc::clone(verdict));
4388        }
4389        let wanted = literal.windows(4).map(|gram| gram_bits(gram, self.width)).collect::<Vec<_>>();
4390        let mut verdict = Vec::with_capacity(self.length / self.width);
4391        let window = GRAM_WINDOW / self.width * self.width;
4392        let hash = walk_checksummed(file, self.start, self.length, window, |bytes| {
4393            verdict.extend(bytes.chunks(self.width).map(|bits| {
4394                wanted
4395                    .iter()
4396                    .flatten()
4397                    .all(|&bit| bits.get(bit / 8).is_some_and(|byte| byte & (1 << (bit % 8)) != 0))
4398            }));
4399            Ok(())
4400        })?;
4401        if hash != self.hash {
4402            return Err(invalid("global dictionary substring signatures checksum differs"));
4403        }
4404        let verdict: Arc<[bool]> = verdict.into();
4405        if held.len() >= GRAM_VERDICTS {
4406            held.remove(0);
4407        }
4408        held.push((literal.to_vec(), Arc::clone(&verdict)));
4409        Ok(verdict)
4410    }
4411
4412    fn footprint(&self) -> usize {
4413        self.verdicts.lock().map_or(0, |held| {
4414            held.iter().map(|(asked, verdict)| asked.capacity() + verdict.len()).sum()
4415        })
4416    }
4417}
4418
4419/// How many searched for values a column's dictionary remembers the boundary of.
4420///
4421/// See [`NativeText::searched`]. Small because the case it is for repeats one value, not because a
4422/// larger one would be wrong.
4423const TEXT_SEARCH_MEMO: usize = 64;
4424
4425/// How many values of a dictionary go in one block of the payload.
4426///
4427/// The block is the unit the string cascade encodes, the unit a checksum covers, and the unit a
4428/// reader has to decode to get at a single value, so it is the one number the payload format turns
4429/// on. Blocking by values rather than by bytes is what keeps a value out of two blocks at once: the
4430/// block holding a code is `code / TEXT_PAYLOAD_VALUES` and nothing has to be stitched.
4431///
4432/// A probe on the five ClickBench columns that have a dictionary worth the name, written up on
4433/// #347, measured the ratio and the decode speed at 128, 256, 512, 1,024 and 4,096 values. Both get
4434/// better all the way up, because front coding and the LZ matcher have more to look back at and
4435/// because the per chunk setup is spread over more values. What stops it is the point read: a query
4436/// that wants ten values has to decode ten blocks, so the block is what a lookup costs. At 1,024
4437/// values a block is between 67 KB and 394 KB decoded across those five columns, and the ratios are
4438/// 2.3 to 4.5. Going up to 4,096 buys two to six percent more and makes a block as much as 1.5 MB.
4439/// Going down to 512 gives up five to nine percent.
4440const TEXT_PAYLOAD_VALUES: usize = 1024;
4441
4442/// Eight KiB per payload block, which is what makes a four-byte substring a useful negative test on
4443/// a column of URLs.
4444///
4445/// Two KiB was the first answer and on ClickBench `URL` it proved almost nothing. A block of 1,024
4446/// sorted URLs holds about seventeen thousand distinct four-byte grams, and at two bits each that
4447/// set nine in ten of the sixteen thousand bits there were, so `LIKE '%google%'` passed most blocks
4448/// it had no match in and decoded them. At eight KiB four bits in ten are set, and of the 2,650
4449/// blocks of `URL` in ten million rows a needle that is in none of them passes 36. The signatures
4450/// are not read into memory, see [`NativeGrams::verdicts`], so the width costs file and not
4451/// resident memory.
4452const TEXT_GRAM_BYTES: usize = 8192;
4453
4454/// The signature width of a format 28 file, which is still read.
4455const NARROW_GRAM_BYTES: usize = 2048;
4456
4457/// How much of a column's signatures a verdict reads at a time.
4458const GRAM_WINDOW: usize = 256 << 10;
4459
4460/// A fast mixing step for exactly four bytes, shared by load and query, into a signature of
4461/// `width` bytes.
4462fn gram_bits(bytes: &[u8], width: usize) -> [usize; 2] {
4463    let original = u32::from_le_bytes(bytes.try_into().expect("a four-byte gram"));
4464    let mut first = original ^ (original >> 16);
4465    first = first.wrapping_mul(0x7feb_352d);
4466    first ^= first >> 15;
4467    let mut second = original ^ (original >> 17);
4468    second = second.wrapping_mul(0x846c_a68b);
4469    second ^= second >> 16;
4470    let mask = width * 8 - 1;
4471    [(first as usize) & mask, (second as usize) & mask]
4472}
4473
4474/// How many decoded payload bytes one dictionary keeps before a sweep stops keeping what it reads.
4475///
4476/// A sweep of the whole dictionary decodes every block whatever it does, and the only question is
4477/// whether it hangs on to them. Keeping all of them is 4.2 GB on ClickBench `URL` at a hundred
4478/// million rows, which is what #997 was right to stop. Keeping none of them means the next query
4479/// asking the same thing decodes all of it again, and on the same column at a million rows that
4480/// took a `LIKE` from 2.7 ms to 16.2 ms, because the decode used to be paid once by a session and
4481/// is now paid by every statement in it. Neither end is the answer. A bound is.
4482///
4483/// So a sweep keeps what it decodes until the column is holding this much and decodes without
4484/// keeping after that. At a million rows the five ClickBench string columns decode to between 8 MB
4485/// and 85 MB, so they sit inside it and a repeated `LIKE` reads a decoded block rather than a
4486/// stored one. At a hundred million rows `URL` fills it and the rest of that column is read and
4487/// dropped, which is the old cost on the part that does not fit and none of the old footprint.
4488///
4489/// Two hundred and fifty six megabytes a column is a number and not a policy, and the policy is
4490/// what should replace it: this wants to be a buffer pool over the whole database, sized against
4491/// the memory limit the session was given, with the blocks of every column competing for it and the
4492/// least useful one evicted. That is F2 work. What is here is the part of it that can be written
4493/// without an eviction order, which is a ceiling.
4494const TEXT_KEEP_BUDGET: usize = 256 * 1024 * 1024;
4495
4496/// The length of every value of a column, as narrow as the longest of them allows.
4497///
4498/// The table is read at the codes a vector holds, which on a column the size of ClickBench `URL`
4499/// land all over it, so what a length costs is whether its line is in cache. Half a million URLs
4500/// are two megabytes at four bytes a length and one at two, which is the difference between the
4501/// table sitting in the second level cache or not.
4502#[derive(Debug)]
4503enum Lengths {
4504    /// Every length fits in sixteen bits.
4505    Narrow(Vec<u16>),
4506    /// Some value is longer than that.
4507    Wide(Vec<u32>),
4508}
4509
4510impl Lengths {
4511    /// The lengths at `indices`, appended to `into`, and zero for a position past the end, which
4512    /// is what a row at a time read says.
4513    fn extend_at(&self, indices: &[u32], into: &mut Vec<i64>) {
4514        match self {
4515            Lengths::Narrow(lens) => into.extend(
4516                indices
4517                    .iter()
4518                    .map(|&index| lens.get(index as usize).map_or(0, |&len| i64::from(len))),
4519            ),
4520            Lengths::Wide(lens) => into.extend(
4521                indices
4522                    .iter()
4523                    .map(|&index| lens.get(index as usize).map_or(0, |&len| i64::from(len))),
4524            ),
4525        }
4526    }
4527
4528    /// The bytes the table holds on to.
4529    fn footprint(&self) -> usize {
4530        match self {
4531            Lengths::Narrow(lens) => lens.capacity() * size_of::<u16>(),
4532            Lengths::Wide(lens) => lens.capacity() * size_of::<u32>(),
4533        }
4534    }
4535}
4536
4537/// The length of every value out of where each one ends inside its payload block, or `None` for
4538/// ends that go backwards somewhere inside a block.
4539///
4540/// A value that opens a block starts at zero and every other one starts where the value before it
4541/// ends, so a block is a run of differences.
4542///
4543/// Built at two bytes a length straight away, and built again at four only when some value turns
4544/// out too long for that, which is rare enough that the second pass is not worth avoiding.
4545fn lengths_of(ends: &[u32]) -> Option<Lengths> {
4546    match lengths_as::<u16>(ends)? {
4547        Some(narrow) => Some(Lengths::Narrow(narrow)),
4548        None => lengths_as::<u32>(ends)?.map(Lengths::Wide),
4549    }
4550}
4551
4552/// [`lengths_of`] at one width: `None` for ends that go backwards, and `Some(None)` for a length
4553/// that does not fit in `T`.
4554fn lengths_as<T: TryFrom<u32>>(ends: &[u32]) -> Option<Option<Vec<T>>> {
4555    let mut lens = Vec::with_capacity(ends.len());
4556    for block in ends.chunks(TEXT_PAYLOAD_VALUES) {
4557        let mut start = 0;
4558        for &end in block {
4559            let Ok(len) = T::try_from(end.checked_sub(start)?) else {
4560                return Some(None);
4561            };
4562            lens.push(len);
4563            start = end;
4564        }
4565    }
4566    Some(Some(lens))
4567}
4568
4569/// How many offsets go in one packed run.
4570///
4571/// A payload block holds 1,024 values and `bitpack::pack_tail` takes fewer than 1,024 at a time,
4572/// since a whole unit of that many belongs in the transposed layout instead. So the offsets of a
4573/// block go in two runs. Five hundred and twelve values at any width is a whole number of bytes, so
4574/// a run starts where a multiply says it does and nothing is padded.
4575const TEXT_OFFSET_RUN: usize = 512;
4576
4577/// Bytes at the front of a global dictionary index: the value count, the values a payload block
4578/// holds, the block count and the bits an offset is packed at.
4579const DICTIONARY_HEADER: usize = 16;
4580
4581/// Set beside the offset width in the fourth word of a global dictionary index, meaning each
4582/// payload block says where in the file it starts and how long it is, rather than sitting directly
4583/// behind the block before it.
4584///
4585/// In that word rather than in a word of its own because the width is at most 32 and lives in a
4586/// `u32`, so the top of it has never been anything. A build old enough not to know the flag reads
4587/// the file's format before it reads any of this and refuses it there, and if it somehow did get
4588/// here it would find an offset width of two billion and say so.
4589///
4590/// The point of the flag is that a block written the moment it fills does not know what will be
4591/// written after it, so the payload of a column cannot be one run of bytes unless the whole column
4592/// is held until the file is closed. That is the memory the load cannot afford. What it costs is
4593/// eight bytes a block, against the block being a thousand values.
4594const DICTIONARY_SCATTERED: u32 = 1 << 31;
4595/// The dictionary index carries one four-byte substring signature per payload block.
4596const DICTIONARY_GRAMS: u32 = 1 << 30;
4597/// Each signature is [`TEXT_GRAM_BYTES`] long rather than the [`NARROW_GRAM_BYTES`] a format 28
4598/// file wrote.
4599const DICTIONARY_WIDE_GRAMS: u32 = 1 << 29;
4600/// Every flag the width word of a dictionary can carry above the offset width.
4601const DICTIONARY_FLAGS: u32 = DICTIONARY_SCATTERED | DICTIONARY_GRAMS | DICTIONARY_WIDE_GRAMS;
4602
4603/// How many entries of a dictionary's sorted order sit in one block that is read and checked as a
4604/// unit.
4605///
4606/// Five hundred and twelve entries is between two and three kilobytes on the ClickBench string
4607/// columns, which is well under a page. A binary search over half a million entries makes nineteen
4608/// probes, and the first ten land in ten different blocks while the last nine land in the one block
4609/// that holds the answer, so the whole search reads about thirty kilobytes of a megabyte of order. A
4610/// smaller block would save a little on the early probes, cost a checksum and an end list four times
4611/// as long, and give the heads less to share a base with. A larger one would read more than it uses
4612/// on every probe.
4613const TEXT_RANK_BLOCK: usize = 512;
4614
4615/// Bytes at the front of a rank block, which is the base of its heads and the width they are packed
4616/// at.
4617///
4618/// An entry used to be twelve bytes flat, eight for the head and four for the code, and on the five
4619/// ClickBench columns that have a dictionary worth the name that was 744 MB of a 12.2 GB file. Both
4620/// halves of it are nearly empty. The heads are the first eight bytes of the values in sorted order,
4621/// so a block of five hundred and twelve of them spans a tiny slice of the column, and on a column of
4622/// URLs they are all `http://w` and the block holds one distinct head. The codes are positions in a
4623/// dictionary of eighteen million, which is twenty five bits and not thirty two.
4624///
4625/// So a block now writes the smallest head in it, the bits the largest is above that, and the heads
4626/// and the codes packed at the width each needs. A block where every head agrees costs nine bytes
4627/// and the codes.
4628const RANK_BLOCK_HEADER: usize = size_of::<u64>() + 1;
4629
4630impl NativeText {
4631    /// One block of the payload, read and decoded the first time anything asks for a value in it.
4632    ///
4633    /// The bytes handed back are the values of the block laid end to end, which is what the offsets
4634    /// describe, so a caller slices it with the offsets it already has. Where the block sits in the
4635    /// file is the only thing the caller cannot work out for itself, because the stored form is
4636    /// shorter than the decoded one and by a different amount in every block.
4637    fn payload_block(&self, block: usize) -> Result<Option<&[u8]>> {
4638        let Some(slot) = self.blocks.get(block) else { return Ok(None) };
4639        let bytes = slot.get_or_init(|| self.decode_block(block)).as_ref().map_err(Clone::clone)?;
4640        Ok(Some(bytes.as_slice()))
4641    }
4642
4643    /// The character length of every value in one block, counted the first time it is asked for.
4644    ///
4645    /// The block is read out of [`Self::blocks`] where something already kept it and decoded and
4646    /// dropped where nothing did, so counting never adds a block to what this column holds. Two
4647    /// threads asking for the same block at once both count it and one of the two answers is kept,
4648    /// which costs a decode and is cheaper than a lock on every lookup.
4649    fn block_chars(&self, block: usize) -> Result<&[u32]> {
4650        let slot = self
4651            .char_lens
4652            .get(block)
4653            .ok_or_else(|| invalid("a block past the global dictionary"))?;
4654        if let Some(lens) = slot.get() {
4655            return Ok(lens);
4656        }
4657        let decoded;
4658        let bytes: &[u8] = match self.blocks.get(block).and_then(OnceLock::get) {
4659            Some(Ok(kept)) => kept,
4660            _ => {
4661                decoded = self.decode_block(block)?;
4662                &decoded
4663            }
4664        };
4665        let first = block * TEXT_PAYLOAD_VALUES;
4666        let last = (first + TEXT_PAYLOAD_VALUES).min(self.values);
4667        let ends = self.ends_within(first, last)?;
4668        if ends.len() != last - first {
4669            return Err(invalid("global dictionary offsets are short"));
4670        }
4671        let mut lens = Vec::with_capacity(ends.len());
4672        let mut start = u64::from(self.start_within(first)?);
4673        for &end in &ends {
4674            let value = usize::try_from(start)
4675                .ok()
4676                .zip(usize::try_from(end).ok())
4677                .and_then(|(from, to)| bytes.get(from..to))
4678                .ok_or_else(|| invalid("global dictionary value is past its block"))?;
4679            // A continuation byte of UTF-8 is `0b10xx_xxxx` and every other byte starts a
4680            // character, so the bytes that are not continuations are the characters.
4681            let characters = value.iter().filter(|byte| (**byte as i8) >= -0x40).count();
4682            lens.push(u32::try_from(characters).unwrap_or(u32::MAX));
4683            start = end;
4684        }
4685        Ok(slot.get_or_init(|| lens.into_boxed_slice()))
4686    }
4687
4688    /// Reads and decodes one block of the payload, without deciding who keeps it.
4689    ///
4690    /// [`Self::payload_block`] keeps it forever, which is what a point read wants and what a walk
4691    /// of the whole dictionary must not do. Both call this and they differ in nothing else.
4692    fn decode_block(&self, block: usize) -> Result<Vec<u8>> {
4693        let len = self.lengths[block];
4694        let mut stored = vec![
4695            0;
4696            usize::try_from(len).map_err(|_| invalid(
4697                "global dictionary block does not fit in memory"
4698            ))?
4699        ];
4700        read_at(&self.file, self.starts[block], &mut stored)?;
4701        if checksum(&stored) != self.hashes[block] {
4702            return Err(invalid("global dictionary payload checksum differs"));
4703        }
4704        let first = block * TEXT_PAYLOAD_VALUES;
4705        let last = (first + TEXT_PAYLOAD_VALUES).min(self.values);
4706        let want = self.end_within(last - 1)? as usize;
4707        let values = string::decode_flat(&stored)?;
4708        if values.len() != last - first {
4709            return Err(invalid("global dictionary block holds the wrong value count"));
4710        }
4711        let bytes = values.into_bytes();
4712        if bytes.len() != want {
4713            return Err(invalid("global dictionary block decodes to the wrong length"));
4714        }
4715        Ok(bytes)
4716    }
4717
4718    /// The block holding a value that a read hands over on loan, kept or decoded for the call.
4719    ///
4720    /// A block something already kept is read where it is. One nothing kept is kept the second
4721    /// time a loaned read decodes it while the column is holding less than [`Self::keep_budget`],
4722    /// and decoded into `decoded` and dropped with it otherwise, which is the policy
4723    /// [`TextSource::sweep`] explains. `scattered` is a read by code rather than in order, which
4724    /// stops dropping once it has dropped a column's worth of blocks, for the reason
4725    /// [`Self::visit_dropped`] gives.
4726    fn loaned_block<'a>(
4727        &'a self,
4728        block: usize,
4729        decoded: &'a mut Vec<u8>,
4730        scattered: bool,
4731    ) -> Result<&'a [u8]> {
4732        let kept = self.blocks.get(block).and_then(OnceLock::get);
4733        if let Some(Ok(kept)) = kept {
4734            return Ok(kept);
4735        }
4736        let again = kept.is_none()
4737            && self.swept.get(block).is_some_and(|swept| swept.swap(true, Atomic::Relaxed));
4738        let keep = again
4739            && (self.payload_kept.load(Atomic::Relaxed) < self.keep_budget
4740                || (scattered && self.visit_dropped.load(Atomic::Relaxed) >= self.blocks.len()));
4741        if keep {
4742            let kept = self
4743                .payload_block(block)?
4744                .ok_or_else(|| invalid("global dictionary block is past the payload"))?;
4745            self.payload_kept.fetch_add(kept.len(), Atomic::Relaxed);
4746            return Ok(kept);
4747        }
4748        *decoded = self.decode_block(block)?;
4749        if scattered && again {
4750            self.visit_dropped.fetch_add(1, Atomic::Relaxed);
4751        }
4752        Ok(decoded)
4753    }
4754
4755    /// How many single offset reads make [`Self::value_ends`] worth building.
4756    ///
4757    /// As many reads as the dictionary has values. Building the table costs about thirty
4758    /// instructions a value once the fresh pages it lands in are counted, and a read out of it saves
4759    /// about thirty five, so it repays itself after roughly one read per value. The reads so far are
4760    /// the only guess there is at the reads to come, and waiting until they match the size of the
4761    /// dictionary is betting that a column read that much will be read that much again.
4762    ///
4763    /// A sixteenth was the first answer, from counting the unpacking alone at three instructions a
4764    /// value. ClickBench 38 showed what that missed: it reads about twenty thousand titles a
4765    /// statement out of a dictionary of three hundred and fifty thousand, crossed a sixteenth in its
4766    /// second statement and was two percent slower for a table it did not read enough to repay. A
4767    /// scan asking for the length of every row crosses it part way through its first statement on
4768    /// ClickBench, where a string column has about two rows for every value, and a filter that keeps
4769    /// a few thousand rows never does. The floor is there
4770    /// because a short dictionary would otherwise build a table for a handful of reads.
4771    fn ends_worth_unpacking(&self) -> usize {
4772        self.values.max(TEXT_PAYLOAD_VALUES)
4773    }
4774
4775    /// The unpacked ends, if they are built or if this read is the one that makes them worth it.
4776    fn value_ends(&self) -> Option<&[u32]> {
4777        if let Some(built) = self.value_ends.get() {
4778            return built.as_deref();
4779        }
4780        if self.ends_asked.fetch_add(1, Atomic::Relaxed) < self.ends_worth_unpacking() {
4781            return None;
4782        }
4783        self.value_ends.get_or_init(|| self.unpack_ends()).as_deref()
4784    }
4785
4786    /// Every end of the column, a run at a time.
4787    ///
4788    /// `None` rather than an error on anything wrong, because this is a cache in front of a reader
4789    /// that answers the same question. A column whose offsets are short or whose ends do not fit in
4790    /// four bytes gets no table and the same error it would have got, from the read that wanted it.
4791    fn unpack_ends(&self) -> Option<Vec<u32>> {
4792        let mut ends = vec![0u32; self.values];
4793        for (run, into) in ends.chunks_mut(TEXT_OFFSET_RUN).enumerate() {
4794            let bytes = self.packed().get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)?;
4795            bitpack::unpack_tail_into(bytes, self.offset_bits, into, |bits| {
4796                u32::try_from(bits).unwrap_or(u32::MAX)
4797            })
4798            .ok()?;
4799        }
4800        // An end that did not fit was stored as the sentinel, and a real one cannot reach it because
4801        // a payload block is far smaller than four gigabytes. So the column keeps the packed reader.
4802        if ends.contains(&u32::MAX) { None } else { Some(ends) }
4803    }
4804
4805    /// The packed offsets, which is the index past its header.
4806    fn packed(&self) -> &[u8] {
4807        self.offsets.get(DICTIONARY_HEADER..).unwrap_or_default()
4808    }
4809
4810    /// Where the value at `index` ends inside its payload block.
4811    fn end_within(&self, index: usize) -> Result<u32> {
4812        if let Some(ends) = self.value_ends() {
4813            return ends
4814                .get(index)
4815                .copied()
4816                .ok_or_else(|| invalid("global dictionary offsets are short"));
4817        }
4818        let run = index / TEXT_OFFSET_RUN;
4819        let bytes = self
4820            .packed()
4821            .get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)
4822            .ok_or_else(|| invalid("global dictionary offsets are short"))?;
4823        let end = bitpack::tail_at(bytes, self.offset_bits, index % TEXT_OFFSET_RUN)
4824            .map_err(|_| invalid("global dictionary offsets are short"))?;
4825        u32::try_from(end).map_err(|_| invalid("global dictionary offset is past the payload"))
4826    }
4827
4828    /// Where every value in `first..last` ends inside its payload block, in one pass over the runs.
4829    ///
4830    /// [`Self::end_within`] answers for one value and pays for it twice over: it shifts a window to
4831    /// the bit the value starts at, and the copy that fills that window is a length the compiler does
4832    /// not know, so it is a call to `memcpy` rather than a load. A sweep asked for two of those per
4833    /// value, one for the end and one for the start that is the end before it, and on the ClickBench
4834    /// `URL` dictionary of eighteen million that was most of the half second a `LIKE` over it took.
4835    ///
4836    /// [`bitpack::unpack_tail_into`] walks the run instead, which makes the window a fixed width and
4837    /// so an unaligned load, and reads the bit position off a counter. A run is five hundred and
4838    /// twelve values and a block is two of them, so a block of a thousand and twenty four values
4839    /// costs two calls here and nothing per value.
4840    ///
4841    /// The answer is written straight into the result. A run that is wanted from its first value,
4842    /// which is every run but the one the sweep starts in, unpacks into its own window of the result
4843    /// and is never copied. Only a run joined part way through needs the scratch buffer, and there is
4844    /// at most one of those per sweep, so the buffer is allocated the first time one turns up.
4845    fn ends_within(&self, first: usize, last: usize) -> Result<Vec<u64>> {
4846        let mut ends = vec![0u64; last.saturating_sub(first)];
4847        let mut scratch = Vec::new();
4848        let mut at = first;
4849        while at < last {
4850            let run = at / TEXT_OFFSET_RUN;
4851            let stop = ((run + 1) * TEXT_OFFSET_RUN).min(last);
4852            let held = self.values.saturating_sub(run * TEXT_OFFSET_RUN).min(TEXT_OFFSET_RUN);
4853            let bytes = self
4854                .packed()
4855                .get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)
4856                .ok_or_else(|| invalid("global dictionary offsets are short"))?;
4857            let from = at % TEXT_OFFSET_RUN;
4858            let upto = stop - run * TEXT_OFFSET_RUN;
4859            if upto > held || bytes.len() < bitpack::tail_len(held, self.offset_bits) {
4860                return Err(invalid("global dictionary offsets are short"));
4861            }
4862            let into = &mut ends[at - first..stop - first];
4863            if from == 0 {
4864                bitpack::unpack_tail_into(bytes, self.offset_bits, into, |bits| bits)
4865                    .map_err(|_| invalid("global dictionary offsets are short"))?;
4866            } else {
4867                scratch.resize(held, 0);
4868                bitpack::unpack_tail_into(bytes, self.offset_bits, &mut scratch, |bits| bits)
4869                    .map_err(|_| invalid("global dictionary offsets are short"))?;
4870                into.copy_from_slice(&scratch[from..upto]);
4871            }
4872            at = stop;
4873        }
4874        Ok(ends)
4875    }
4876
4877    /// Where the value at `index` starts inside its payload block, which is where the value before
4878    /// it ended unless it is the first of the block.
4879    fn start_within(&self, index: usize) -> Result<u32> {
4880        if index % TEXT_PAYLOAD_VALUES == 0 { Ok(0) } else { self.end_within(index - 1) }
4881    }
4882
4883    /// Where the value at `index` starts and ends inside its payload block.
4884    ///
4885    /// The two offsets sit next to each other in the same run unless the value opens one, and a run
4886    /// of seventeen bit offsets, which is what a block of a thousand strings needs, puts a pair of
4887    /// them inside one eight byte load. So the common case reads the packed bytes once rather than
4888    /// twice and does the bounds arithmetic once. This is asked once per string a text column hands
4889    /// out, and on ClickBench 27 the two reads together were a quarter of the query.
4890    fn span_within(&self, index: usize) -> Result<(u32, u32)> {
4891        if let Some(ends) = self.value_ends() {
4892            let end =
4893                *ends.get(index).ok_or_else(|| invalid("global dictionary offsets are short"))?;
4894            // The value before it in the same block, and zero where there is no value before it.
4895            // `index` is inside the table, so the one under it is too.
4896            let start = if index % TEXT_PAYLOAD_VALUES == 0 { 0 } else { ends[index - 1] };
4897            if start > end {
4898                return Err(invalid("global dictionary value ends before it starts"));
4899            }
4900            return Ok((start, end));
4901        }
4902        let within = index % TEXT_OFFSET_RUN;
4903        let (start, end) = if within == 0 {
4904            (self.start_within(index)?, self.end_within(index)?)
4905        } else {
4906            let run = index / TEXT_OFFSET_RUN;
4907            let bytes = self
4908                .packed()
4909                .get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)
4910                .ok_or_else(|| invalid("global dictionary offsets are short"))?;
4911            let (start, end) = bitpack::tail_pair(bytes, self.offset_bits, within)
4912                .map_err(|_| invalid("global dictionary offsets are short"))?;
4913            let ends = u32::try_from(end)
4914                .map_err(|_| invalid("global dictionary offset is past the payload"))?;
4915            let starts = u32::try_from(start)
4916                .map_err(|_| invalid("global dictionary offset is past the payload"))?;
4917            (starts, ends)
4918        };
4919        if start > end {
4920            return Err(invalid("global dictionary value ends before it starts"));
4921        }
4922        Ok((start, end))
4923    }
4924
4925    /// The block of the sorted order that holds `rank`, and where in it that rank sits.
4926    ///
4927    /// The block is read from the file and checked against the hash the index carries for it the
4928    /// first time anything asks, and kept after that, the same way a payload block is. A search
4929    /// makes about as many probes as the order has bits, so the whole search reads a handful of
4930    /// these and never the rest.
4931    fn rank_parts(&self, rank: usize) -> Result<(&[u8], usize)> {
4932        let slot = self
4933            .rank_blocks
4934            .get(rank / TEXT_RANK_BLOCK)
4935            .ok_or_else(|| invalid("global dictionary rank is past the order"))?;
4936        let block = slot
4937            .get_or_init(|| {
4938                let mut bytes = Vec::new();
4939                self.read_rank_block(rank / TEXT_RANK_BLOCK, &mut bytes)?;
4940                Ok(bytes)
4941            })
4942            .as_ref()
4943            .map_err(Clone::clone)?;
4944        Ok((block.as_slice(), rank % TEXT_RANK_BLOCK))
4945    }
4946
4947    /// Reads block `which` of the sorted order into `bytes`, checked against the hash the index
4948    /// carries for it.
4949    fn read_rank_block(&self, which: usize, bytes: &mut Vec<u8>) -> Result<()> {
4950        let start = if which == 0 { 0 } else { self.rank_ends[which - 1] };
4951        let end = self.rank_ends[which];
4952        bytes.clear();
4953        bytes.resize((end - start) as usize, 0);
4954        read_at(&self.file, self.rank_at + start, bytes)?;
4955        let expected = self
4956            .rank_hashes
4957            .get(which)
4958            .ok_or_else(|| invalid("global dictionary rank block has no checksum"))?;
4959        if checksum(bytes) != *expected {
4960            return Err(invalid("global dictionary rank checksum differs"));
4961        }
4962        Ok(())
4963    }
4964
4965    /// The first eight bytes of the value at `rank`, as the integer a comparison reads.
4966    fn head_at(&self, rank: usize) -> Result<u64> {
4967        let (block, within) = self.rank_parts(rank)?;
4968        let (base, width, packed) = rank_heads(block)?;
4969        let above = bitpack::tail_at(packed, width, within)
4970            .map_err(|_| invalid("global dictionary rank block is short of heads"))?;
4971        Ok(base.wrapping_add(above))
4972    }
4973
4974    /// The packed codes of one rank block, which follow the heads on the next byte boundary.
4975    fn rank_codes<'block>(&self, block: &'block [u8], count: usize) -> Result<&'block [u8]> {
4976        let (_, width, packed) = rank_heads(block)?;
4977        packed
4978            .get(bitpack::tail_len(count, width)..)
4979            .ok_or_else(|| invalid("global dictionary rank block is short of codes"))
4980    }
4981
4982    /// How many entries the block holding `rank` has, which is a full block except at the end.
4983    fn rank_block_len(&self, rank: usize) -> usize {
4984        let first = rank / TEXT_RANK_BLOCK * TEXT_RANK_BLOCK;
4985        TEXT_RANK_BLOCK.min(self.ranks - first)
4986    }
4987}
4988
4989/// The base, the width and the packed bytes of one rank block's heads.
4990fn rank_heads(block: &[u8]) -> Result<(u64, usize, &[u8])> {
4991    let header = block
4992        .get(..RANK_BLOCK_HEADER)
4993        .ok_or_else(|| invalid("global dictionary rank block is short"))?;
4994    let base = u64::from_le_bytes(header[..8].try_into().expect("eight bytes"));
4995    let width = header[8] as usize;
4996    if width > 64 {
4997        return Err(invalid("global dictionary rank block packs heads past a word"));
4998    }
4999    Ok((base, width, &block[RANK_BLOCK_HEADER..]))
5000}
5001
5002/// Bits one offset of a dictionary takes, which is what its widest payload block spans.
5003///
5004/// One width for the whole column rather than one a block. A block is 1,024 values of the same
5005/// column, so the blocks of a column are within a factor of two of each other on every ClickBench
5006/// string column, and a width a block would save a fraction of a bit and cost a byte a block plus
5007/// the arithmetic that finds where a block starts.
5008fn offset_width(ends: &[u32]) -> usize {
5009    // The ends are already relative to the block the value is in, so the last end of a block is that
5010    // block's total and the largest end anywhere is the widest block. There is no subtraction left
5011    // to do and no need to walk the blocks to find where one starts.
5012    let span = ends.iter().copied().max().unwrap_or(0);
5013    (u32::BITS - span.leading_zeros()) as usize
5014}
5015
5016/// How many bytes `values` offsets take at `bits`, which is what the reader has to know before it
5017/// has read any of them.
5018fn offset_bytes(values: usize, bits: usize) -> usize {
5019    let full = values / TEXT_OFFSET_RUN;
5020    let rest = values % TEXT_OFFSET_RUN;
5021    full * TEXT_OFFSET_RUN / 8 * bits + bitpack::tail_len(rest, bits)
5022}
5023
5024/// The end of every value within its payload block, packed a run at a time.
5025/// A run never straddles a block, because [`TEXT_OFFSET_RUN`] divides [`TEXT_PAYLOAD_VALUES`], which
5026/// is what lets this be a walk of the ends rather than arithmetic against a per block base.
5027fn encode_offsets(ends: &[u32], bits: usize, out: &mut Vec<u8>) -> Result<()> {
5028    let mut run = Vec::with_capacity(TEXT_OFFSET_RUN);
5029    for chunk in ends.chunks(TEXT_OFFSET_RUN) {
5030        run.clear();
5031        run.extend(chunk.iter().map(|&end| u64::from(end)));
5032        bitpack::pack_tail(&run, bits, out)
5033            .map_err(|_| invalid("global dictionary offsets do not pack"))?;
5034    }
5035    Ok(())
5036}
5037
5038/// How many bits a code of a dictionary of `values` entries takes.
5039fn code_width(values: usize) -> usize {
5040    match u64::try_from(values).unwrap_or(u64::MAX) {
5041        0 | 1 => 0,
5042        last => (u64::BITS - (last - 1).leading_zeros()) as usize,
5043    }
5044}
5045
5046impl TextSource for NativeText {
5047    fn len(&self) -> usize {
5048        self.values
5049    }
5050
5051    fn might_contain(&self, first: usize, literal: &[u8]) -> Result<bool> {
5052        let Some(grams) = &self.grams else { return Ok(true) };
5053        if literal.len() < 4 || first >= self.values {
5054            return Ok(true);
5055        }
5056        let verdict = grams.verdicts(&self.file, literal)?;
5057        Ok(verdict.get(first / TEXT_PAYLOAD_VALUES).copied().unwrap_or(true))
5058    }
5059
5060    fn bytes_at(&self, index: usize) -> Result<Option<&[u8]>> {
5061        if index >= self.values {
5062            return Ok(None);
5063        }
5064        let (start, end) = self.span_within(index)?;
5065        if start == end {
5066            return Ok(Some(&[]));
5067        }
5068        // A block holds a fixed number of values rather than a fixed number of bytes, so the value
5069        // is in one block and the offsets already say where in it.
5070        let block = index / TEXT_PAYLOAD_VALUES;
5071        let Some(bytes) = self.payload_block(block)? else { return Ok(None) };
5072        Ok(bytes.get(start as usize..end as usize))
5073    }
5074
5075    fn bytes_len_at(&self, index: usize) -> Result<Option<usize>> {
5076        if index >= self.values {
5077            return Ok(None);
5078        }
5079        let (start, end) = self.span_within(index)?;
5080        Ok(Some((end - start) as usize))
5081    }
5082
5083    /// Every length out of the unpacked ends in one loop, which is the point of having them.
5084    ///
5085    /// The whole run of positions counts towards [`Self::ends_worth_unpacking`] at once, because a
5086    /// caller asking for a vector of lengths has said how many it wants, and a vector of them is
5087    /// usually enough on its own. Until the table is worth building this is the row at a time read,
5088    /// the same as the default.
5089    fn bytes_lens_at(&self, indices: &[u32], into: &mut Vec<i64>) -> Result<()> {
5090        self.ends_asked.fetch_add(indices.len(), Atomic::Relaxed);
5091        into.reserve(indices.len());
5092        let Some(ends) = self.value_ends() else {
5093            for &index in indices {
5094                into.push(
5095                    self.bytes_len_at(index as usize)?
5096                        .map_or(0, |len| i64::try_from(len).unwrap_or(i64::MAX)),
5097                );
5098            }
5099            return Ok(());
5100        };
5101        if let Some(lens) = self.value_lens.get_or_init(|| lengths_of(ends)) {
5102            lens.extend_at(indices, into);
5103            return Ok(());
5104        }
5105        for &index in indices {
5106            let index = index as usize;
5107            // Past the end is no value and so no length, which is what a row at a time read says.
5108            let Some(&end) = ends.get(index) else {
5109                into.push(0);
5110                continue;
5111            };
5112            let start = if index % TEXT_PAYLOAD_VALUES == 0 { 0 } else { ends[index - 1] };
5113            if start > end {
5114                return Err(invalid("global dictionary value ends before it starts"));
5115            }
5116            into.push(i64::from(end - start));
5117        }
5118        Ok(())
5119    }
5120
5121    /// Every length in characters out of the counts kept a block at a time, which is what keeps a
5122    /// scan of `length` from holding the column decoded. See [`NativeText::char_lens`].
5123    fn chars_lens_at(&self, indices: &[u32], into: &mut Vec<i64>) -> Result<()> {
5124        into.reserve(indices.len());
5125        for &index in indices {
5126            let index = index as usize;
5127            // Past the end is no value and so no length, which is what a row at a time read says.
5128            if index >= self.values {
5129                into.push(0);
5130                continue;
5131            }
5132            let lens = self.block_chars(index / TEXT_PAYLOAD_VALUES)?;
5133            let len = lens
5134                .get(index % TEXT_PAYLOAD_VALUES)
5135                .ok_or_else(|| invalid("global dictionary block holds the wrong value count"))?;
5136            into.push(i64::from(*len));
5137        }
5138        Ok(())
5139    }
5140
5141    /// The rest of the block holding `first`, decoded into a buffer that may die with the call.
5142    ///
5143    /// A block is the unit this format decodes, so a walk that wants every value is going to decode
5144    /// every block whatever it does. The question is whether it keeps them, and both answers are
5145    /// wrong on their own. [`Self::payload_block`] keeps every block it is asked for, so a reader
5146    /// that walked the whole dictionary through `bytes_at` ended up holding the whole dictionary
5147    /// decoded, 4.2 GB on ClickBench `URL`. Keeping none of them makes the next statement asking
5148    /// the same question decode all of it again, which on the same column at a million rows is a
5149    /// `LIKE` going from 2.7 ms to 16.2 ms.
5150    ///
5151    /// So a sweep keeps what it decodes for the second time while the column is under
5152    /// [`TEXT_KEEP_BUDGET`] and drops it after that. A block already in hand is used where it is there and costs nothing either way.
5153    fn sweep(
5154        &self,
5155        first: usize,
5156        limit: usize,
5157        body: &mut dyn FnMut(usize, &[u8]) -> Result<()>,
5158    ) -> Result<usize> {
5159        let limit = limit.min(self.values);
5160        if first >= limit {
5161            return Ok(first);
5162        }
5163        let block = first / TEXT_PAYLOAD_VALUES;
5164        let last = ((block + 1) * TEXT_PAYLOAD_VALUES).min(limit);
5165        let mut decoded = Vec::new();
5166        let bytes = self.loaned_block(block, &mut decoded, false)?;
5167        let ends = self.ends_within(first, last)?;
5168        if ends.len() != last - first {
5169            return Err(invalid("global dictionary offsets are short"));
5170        }
5171        let mut start = u64::from(self.start_within(first)?);
5172        // row at a time: the caller is handed one value after another, and what it does with one is
5173        // its own business, so there is no shape here for anything but a walk.
5174        for (index, &end) in (first..last).zip(&ends) {
5175            let value = usize::try_from(start)
5176                .ok()
5177                .zip(usize::try_from(end).ok())
5178                .and_then(|(from, to)| bytes.get(from..to))
5179                .ok_or_else(|| invalid("global dictionary value is past its block"))?;
5180            body(index, value)?;
5181            start = end;
5182        }
5183        Ok(last)
5184    }
5185
5186    /// The values at `indices` a block at a time, each block read once for the call.
5187    ///
5188    /// The positions are put in code order first, because the codes of a vector are in row order
5189    /// and land all over the dictionary, and read in that order each block a vector touches would
5190    /// be looked up once for every row in it. Whether a block is kept is
5191    /// [`NativeText::loaned_block`]'s decision, which keeps at most the budget of this column
5192    /// until the reads have shown they come back to the same blocks too often for dropping them to
5193    /// be cheap.
5194    fn visit_at(
5195        &self,
5196        indices: &[u32],
5197        body: &mut dyn FnMut(usize, &[u8]) -> Result<()>,
5198    ) -> Result<()> {
5199        let mut order = (0..indices.len()).collect::<Vec<_>>();
5200        order.sort_unstable_by_key(|&at| indices[at]);
5201        let block_of = |at: usize| {
5202            let index = indices[at] as usize;
5203            (index < self.values).then_some(index / TEXT_PAYLOAD_VALUES)
5204        };
5205        let mut decoded = Vec::new();
5206        let mut run = 0;
5207        while run < order.len() {
5208            let Some(block) = block_of(order[run]) else {
5209                // Past the end is no value, and every position after this one is past it too.
5210                for &at in &order[run..] {
5211                    body(at, &[])?;
5212                }
5213                break;
5214            };
5215            let upto = run + order[run..].partition_point(|&at| block_of(at) == Some(block));
5216            let bytes = self.loaned_block(block, &mut decoded, true)?;
5217            for &at in &order[run..upto] {
5218                let (start, end) = self.span_within(indices[at] as usize)?;
5219                let value = bytes
5220                    .get(start as usize..end as usize)
5221                    .ok_or_else(|| invalid("global dictionary value is past its block"))?;
5222                body(at, value)?;
5223            }
5224            run = upto;
5225        }
5226        Ok(())
5227    }
5228
5229    /// Each block the indices land in, decoded once and dropped, or read where it is already kept.
5230    ///
5231    /// Never kept, unlike [`Self::sweep`] under its budget, because a scattered read is a one off:
5232    /// a synopsis turned into values is turned once and remembered by the reader as values, a few
5233    /// kilobytes, where the blocks it went through are megabytes nobody asks for again.
5234    fn visit(
5235        &self,
5236        indices: &[usize],
5237        body: &mut dyn FnMut(usize, &[u8]) -> Result<()>,
5238    ) -> Result<()> {
5239        let mut at = 0;
5240        while at < indices.len() {
5241            let block = indices[at] / TEXT_PAYLOAD_VALUES;
5242            let upto =
5243                at + indices[at..].partition_point(|&index| index / TEXT_PAYLOAD_VALUES == block);
5244            let wanted = &indices[at..upto];
5245            if wanted.iter().any(|&index| index >= self.values) {
5246                return Err(invalid("a visited value is past the global dictionary"));
5247            }
5248            let decoded;
5249            let bytes: &[u8] = match self.blocks.get(block).and_then(OnceLock::get) {
5250                Some(Ok(kept)) => kept,
5251                _ => {
5252                    decoded = self.decode_block(block)?;
5253                    &decoded
5254                }
5255            };
5256            for (offset, &index) in wanted.iter().enumerate() {
5257                let (start, end) = self.span_within(index)?;
5258                let value = bytes
5259                    .get(start as usize..end as usize)
5260                    .ok_or_else(|| invalid("global dictionary value is past its block"))?;
5261                body(at + offset, value)?;
5262            }
5263            at = upto;
5264        }
5265        Ok(())
5266    }
5267
5268    fn ranks(&self) -> Option<usize> {
5269        (self.ranks > 0).then_some(self.ranks)
5270    }
5271
5272    /// The boundary for `wanted`, out of [`Self::searched`] where it is there and put there where
5273    /// it is not.
5274    ///
5275    /// The lock is held over the search rather than dropped and taken again, so that two threads
5276    /// asking for the same value at the same time do the work once between them. That is the shape
5277    /// the scan actually arrives in: sixteen instances of a top N, all reading the same column, all
5278    /// improving their bound over the same early chunks.
5279    fn below(&self, ranks: usize, wanted: &[u8]) -> Result<(usize, bool)> {
5280        let mut memo = self.searched.lock().map_err(|_| invalid("a poisoned dictionary search"))?;
5281        if let Some(&answer) = memo.get(wanted) {
5282            return Ok(answer);
5283        }
5284        let answer = search_below(self, ranks, wanted)?;
5285        if memo.len() >= TEXT_SEARCH_MEMO {
5286            memo.clear();
5287        }
5288        memo.insert(wanted.to_vec(), answer);
5289        Ok(answer)
5290    }
5291
5292    fn compare_rank(&self, rank: usize, wanted: &[u8]) -> Result<Ordering> {
5293        // The head settles the probe unless the two values start with the same eight bytes, and
5294        // only then is a value read. On a column of URLs that is the difference between a search
5295        // that touches one block of the payload and a search that touches nineteen of them.
5296        let settled = self.head_at(rank)?.cmp(&head(wanted));
5297        if settled != Ordering::Equal {
5298            return Ok(settled);
5299        }
5300        let code = self.code_at_rank(rank)?;
5301        let bytes = self
5302            .bytes_at(code as usize)?
5303            .ok_or_else(|| invalid("global dictionary order names a code it does not have"))?;
5304        Ok(bytes.cmp(wanted))
5305    }
5306
5307    fn code_at_rank(&self, rank: usize) -> Result<u32> {
5308        let (block, within) = self.rank_parts(rank)?;
5309        let codes = self.rank_codes(block, self.rank_block_len(rank))?;
5310        let code = bitpack::tail_at(codes, self.code_bits, within)
5311            .map_err(|_| invalid("global dictionary rank block is short of codes"))?;
5312        let code = u32::try_from(code)
5313            .map_err(|_| invalid("global dictionary order names a code it does not have"))?;
5314        if code as usize >= self.len() {
5315            return Err(invalid("global dictionary order names a code it does not have"));
5316        }
5317        Ok(code)
5318    }
5319
5320    fn code_ranks(&self) -> Option<&[u32]> {
5321        // The order is a permutation of the positions, so inverting it needs every position to be
5322        // named exactly once. Anything else and the slice would have holes, and a caller indexing
5323        // it by a code would read a rank that belongs to nothing.
5324        if self.ranks == 0 || self.ranks != self.len() {
5325            return None;
5326        }
5327        self.code_ranks
5328            .get_or_init(|| {
5329                let mut ranks = vec![u32::MAX; self.ranks];
5330                // A block at a time rather than a rank at a time, because reading it per rank pays
5331                // for the bounds check, the division and the lock on every one of them.
5332                //
5333                // A block nothing has read yet is read into one buffer that is reused, rather than
5334                // through `rank_parts`, which would keep every block of the order once this is
5335                // done with it. The inverse is all anything wants after this, and on the `Referer`
5336                // column of the ClickBench file the blocks are tens of megabytes held for nothing.
5337                let mut scratch = Vec::new();
5338                let mut codes = vec![0u64; TEXT_RANK_BLOCK];
5339                for first in (0..self.ranks).step_by(TEXT_RANK_BLOCK) {
5340                    let which = first / TEXT_RANK_BLOCK;
5341                    let block = match self.rank_blocks.get(which)?.get() {
5342                        Some(kept) => kept.as_ref().ok()?.as_slice(),
5343                        None => {
5344                            self.read_rank_block(which, &mut scratch).ok()?;
5345                            scratch.as_slice()
5346                        }
5347                    };
5348                    let count = self.rank_block_len(first);
5349                    let packed = self.rank_codes(block, count).ok()?;
5350                    let codes = codes.get_mut(..count)?;
5351                    bitpack::unpack_tail_into(packed, self.code_bits, codes, |bits| bits).ok()?;
5352                    for (within, &code) in codes.iter().enumerate() {
5353                        let code = usize::try_from(code).ok()?;
5354                        *ranks.get_mut(code)? = u32::try_from(first + within).ok()?;
5355                    }
5356                }
5357                if ranks.contains(&u32::MAX) {
5358                    return None;
5359                }
5360                Some(ranks)
5361            })
5362            .as_deref()
5363    }
5364
5365    fn footprint(&self) -> usize {
5366        self.offsets.capacity()
5367            + self
5368                .value_ends
5369                .get()
5370                .and_then(Option::as_ref)
5371                .map_or(0, |ends| ends.capacity() * size_of::<u32>())
5372            + self.value_lens.get().and_then(Option::as_ref).map_or(0, Lengths::footprint)
5373            + self
5374                .code_ranks
5375                .get()
5376                .and_then(Option::as_ref)
5377                .map_or(0, |ranks| ranks.capacity() * size_of::<u32>())
5378            + self.rank_hashes.capacity() * size_of::<u64>()
5379            + self.rank_ends.capacity() * size_of::<u64>()
5380            + self.rank_blocks.capacity() * size_of::<OnceLock<Result<Vec<u8>>>>()
5381            + self
5382                .rank_blocks
5383                .iter()
5384                .filter_map(OnceLock::get)
5385                .filter_map(|result| result.as_ref().ok())
5386                .map(Vec::capacity)
5387                .sum::<usize>()
5388            + self.blocks.capacity() * size_of::<OnceLock<Result<Vec<u8>>>>()
5389            + self.char_lens.capacity() * size_of::<OnceLock<Box<[u32]>>>()
5390            + self
5391                .char_lens
5392                .iter()
5393                .filter_map(OnceLock::get)
5394                .map(|lens| lens.len() * size_of::<u32>())
5395                .sum::<usize>()
5396            + self.hashes.capacity() * size_of::<u64>()
5397            + self.starts.capacity() * size_of::<u64>()
5398            + self.lengths.capacity() * size_of::<u64>()
5399            + self.grams.as_ref().map_or(0, NativeGrams::footprint)
5400            + self
5401                .blocks
5402                .iter()
5403                .filter_map(OnceLock::get)
5404                .filter_map(|result| result.as_ref().ok())
5405                .map(Vec::capacity)
5406                .sum::<usize>()
5407    }
5408}
5409
5410/// Every table wide part number in order, with the stripe it belongs to.
5411fn places(table: &Table) -> Result<Vec<Place>> {
5412    let mut places = Vec::with_capacity(table.stripes.len().saturating_mul(STRIPE_PARTS));
5413    for (at, stripe) in table.stripes.iter().enumerate() {
5414        let index = u32::try_from(at).map_err(|_| invalid("too many stripes"))?;
5415        for (part, &rows) in stripe.parts.iter().enumerate() {
5416            places.push(Place {
5417                stripe: index,
5418                part: u32::try_from(part).map_err(|_| invalid("too many parts in a stripe"))?,
5419                rows,
5420            });
5421        }
5422    }
5423    Ok(places)
5424}
5425
5426/// Reads one column's section of a stripe's index page.
5427///
5428/// The section carries its own checksum, so a reader that wants one column out of a hundred and
5429/// five preads a few hundred bytes and still knows that what it got is what was written.
5430fn read_index<F: Positional + ?Sized>(
5431    file: &F,
5432    stripe: &Stripe,
5433    column: usize,
5434) -> Result<Vec<PartSpan>> {
5435    let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
5436    read_index_span(file, stripe.index, *page, stripe.parts.len(), column)
5437}
5438
5439fn read_index_span<F: Positional + ?Sized>(
5440    file: &F,
5441    index: Span,
5442    page: Span,
5443    parts: usize,
5444    column: usize,
5445) -> Result<Vec<PartSpan>> {
5446    let section = index_section(parts)?;
5447    let at = column.checked_mul(section).ok_or_else(|| invalid("index page offset overflow"))?;
5448    let end = at.checked_add(section).ok_or_else(|| invalid("index page offset overflow"))?;
5449    if end > index.length as usize {
5450        return Err(invalid("index page is shorter than its columns"));
5451    }
5452    let mut bytes = vec![0; section];
5453    let offset =
5454        index.offset.checked_add(at as u64).ok_or_else(|| invalid("index page offset overflow"))?;
5455    read_at(file, offset, &mut bytes)?;
5456    let entries = section - size_of::<u64>();
5457    let stored = u64::from_le_bytes(bytes[entries..].try_into().expect("eight bytes"));
5458    if checksum(&bytes[..entries]) != stored {
5459        // With where it was read from, because the two ways this fires look identical from the
5460        // message alone: a file somebody damaged, and a file we wrote to the wrong offset.
5461        return Err(invalid(&format!(
5462            "index page section checksum differs, column {column} of {parts} parts at {offset}, \
5463             wanted {stored:016x} and got {:016x}",
5464            checksum(&bytes[..entries]),
5465        )));
5466    }
5467    let mut spans = Vec::with_capacity(parts);
5468    let mut start = 0_usize;
5469    for part in 0..parts {
5470        let at = part * INDEX_ENTRY;
5471        let length = u32::from_le_bytes(bytes[at..at + 4].try_into().expect("four bytes")) as usize;
5472        let hash = u64::from_le_bytes(bytes[at + 4..at + 12].try_into().expect("eight bytes"));
5473        spans.push(PartSpan { start, length, hash });
5474        start = start.checked_add(length).ok_or_else(|| invalid("column page length overflow"))?;
5475    }
5476    if start != page.length as usize {
5477        return Err(invalid("column page length differs from its index"));
5478    }
5479    Ok(spans)
5480}
5481
5482/// One part's bytes out of a whole column page.
5483fn part_bytes(page: &[u8], span: PartSpan) -> Result<&[u8]> {
5484    let end = span.start.checked_add(span.length).ok_or_else(|| invalid("part range overflow"))?;
5485    page.get(span.start..end).ok_or_else(|| invalid("part exceeds its column page"))
5486}
5487
5488/// Puts one stripe of one column in the cache, and hands back the page for the pool to count when
5489/// it is a page the column did not already hold.
5490///
5491/// The index goes in its own slot and stays. Only the page is under the budget, and the pool is
5492/// what enforces it, once the caller has let go of the column's lock.
5493fn remember(cached: &mut Cached, held: &CachedColumn) -> Option<(usize, Arc<AtomicBool>)> {
5494    if let Some(slot) = cached.index.get_mut(held.stripe) {
5495        if slot.is_none() {
5496            *slot = Some(Arc::clone(&held.index));
5497        }
5498    }
5499    let page = held.page.clone()?;
5500    let slot = cached.pages.get_mut(held.stripe)?;
5501    if slot.is_some() {
5502        return None;
5503    }
5504    let bytes = page.len();
5505    // Set, so that the page a worker has just paid to read is not the one the pass it pays for
5506    // lets go of before the worker has read a part out of it.
5507    let used = Arc::new(AtomicBool::new(true));
5508    *slot = Some(Resident { page, used: Arc::clone(&used) });
5509    Some((bytes, used))
5510}
5511
5512/// Every table a native file holds, without the directory of any of them.
5513///
5514/// This is what opening a database reads. It is the small level of the directory, so the cost is
5515/// proportional to how many tables there are rather than to how much data they hold, and a session
5516/// that touches two tables of eight decodes two table directories.
5517///
5518/// The file handle is shared with every reader this hands out. Eight tables in one file is one open
5519/// file descriptor, not eight, which is the other thing one file buys over a file per table.
5520#[derive(Debug, Clone)]
5521pub struct Catalog {
5522    file: Arc<File>,
5523    size: u64,
5524    entries: Arc<Vec<Entry>>,
5525    /// The views the file holds, whole, since a view has no second level to read later.
5526    views: Arc<Vec<ViewEntry>>,
5527    opening: Opening,
5528    /// Where every reader this hands out counts its pages.
5529    pool: PagePool,
5530}
5531
5532/// Signed integer sums and non-null counts for selected columns, plus total table rows.
5533#[derive(Debug, Clone, PartialEq, Eq)]
5534pub struct CertifiedSums {
5535    pub columns: Vec<(i128, u64)>,
5536    pub rows: u64,
5537}
5538
5539/// Exact ends of an integer or date column, including a certified all-null column.
5540#[derive(Debug, Clone, Copy, PartialEq, Eq)]
5541pub enum IntegerExtremes {
5542    Null,
5543    Values { low: i128, high: i128 },
5544}
5545
5546/// A complete numeric value-to-row-count synopsis; `None` represents SQL NULL.
5547pub type NumericFrequencies = Vec<(Option<i128>, u64)>;
5548
5549impl Catalog {
5550    /// Reads the highest valid catalog slot and nothing under it.
5551    ///
5552    /// The readers it hands out keep pages in a pool of their own with no budget, so each column
5553    /// holds its floor of four stripes and no more. A database opens with [`Catalog::open_in`].
5554    ///
5555    /// # Errors
5556    ///
5557    /// If the file has no valid committed catalog or a catalog pointer is out of bounds.
5558    pub fn open(path: impl AsRef<Path>) -> Result<Self> {
5559        Self::open_in(path, &PagePool::default())
5560    }
5561
5562    /// The same, with every reader it hands out keeping its pages in `pool`.
5563    ///
5564    /// # Errors
5565    ///
5566    /// If the file has no valid committed catalog or a catalog pointer is out of bounds.
5567    pub fn open_in(path: impl AsRef<Path>, pool: &PagePool) -> Result<Self> {
5568        let (file, size, _, bytes, opening) = slot_bytes(path)?;
5569        let (entries, views) = decode_catalog(&bytes, size)?;
5570        Ok(Self {
5571            file: Arc::new(file),
5572            size,
5573            entries: Arc::new(entries),
5574            views: Arc::new(views),
5575            opening,
5576            pool: pool.clone(),
5577        })
5578    }
5579
5580    /// The tables in the file, in the order they were written.
5581    pub fn names(&self) -> impl ExactSizeIterator<Item = &str> {
5582        self.entries.iter().map(|entry| entry.name.as_str())
5583    }
5584
5585    /// The same tables with how many rows each of them holds.
5586    ///
5587    /// The names alone answer which tables the file has, which is what a checkpoint needs to know.
5588    /// A load asks a second question: whether a table already in the file is really in the way of
5589    /// the one it wants to write. A table with no rows is not, because it has no pages the next
5590    /// generation would have to carry, so the count has to come out of the catalog beside the name.
5591    pub fn rows(&self) -> impl ExactSizeIterator<Item = (&str, usize)> {
5592        self.entries.iter().map(|entry| (entry.name.as_str(), entry.rows))
5593    }
5594
5595    /// The views in the file, in the order they were written.
5596    ///
5597    /// Whole, unlike [`Catalog::names`], which hands back names and makes the caller ask for a table
5598    /// by one. A view is a few strings and a column list and it was all read at open, so there is
5599    /// nothing left to go and fetch and no reason to make the caller ask twice.
5600    pub fn views(&self) -> impl ExactSizeIterator<Item = &ViewEntry> {
5601        self.views.iter()
5602    }
5603
5604    /// How many tables the file holds.
5605    #[must_use]
5606    pub fn len(&self) -> usize {
5607        self.entries.len()
5608    }
5609
5610    /// Whether the file holds no table at all, which is what [`Writer::empty`] writes and what a
5611    /// database somebody dropped the last table out of comes back as.
5612    #[must_use]
5613    pub fn is_empty(&self) -> bool {
5614        self.entries.is_empty()
5615    }
5616
5617    /// Opens one table by name, decoding its directory now.
5618    ///
5619    /// # Errors
5620    ///
5621    /// If there is no table by that name, or its directory is torn or points outside the file.
5622    pub fn table(&self, name: &str) -> Result<Reader> {
5623        let entry = self
5624            .entries
5625            .iter()
5626            .find(|entry| entry.name == name)
5627            .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
5628        // Checked and then decoded a window at a time, so that the directory's own bytes are never
5629        // all in memory beside the table they decode into. It is read twice, and the second read
5630        // comes out of the page cache the first one filled.
5631        let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
5632        if file_checksum(&self.file, offset, length)? != entry.directory.hash {
5633            return Err(invalid(&format!("the directory of table {name} does not checksum")));
5634        }
5635        let mut opening = self.opening;
5636        opening.reads += 1;
5637        opening.bytes += u64::from(entry.directory.length);
5638        Reader::build(
5639            Arc::clone(&self.file),
5640            self.size,
5641            read_directory(Cursor::over(&self.file, offset, length), self.size, Some(offset))?,
5642            u64::from(entry.directory.length),
5643            opening,
5644            self.pool.clone(),
5645        )
5646    }
5647
5648    /// Counts one signed integer column from its encoded parts without building metadata for
5649    /// unrelated columns. The counts are computed from row encodings when this is called.
5650    /// Nullable and non-cascade parts use the ordinary decoder for that part.
5651    ///
5652    /// # Errors
5653    ///
5654    /// If the directory, selected page index, checksum, or encoded integer is invalid.
5655    pub fn integer_tally(&self, name: &str, column: usize) -> Result<Option<Vec<(i64, u64)>>> {
5656        let mut counts = BTreeMap::<i64, u64>::new();
5657        let Some(()) = self.integer_fold(name, column, |value, count| {
5658            let held = counts.entry(value).or_default();
5659            *held = held.checked_add(count).ok_or_else(|| invalid("integer count overflow"))?;
5660            Ok(())
5661        })?
5662        else {
5663            return Ok(None);
5664        };
5665        Ok(Some(counts.into_iter().collect()))
5666    }
5667
5668    /// Visits a signed integer column's row values without building per-part or table-wide count
5669    /// maps. The caller combines the emitted counts for its query at runtime.
5670    ///
5671    /// # Errors
5672    ///
5673    /// If the selected file data is invalid or the callback rejects a count.
5674    pub fn integer_fold(
5675        &self,
5676        name: &str,
5677        column: usize,
5678        mut emit: impl FnMut(i64, u64) -> Result<()>,
5679    ) -> Result<Option<()>> {
5680        let entry = self
5681            .entries
5682            .iter()
5683            .find(|entry| entry.name == name)
5684            .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
5685        let field =
5686            entry.fields.get(column).ok_or_else(|| invalid("integer column index out of range"))?;
5687        if !signed_integer(&field.ty) {
5688            return Ok(None);
5689        }
5690        let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
5691        if file_checksum(&self.file, offset, length)? != entry.directory.hash {
5692            return Err(invalid(&format!("the directory of table {name} does not checksum")));
5693        }
5694        quick_integer_fold(
5695            &self.file,
5696            Cursor::over(&self.file, offset, length),
5697            entry,
5698            self.size,
5699            column,
5700            &mut emit,
5701        )?;
5702        Ok(Some(()))
5703    }
5704
5705    /// Counts non-null, nonzero values from generic column frequencies when complete. For an
5706    /// older file or a partial catalog synopsis, reads the validated native directory without
5707    /// building a reader for every stripe. Returns `None` when the bounded frequency synopsis
5708    /// cannot prove the count, so callers can use the ordinary query path.
5709    pub fn nonzero_count(&self, name: &str, column: usize) -> Result<Option<u64>> {
5710        let entry = self
5711            .entries
5712            .iter()
5713            .find(|entry| entry.name == name)
5714            .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
5715        let Some(field) = entry.fields.get(column) else {
5716            return Err(invalid("frequency column index out of range"));
5717        };
5718        if !matches!(
5719            field.ty,
5720            LogicalType::TinyInt
5721                | LogicalType::SmallInt
5722                | LogicalType::Integer
5723                | LogicalType::BigInt
5724                | LogicalType::UTinyInt
5725                | LogicalType::USmallInt
5726                | LogicalType::UInteger
5727                | LogicalType::UBigInt
5728        ) {
5729            return Ok(None);
5730        }
5731        let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
5732        if file_checksum(&self.file, offset, length)? != entry.directory.hash {
5733            return Err(invalid(&format!("the directory of table {name} does not checksum")));
5734        }
5735        if let Some(Some(frequencies)) = entry.frequencies.get(column) {
5736            return frequencies
5737                .iter()
5738                .filter(|(value, _)| value.is_some_and(|value| value != 0))
5739                .try_fold(0_u64, |total, (_, count)| total.checked_add(*count))
5740                .map(Some)
5741                .ok_or_else(|| invalid("numeric frequency count overflow"));
5742        }
5743        quick_nonzero(
5744            Cursor::over(&self.file, offset, length),
5745            &entry.name,
5746            &entry.fields,
5747            entry.rows,
5748            column,
5749        )
5750    }
5751
5752    /// Exact signed-integer sums and non-null counts from the small catalog. The table directory
5753    /// checksum is still checked once before any certificate can answer a query.
5754    pub fn aggregate_sums(&self, name: &str, columns: &[usize]) -> Result<Option<CertifiedSums>> {
5755        let entry = self
5756            .entries
5757            .iter()
5758            .find(|entry| entry.name == name)
5759            .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
5760        let mut sums = Vec::with_capacity(columns.len());
5761        for &column in columns {
5762            let Some(field) = entry.fields.get(column) else {
5763                return Err(invalid("aggregate column index out of range"));
5764            };
5765            if !signed_integer(&field.ty) {
5766                return Ok(None);
5767            }
5768            let Some(sum) = entry.aggregates[column] else {
5769                return Ok(None);
5770            };
5771            sums.push(sum);
5772        }
5773        let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
5774        if file_checksum(&self.file, offset, length)? != entry.directory.hash {
5775            return Err(invalid(&format!("the directory of table {name} does not checksum")));
5776        }
5777        Ok(Some(CertifiedSums { columns: sums, rows: entry.rows as u64 }))
5778    }
5779
5780    /// Exact non-null distinct count from the small catalog, after checking the table directory.
5781    pub fn distinct_count(&self, name: &str, column: usize) -> Result<Option<u64>> {
5782        let entry = self
5783            .entries
5784            .iter()
5785            .find(|entry| entry.name == name)
5786            .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
5787        let Some(count) = entry.distincts.get(column).copied() else {
5788            return Err(invalid("distinct column index out of range"));
5789        };
5790        let Some(count) = count else { return Ok(None) };
5791        let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
5792        if file_checksum(&self.file, offset, length)? != entry.directory.hash {
5793            return Err(invalid(&format!("the directory of table {name} does not checksum")));
5794        }
5795        Ok(Some(count))
5796    }
5797
5798    /// Exact integer or date ends from the small catalog after checking the table directory.
5799    pub fn integer_extremes(&self, name: &str, column: usize) -> Result<Option<IntegerExtremes>> {
5800        let entry = self
5801            .entries
5802            .iter()
5803            .find(|entry| entry.name == name)
5804            .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
5805        let Some(extremes) = entry.extremes.get(column).copied() else {
5806            return Err(invalid("extremes column index out of range"));
5807        };
5808        let Some(extremes) = extremes else { return Ok(None) };
5809        let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
5810        if file_checksum(&self.file, offset, length)? != entry.directory.hash {
5811            return Err(invalid(&format!("the directory of table {name} does not checksum")));
5812        }
5813        Ok(Some(match extremes {
5814            None => IntegerExtremes::Null,
5815            Some((low, high)) => IntegerExtremes::Values { low, high },
5816        }))
5817    }
5818
5819    /// Complete numeric frequencies from the small catalog, after checking the table directory.
5820    pub fn exact_numeric_frequencies(
5821        &self,
5822        name: &str,
5823        column: usize,
5824    ) -> Result<Option<NumericFrequencies>> {
5825        let entry = self
5826            .entries
5827            .iter()
5828            .find(|entry| entry.name == name)
5829            .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
5830        let Some(frequencies) = entry.frequencies.get(column).cloned() else {
5831            return Err(invalid("numeric frequency column index out of range"));
5832        };
5833        let Some(frequencies) = frequencies else { return Ok(None) };
5834        let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
5835        if file_checksum(&self.file, offset, length)? != entry.directory.hash {
5836            return Err(invalid(&format!("the directory of table {name} does not checksum")));
5837        }
5838        Ok(Some(frequencies))
5839    }
5840
5841    /// The schema copied into the small file catalog, available without opening the table directory.
5842    pub fn table_fields(&self, name: &str) -> Option<&[Field]> {
5843        self.entries.iter().find(|entry| entry.name == name).map(|entry| entry.fields.as_slice())
5844    }
5845}
5846
5847/// Where the slot naming `generation` goes, which is the one the generation before it did not use.
5848///
5849/// Generation 1 takes the slot at 16, so a file written once is byte for byte the file this wrote
5850/// before there was a second generation to write.
5851fn slot_offset(generation: u64) -> u64 {
5852    16 + (generation - 1) % 2 * SLOT_BYTES as u64
5853}
5854
5855/// The header and the bytes the highest valid slot points at.
5856///
5857/// Both levels of the directory are reached this way, so the magic check, the version check and the
5858/// choice between the two slots live here rather than being written out twice.
5859fn slot_bytes(path: impl AsRef<Path>) -> Result<(File, u64, Slot, Vec<u8>, Opening)> {
5860    let file = File::open(path).map_err(io)?;
5861    let size = file.metadata().map_err(io)?.len();
5862    let (slot, bytes, opening) = committed_slot(&file, size)?;
5863    Ok((file, size, slot, bytes, opening))
5864}
5865
5866/// The committed slot of a file that is `size` bytes long, and the catalog it points at.
5867///
5868/// The half of [`slot_bytes`] that does not care how the file was opened. A reader comes here with
5869/// the `std::fs::File` it goes on to share between its threads, and a writer with the `rudb_io`
5870/// file it is about to append to.
5871fn committed_slot<F: Positional + ?Sized>(file: &F, size: u64) -> Result<(Slot, Vec<u8>, Opening)> {
5872    if size < HEADER {
5873        return Err(invalid("file is shorter than its header"));
5874    }
5875    let mut header = [0; HEADER as usize];
5876    read_at(file, 0, &mut header)?;
5877    let mut opening = Opening { reads: 1, bytes: HEADER };
5878    let version = u32::from_le_bytes([header[8], header[9], header[10], header[11]]);
5879    // The two halves are worth telling apart. A wrong magic is a file that was never ours and
5880    // the answer is to look at the path. A wrong version is our own file from another build,
5881    // and the number this build wants is the only thing that tells the reader whether to
5882    // rebuild the file or to go back to the binary that wrote it.
5883    if &header[..8] != MAGIC {
5884        return Err(invalid("the header does not begin with a rudb native magic"));
5885    }
5886    if !READABLE.contains(&version) {
5887        return Err(invalid(&format!(
5888            "the file is format {version} and this build reads format {FORMAT}, so it has to \
5889                 be written again"
5890        )));
5891    }
5892    let mut selected = None;
5893    for start in [16, 16 + SLOT_BYTES] {
5894        let slot = Slot::read(&header[start..start + SLOT_BYTES]);
5895        if slot.generation == 0 || slot.length == 0 || slot.length as usize > MAX_DIRECTORY {
5896            continue;
5897        }
5898        let Some(end) = slot.offset.checked_add(u64::from(slot.length)) else { continue };
5899        if slot.offset < HEADER || end > size {
5900            continue;
5901        }
5902        let mut bytes = vec![0; slot.length as usize];
5903        read_at(file, slot.offset, &mut bytes)?;
5904        opening.reads += 1;
5905        opening.bytes += u64::from(slot.length);
5906        if checksum(&bytes) == slot.hash
5907            && selected
5908                .as_ref()
5909                .is_none_or(|(old, _): &(Slot, Vec<u8>)| old.generation < slot.generation)
5910        {
5911            selected = Some((slot, bytes));
5912        }
5913    }
5914    let (slot, bytes) = selected.ok_or_else(|| invalid("no committed directory slot is valid"))?;
5915    Ok((slot, bytes, opening))
5916}
5917
5918impl Reader {
5919    /// Opens a file that holds exactly one table.
5920    ///
5921    /// # Errors
5922    ///
5923    /// If the file has no valid committed directory, a directory pointer is out of bounds, or the
5924    /// file holds more than one table, which is a file that has to be opened by name.
5925    pub fn open(path: impl AsRef<Path>) -> Result<Self> {
5926        let catalog = Catalog::open(path)?;
5927        let mut names = catalog.names();
5928        let name = names.next().ok_or_else(|| invalid("the file holds no table"))?.to_string();
5929        if names.next().is_some() {
5930            return Err(invalid(
5931                "the file holds more than one table, so it has to be opened by name",
5932            ));
5933        }
5934        catalog.table(&name)
5935    }
5936
5937    /// Builds a reader over one decoded table directory.
5938    fn build(
5939        file: Arc<File>,
5940        size: u64,
5941        table: Table,
5942        directory: u64,
5943        opening: Opening,
5944        pool: PagePool,
5945    ) -> Result<Self> {
5946        let places = places(&table)?;
5947        let dictionaries = (0..table.fields.len()).map(|_| OnceLock::new()).collect();
5948        let table_fields = table.fields.len();
5949        let stripes = table.stripes.len();
5950        let columns = (0..table.fields.len())
5951            .map(|_| {
5952                Mutex::new(Cached {
5953                    pages: (0..stripes).map(|_| None).collect(),
5954                    index: (0..stripes).map(|_| None).collect(),
5955                    ..Cached::default()
5956                })
5957            })
5958            .collect::<Vec<_>>();
5959        let cache = Shelf {
5960            columns,
5961            held: (0..table_fields).map(|_| AtomicUsize::new(0)).collect(),
5962            kept: AtomicUsize::new(CACHED_STRIPES_PER_COLUMN),
5963        };
5964        let sieves: Vec<Vec<SieveSlot>> = (0..table.fields.len())
5965            .map(|_| table.stripes.iter().map(|_| OnceLock::new()).collect())
5966            .collect();
5967        let part_ranges: Vec<Vec<RangeSlot>> = (0..table.fields.len())
5968            .map(|_| table.stripes.iter().map(|_| OnceLock::new()).collect())
5969            .collect();
5970        Ok(Self {
5971            file,
5972            table: Arc::new(table),
5973            dictionaries: Arc::new(dictionaries),
5974            loading: Arc::new((0..table_fields).map(|_| Mutex::new(())).collect()),
5975            frequency_values: Arc::new((0..table_fields).map(|_| OnceLock::new()).collect()),
5976            frequency_summaries: Arc::new((0..table_fields).map(|_| OnceLock::new()).collect()),
5977            opened: Arc::new(AtomicUsize::new(0)),
5978            sieves: Arc::new(sieves),
5979            part_ranges: Arc::new(part_ranges),
5980            places: Arc::new(places),
5981            cache: Arc::new(cache),
5982            pool,
5983            pages: Arc::new(AtomicUsize::new(0)),
5984            indexes: Arc::new(AtomicUsize::new(0)),
5985            size,
5986            directory,
5987            opening,
5988        })
5989    }
5990
5991    /// What this reader has read so far, and what opening it cost.
5992    ///
5993    /// Public because the claim of `spec/stats/04-in-memory.md` section 4.2 is about this number
5994    /// and a claim nobody can check is a comment. A caller that wants to know whether opening a
5995    /// file touched the data asks here, and gets an answer that does not depend on what the page
5996    /// cache happened to hold.
5997    #[must_use]
5998    pub fn reads(&self) -> Reads {
5999        Reads {
6000            opening: self.opening,
6001            pages: self.pages.load(Atomic::Relaxed),
6002            indexes: self.indexes.load(Atomic::Relaxed),
6003            dictionaries: self.opened.load(Atomic::Relaxed),
6004        }
6005    }
6006
6007    /// Where the file's bytes went, from the directory alone.
6008    ///
6009    /// No page is read, so this costs the same on a 45 GB table as on an empty one. See [`Layout`]
6010    /// for what is charged where and for why the three things that are not columns stay separate.
6011    #[must_use]
6012    pub fn layout(&self) -> Layout {
6013        let table = &self.table;
6014        let stripes = table.stripes.as_slice();
6015        let columns = table
6016            .fields
6017            .iter()
6018            .enumerate()
6019            .map(|(at, field)| ColumnLayout {
6020                name: field.name.clone(),
6021                kind: field.ty.to_string(),
6022                pages: sum(stripes.iter().map(|stripe| span_bytes(&stripe.pages, at))),
6023                memberships: sum(stripes.iter().map(|stripe| stripe.memberships.bytes(at))),
6024                sieves: sum(stripes.iter().map(|stripe| stripe.sieves.bytes(at))),
6025                part_ranges: sum(stripes.iter().map(|stripe| stripe.part_ranges.bytes(at))),
6026                dictionary: dictionary_bytes(table, at),
6027            })
6028            .collect();
6029        Layout {
6030            file: self.size,
6031            rows: table.rows,
6032            stripes: stripes.len(),
6033            parts: self.places.len(),
6034            columns,
6035            indexes: sum(stripes.iter().map(|stripe| u64::from(stripe.index.length))),
6036            directory: self.directory,
6037            header: HEADER,
6038        }
6039    }
6040
6041    /// What every part of one column is stored as, which is what `pragma_storage_info` reports.
6042    ///
6043    /// Unlike [`Self::layout`] this reads the data, because the encoder's choice is in the page and
6044    /// nowhere else. The directory says how many bytes a column took and says nothing about what
6045    /// shape they are in, and the shape is the question worth asking: the same rows in a different
6046    /// order come back bit packed on one file and plain on another, and that is the difference a
6047    /// clustered load makes to a scan.
6048    ///
6049    /// One read per stripe rather than one per part. A part is a few kilobytes out of a page that
6050    /// is a quarter of a megabyte, so asking part by part would read the same page sixty four
6051    /// times. Nothing is put in the page cache, because a caller asking what a file looks like is
6052    /// not about to scan it and evicting the pages a real query wants would be a poor trade.
6053    ///
6054    /// # Errors
6055    ///
6056    /// If the column is outside the schema, or a page, index section or checksum is invalid.
6057    pub fn stored(&self, column: usize) -> Result<Vec<StoredPart>> {
6058        let field = self
6059            .table
6060            .fields
6061            .get(column)
6062            .ok_or_else(|| invalid("stored column index out of range"))?;
6063        let mut stored = Vec::with_capacity(self.places.len());
6064        let mut row = 0;
6065        for (at, stripe) in self.table.stripes.iter().enumerate() {
6066            let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
6067            let index = read_index(&self.file, stripe, column)?;
6068            let mut bytes = vec![0; page.length as usize];
6069            read_at(&self.file, page.offset, &mut bytes)?;
6070            let ranges = self.stripe_part_ranges(at, column);
6071            for (part, &rows) in stripe.parts.iter().enumerate() {
6072                let span = *index.get(part).ok_or_else(|| invalid("part index out of range"))?;
6073                let held = part_bytes(&bytes, span)?;
6074                let range = ranges.and_then(|held| held.get(part));
6075                stored.push(StoredPart {
6076                    stripe: at,
6077                    part,
6078                    row,
6079                    rows: rows as usize,
6080                    encoding: page_encoding(&field.ty, rows as usize, held),
6081                    bytes: span.length as u64,
6082                    page: page.offset,
6083                    offset: span.start as u64,
6084                    low: range
6085                        .and_then(|range| range.low.clone())
6086                        .and_then(|bound| bound.into_value(&field.ty)),
6087                    high: range
6088                        .and_then(|range| range.high.clone())
6089                        .and_then(|bound| bound.into_value(&field.ty)),
6090                    nulls: range.map(|range| range.nulls),
6091                });
6092                row += rows as usize;
6093            }
6094        }
6095        Ok(stored)
6096    }
6097
6098    /// How many parts the table has, which is how many chunks a scan of it reads.
6099    #[must_use]
6100    pub fn parts(&self) -> usize {
6101        self.places.len()
6102    }
6103
6104    /// The parts of each stripe, in table wide part numbers.
6105    ///
6106    /// A scan that wants one worker to own the page it reads hands work out in these runs. The
6107    /// stripes are contiguous in part numbering and all but the last hold sixty four parts, but a
6108    /// stripe can be flushed early when rows arrive out of order, so the runs are read off the
6109    /// directory rather than worked out from a constant.
6110    #[must_use]
6111    pub fn stripe_parts(&self) -> Vec<std::ops::Range<usize>> {
6112        let mut runs = Vec::with_capacity(self.table.stripes.len());
6113        let mut start = 0;
6114        for stripe in &self.table.stripes {
6115            let end = start + stripe.parts.len();
6116            runs.push(start..end);
6117            start = end;
6118        }
6119        runs
6120    }
6121
6122    /// How many rows one stripe holds, in the numbering [`Self::stripe_parts`] hands back.
6123    ///
6124    /// Off the directory, which is already in memory, rather than by the caller asking for each
6125    /// part in turn through the catalog. Nothing past the end holds any rows.
6126    #[must_use]
6127    pub fn stripe_rows(&self, stripe: usize) -> usize {
6128        self.table.stripes.get(stripe).map_or(0, |held| held.rows)
6129    }
6130
6131    /// Asks the page cache to keep `stripes` stripes of every column instead of the default.
6132    ///
6133    /// This only ever raises the number. A scan that gives each worker a whole stripe has one page
6134    /// per column per worker open at once, and a cache smaller than that is worse than no cache at
6135    /// all: every worker's page is evicted by the others before it has finished its stripe, so it
6136    /// reads a quarter of a megabyte for every part it takes out of it.
6137    pub fn keep_stripes(&self, stripes: usize) {
6138        self.cache.kept.fetch_max(stripes, Atomic::Relaxed);
6139    }
6140
6141    /// Rows in one part, or zero when the part number is past the table.
6142    #[must_use]
6143    pub fn part_rows(&self, at: usize) -> usize {
6144        self.places.get(at).map_or(0, |place| place.rows as usize)
6145    }
6146
6147    /// The committed table directory.
6148    #[must_use]
6149    pub fn table(&self) -> &Table {
6150        &self.table
6151    }
6152
6153    /// Exact leading frequencies when the stored synopsis proves a count-descending prefix.
6154    ///
6155    /// The returned list can be longer than `top`. Keeping the stored tail lets a later TopN apply
6156    /// additional ordering keys without losing a value tied with the requested boundary.
6157    ///
6158    /// # Errors
6159    ///
6160    /// If the column is outside the schema or a stored value does not fit its declared type.
6161    pub fn top_frequencies(&self, column: usize, top: usize) -> Result<Option<Vec<(Value, u64)>>> {
6162        let field = self
6163            .table
6164            .fields
6165            .get(column)
6166            .ok_or_else(|| invalid("frequency column index out of range"))?;
6167        let Some(summary) = self.frequency_summary(column)? else {
6168            return Ok(None);
6169        };
6170        if top == 0 || summary.entries.len() < top {
6171            return Ok(None);
6172        }
6173        let boundary = summary.entries[top - 1].count;
6174        if boundary <= summary.omitted_max {
6175            return Ok(None);
6176        }
6177        self.decode_frequencies(column, &field.ty, &summary.entries).map(Some)
6178    }
6179
6180    /// Exact leading counts for a numeric key paired with a stable-dictionary string key.
6181    ///
6182    /// The stored prefix is returned only when its requested boundary strictly beats the bound on
6183    /// every pair omitted at load time. The returned tail may be longer than `top`, as with
6184    /// [`Self::top_frequencies`], so downstream ordering can settle ties without reading rows.
6185    ///
6186    /// # Errors
6187    ///
6188    /// If either column is outside the schema or persisted pair metadata is inconsistent with the
6189    /// frequency synopsis or dictionary it names.
6190    pub fn top_pair_frequencies(
6191        &self,
6192        first: usize,
6193        second: usize,
6194        top: usize,
6195    ) -> Result<Option<PairFrequencyCounts>> {
6196        if first >= self.table.fields.len() || second >= self.table.fields.len() {
6197            return Err(invalid("pair frequency column index out of range"));
6198        }
6199        let Some(summary) =
6200            self.table.pair_frequencies.iter().find(|summary| {
6201                summary.first as usize == first && summary.second as usize == second
6202            })
6203        else {
6204            return Ok(None);
6205        };
6206        if top == 0 || summary.entries.len() < top {
6207            return Ok(None);
6208        }
6209        let boundary = summary.entries[top - 1].count;
6210        if boundary <= summary.omitted_max {
6211            return Ok(None);
6212        }
6213        let first_summary = self
6214            .frequency_summary(first)?
6215            .ok_or_else(|| invalid("pair frequency first column has no synopsis"))?;
6216        let anchors = self
6217            .decode_frequencies(first, &self.table.fields[first].ty, &first_summary.entries)?
6218            .into_iter()
6219            .map(|(value, _)| value)
6220            .collect::<Vec<_>>();
6221        let dictionary = self
6222            .dictionary(second)?
6223            .ok_or_else(|| invalid("pair frequency second column has no dictionary"))?;
6224        let mut codes = summary.entries.iter().filter_map(|entry| entry.second).collect::<Vec<_>>();
6225        codes.sort_unstable();
6226        codes.dedup();
6227        let texts = dictionary
6228            .try_values_visited(&codes.iter().map(|&code| code as usize).collect::<Vec<_>>())?;
6229        let mut out = Vec::with_capacity(summary.entries.len());
6230        for entry in &summary.entries {
6231            if entry.count < boundary {
6232                break;
6233            }
6234            let first = anchors
6235                .get(entry.first_entry as usize)
6236                .cloned()
6237                .ok_or_else(|| invalid("pair frequency anchor is outside its values"))?;
6238            let second = match entry.second {
6239                None => Value::Null,
6240                Some(code) => {
6241                    let at = codes
6242                        .binary_search(&code)
6243                        .map_err(|_| invalid("pair frequency code was not among the codes read"))?;
6244                    texts[at].clone()
6245                }
6246            };
6247            out.push((vec![first, second], entry.count));
6248        }
6249        Ok(Some(out))
6250    }
6251
6252    /// Every value of one column with the number of rows holding it, when the synopsis is complete.
6253    ///
6254    /// The heavy hitter pass keeps a bounded set of candidates and decrements them all when it runs
6255    /// out of room, so what it usually ends with is the leading values and a bound on everything it
6256    /// dropped. `omitted_max` of zero says that never happened: no candidate was ever decremented and
6257    /// the entries did not overflow the stored budget, so the list is every distinct value of the
6258    /// column with an exact count, and a null counts as a value of its own rather than being skipped.
6259    ///
6260    /// That makes a whole class of question answerable without reading a row. How many rows hold a
6261    /// value, how many do not, and what a `GROUP BY` of that column with a count over it produces are
6262    /// all in here. It is only ever true of a column with few enough distinct values, which is the
6263    /// case worth having, because that is exactly the column a grouping or an equality filter would
6264    /// otherwise walk every row to answer.
6265    ///
6266    /// `None` when the column has no synopsis, or has one that dropped anything.
6267    ///
6268    /// # Errors
6269    ///
6270    /// If the column is outside the schema or a stored value does not fit its declared type.
6271    pub fn exact_frequencies(&self, column: usize) -> Result<Option<Vec<(Value, u64)>>> {
6272        let Some(prefix) = self.frequency_prefix(column)? else {
6273            return Ok(None);
6274        };
6275        Ok((prefix.omitted_max == 0).then_some(prefix.entries))
6276    }
6277
6278    /// Every value the synopsis lists with the number of rows holding it, and a bound on the rest.
6279    ///
6280    /// The counts are exact whether or not the list is complete. The heavy hitter pass keeps a
6281    /// bounded candidate set and then recounts only the candidates that survived it, so a value that
6282    /// made it into the list carries the number of rows that really hold it rather than whatever the
6283    /// pass had left over. What the pass loses is values, not counts.
6284    ///
6285    /// `omitted_max` is how many rows the most common value left out can hold, and zero says nothing
6286    /// was left out at all, which is what [`exact_frequencies`] asks for. Above zero the list is the
6287    /// leading values of the column and everything else is somewhere between no rows and that bound.
6288    ///
6289    /// That prefix is worth reading on its own. A column with a value in half its rows and a long
6290    /// tail behind it has no complete synopsis and never will, and it is the column where dividing
6291    /// the rows by the distinct count is furthest from the truth.
6292    ///
6293    /// `None` when the column has no synopsis.
6294    ///
6295    /// # Errors
6296    ///
6297    /// If the column is outside the schema or a stored value does not fit its declared type.
6298    ///
6299    /// [`exact_frequencies`]: Self::exact_frequencies
6300    pub fn frequency_prefix(&self, column: usize) -> Result<Option<FrequencyPrefix>> {
6301        let field = self
6302            .table
6303            .fields
6304            .get(column)
6305            .ok_or_else(|| invalid("frequency column index out of range"))?;
6306        let Some(summary) = self.frequency_summary(column)? else {
6307            return Ok(None);
6308        };
6309        let entries = self.decode_frequencies(column, &field.ty, &summary.entries)?;
6310        Ok(Some(FrequencyPrefix { entries, omitted_max: summary.omitted_max }))
6311    }
6312
6313    /// One column's synopsis, read back from the file when the directory left it there.
6314    fn frequency_summary(&self, column: usize) -> Result<Option<Cow<'_, FrequencySummary>>> {
6315        Ok(match self.table.frequencies.get(column) {
6316            None | Some(None) => None,
6317            Some(Some(Frequencies::Held(summary))) => Some(Cow::Borrowed(summary)),
6318            Some(Some(Frequencies::Stored { span, values })) => {
6319                let slot = self
6320                    .frequency_summaries
6321                    .get(column)
6322                    .ok_or_else(|| invalid("frequency column index out of range"))?;
6323                if let Some(summary) = slot.get() {
6324                    return Ok(Some(Cow::Borrowed(summary.as_ref())));
6325                }
6326                let field = self
6327                    .table
6328                    .fields
6329                    .get(column)
6330                    .ok_or_else(|| invalid("frequency column index out of range"))?;
6331                let mut bytes = vec![0; span.length as usize];
6332                read_at(&self.file, span.offset, &mut bytes)?;
6333                let summary =
6334                    decode_summary(&mut Cursor::new(&bytes), field, self.table.rows, *values)?;
6335                let summary = summary.ok_or_else(|| invalid("a stored synopsis is missing"))?;
6336                let _ = slot.set(Arc::new(summary));
6337                Some(Cow::Borrowed(slot.get().expect("the decoded summary was stored").as_ref()))
6338            }
6339        })
6340    }
6341
6342    /// Turns stored frequency entries into values of the column's own type.
6343    ///
6344    /// Remembered per column, because the planner asks once for every estimate that touches the
6345    /// column and the executor asks again, and the answer is a few hundred values. The codes of a
6346    /// string column are read through [`Vector::try_values_visited`], which does not keep the blocks
6347    /// it decodes, so what a query answered out of the synopsis holds is those values and not the
6348    /// hundred or so dictionary blocks they are scattered over.
6349    fn decode_frequencies(
6350        &self,
6351        column: usize,
6352        ty: &LogicalType,
6353        entries: &[FrequencyEntry],
6354    ) -> Result<Vec<(Value, u64)>> {
6355        if let Some(values) = self.frequency_values.get(column).and_then(OnceLock::get) {
6356            return Ok(values.as_ref().clone());
6357        }
6358        let values = self.decode_frequencies_once(column, ty, entries)?;
6359        if let Some(slot) = self.frequency_values.get(column) {
6360            let _ = slot.set(Arc::new(values.clone()));
6361        }
6362        Ok(values)
6363    }
6364
6365    fn decode_frequencies_once(
6366        &self,
6367        column: usize,
6368        ty: &LogicalType,
6369        entries: &[FrequencyEntry],
6370    ) -> Result<Vec<(Value, u64)>> {
6371        let stored_texts = self.table.frequency_texts.get(column).filter(|texts| !texts.is_empty());
6372        if stored_texts.is_some_and(|texts| texts.len() != entries.len()) {
6373            return Err(invalid("frequency text count differs from its synopsis"));
6374        }
6375        let dictionary =
6376            if coded_type(ty) && stored_texts.is_none() { self.dictionary(column)? } else { None };
6377        let mut codes = entries
6378            .iter()
6379            .filter_map(|entry| match entry.value {
6380                FrequencyValue::Code(code) => Some(code as usize),
6381                _ => None,
6382            })
6383            .collect::<Vec<_>>();
6384        codes.sort_unstable();
6385        codes.dedup();
6386        let texts = match &dictionary {
6387            Some(dictionary) if !codes.is_empty() => dictionary.try_values_visited(&codes)?,
6388            _ => Vec::new(),
6389        };
6390        let mut out = Vec::with_capacity(entries.len());
6391        for (entry_at, entry) in entries.iter().enumerate() {
6392            let value = match entry.value {
6393                FrequencyValue::Null => {
6394                    if stored_texts.and_then(|texts| texts[entry_at].as_ref()).is_some() {
6395                        return Err(invalid("a null frequency entry has text"));
6396                    }
6397                    Value::Null
6398                }
6399                FrequencyValue::Integer(value) => match *ty {
6400                    LogicalType::TinyInt => Value::TinyInt(
6401                        i8::try_from(value)
6402                            .map_err(|_| invalid("frequency TINYINT is out of range"))?,
6403                    ),
6404                    LogicalType::UTinyInt => Value::UTinyInt(
6405                        u8::try_from(value)
6406                            .map_err(|_| invalid("frequency UTINYINT is out of range"))?,
6407                    ),
6408                    LogicalType::USmallInt => Value::USmallInt(
6409                        u16::try_from(value)
6410                            .map_err(|_| invalid("frequency USMALLINT is out of range"))?,
6411                    ),
6412                    LogicalType::UInteger => Value::UInteger(
6413                        u32::try_from(value)
6414                            .map_err(|_| invalid("frequency UINTEGER is out of range"))?,
6415                    ),
6416                    LogicalType::UBigInt => Value::UBigInt(
6417                        u64::try_from(value)
6418                            .map_err(|_| invalid("frequency UBIGINT is out of range"))?,
6419                    ),
6420                    LogicalType::SmallInt => Value::SmallInt(
6421                        i16::try_from(value)
6422                            .map_err(|_| invalid("frequency SMALLINT is out of range"))?,
6423                    ),
6424                    LogicalType::Integer => Value::Integer(
6425                        i32::try_from(value)
6426                            .map_err(|_| invalid("frequency INTEGER is out of range"))?,
6427                    ),
6428                    LogicalType::BigInt => Value::BigInt(
6429                        i64::try_from(value)
6430                            .map_err(|_| invalid("frequency BIGINT is out of range"))?,
6431                    ),
6432                    LogicalType::Date => Value::Date(
6433                        i32::try_from(value)
6434                            .map_err(|_| invalid("frequency DATE is out of range"))?,
6435                    ),
6436                    LogicalType::Timestamp => Value::Timestamp(
6437                        i64::try_from(value)
6438                            .map_err(|_| invalid("frequency TIMESTAMP is out of range"))?,
6439                    ),
6440                    _ => return Err(invalid("integer frequency belongs to another type")),
6441                },
6442                FrequencyValue::Code(code) => {
6443                    if let Some(text) = stored_texts.and_then(|texts| texts[entry_at].as_ref()) {
6444                        if *ty == LogicalType::Blob {
6445                            Value::Blob(text.clone())
6446                        } else {
6447                            Value::Varchar(
6448                                String::from_utf8(text.clone())
6449                                    .map_err(|_| invalid("frequency text is not UTF-8"))?,
6450                            )
6451                        }
6452                    } else {
6453                        if dictionary.is_none() {
6454                            return Err(invalid("frequency code has no dictionary or stored text"));
6455                        }
6456                        let at = codes
6457                            .binary_search(&(code as usize))
6458                            .map_err(|_| invalid("frequency code was not among the codes read"))?;
6459                        texts[at].clone()
6460                    }
6461                }
6462            };
6463            out.push((value, entry.count));
6464        }
6465        Ok(out)
6466    }
6467
6468    /// Sparse rows belonging to the bounded numeric frequency candidate set.
6469    ///
6470    /// The list is omitted when collecting it would exceed the fixed storage budget. A composite
6471    /// aggregate may accept a result over these rows only when its requested boundary is strictly
6472    /// greater than `omitted_max`.
6473    ///
6474    /// # Errors
6475    ///
6476    /// If the column is outside the schema.
6477    pub fn frequency_occurrences(&self, column: usize) -> Result<Option<FrequencyOccurrences>> {
6478        let field = self
6479            .table
6480            .fields
6481            .get(column)
6482            .ok_or_else(|| invalid("frequency column index out of range"))?;
6483        let Some(summary) = self.frequency_summary(column)? else {
6484            return Ok(None);
6485        };
6486        if summary.ordinals.is_empty() {
6487            return Ok(None);
6488        }
6489        let (anchors, anchor_indices) = if summary.ordinal_entries.len() == summary.ordinals.len() {
6490            let entries = self.decode_frequencies(column, &field.ty, &summary.entries)?;
6491            (entries.into_iter().map(|(value, _)| value).collect(), summary.ordinal_entries.clone())
6492        } else {
6493            (Vec::new(), Vec::new())
6494        };
6495        Ok(Some(FrequencyOccurrences {
6496            omitted_max: summary.omitted_max,
6497            ordinals: summary.ordinals.clone(),
6498            anchors,
6499            anchor_indices,
6500        }))
6501    }
6502
6503    /// How many distinct values one column holds, counting a null as no value.
6504    ///
6505    /// A string column of this format is written against one dictionary that covers the whole table.
6506    /// A code is handed out the first time a value is seen and nothing ever removes one, so the
6507    /// number of codes is the number of distinct values exactly rather than an estimate. That makes
6508    /// `COUNT(DISTINCT column)` over a whole table a question the directory already knows the answer
6509    /// to, and the alternative is a hash table with a row per distinct value built from a pass over
6510    /// every row.
6511    ///
6512    /// A null in the column used to make this `None` and no longer does. A null row is written as
6513    /// the code for the empty string, so a nullable column's dictionary can hold an empty string
6514    /// that no row of it actually has, and the dictionary on its own does not say which case it is.
6515    /// The writer does know, because it counts the non-null rows that use each code on its way to
6516    /// the frequency summary, so it records how many codes any row holds and the directory carries
6517    /// that number. This reads it rather than the size of the dictionary, which also means the
6518    /// dictionary page is not opened to answer.
6519    ///
6520    /// An integer column has no dictionary, and its count comes from the set the writer keeps on its
6521    /// numeric frequency pass instead, which is exact up to a cap. `None` for a column past that cap
6522    /// and for every column that is neither, where a sketch would answer approximately and SQL asked
6523    /// for the exact number.
6524    ///
6525    /// # Errors
6526    ///
6527    /// If the column is outside the schema.
6528    pub fn distinct_values(&self, column: usize) -> Result<Option<u64>> {
6529        self.table
6530            .distincts
6531            .get(column)
6532            .copied()
6533            .ok_or_else(|| invalid("distinct column index out of range"))
6534    }
6535
6536    /// How many rows of one column are null, added up over the stripes.
6537    ///
6538    /// Every stripe records this exactly when it is written, because a null count is not a bound
6539    /// that is allowed to be wide the way a minimum and a maximum are: a filter that reads one too
6540    /// many is slow and a `COUNT` that reads one too many is wrong. Adding up a few hundred numbers
6541    /// already in memory is what makes `COUNT(column)` over a whole table free.
6542    ///
6543    /// # Errors
6544    ///
6545    /// If the column is outside the schema.
6546    pub fn null_count(&self, column: usize) -> Result<u64> {
6547        if column >= self.table.fields.len() {
6548            return Err(invalid("null count column index out of range"));
6549        }
6550        let mut nulls = 0_u64;
6551        for stripe in &self.table.stripes {
6552            let range = stripe
6553                .zone
6554                .column(column)
6555                .ok_or_else(|| invalid("stripe zone is narrower than the schema"))?;
6556            nulls = nulls
6557                .checked_add(range.nulls as u64)
6558                .ok_or_else(|| invalid("null count overflow"))?;
6559        }
6560        Ok(nulls)
6561    }
6562
6563    /// The smallest and the largest value of one string column, from the order beside its values.
6564    ///
6565    /// The dictionary holds exactly the values the column holds, so the first and the last of them
6566    /// in sorted order are the column's minimum and maximum. Two reads of a rank block settle what
6567    /// otherwise walks a million rows.
6568    ///
6569    /// `None` when the column is not a string, when the file was written before version 9 and so has
6570    /// no order, when the column has no values at all, or when it has a null in it, which is the
6571    /// placeholder again: the empty string a null is written as would sort ahead of every real
6572    /// value and be reported as the minimum.
6573    ///
6574    /// # Errors
6575    ///
6576    /// If the column is outside the schema, or a rank names a code the dictionary does not have.
6577    pub fn text_extremes(&self, column: usize) -> Result<Option<(Value, Value)>> {
6578        if self.null_count(column)? > 0 || self.demoted(column) {
6579            return Ok(None);
6580        }
6581        let Some(dictionary) = self.dictionary(column)? else { return Ok(None) };
6582        let Some(ranks) = dictionary.ranks() else { return Ok(None) };
6583        if ranks == 0 {
6584            return Ok(None);
6585        }
6586        let low = text_at_rank(&dictionary, 0)?;
6587        let high = text_at_rank(&dictionary, ranks - 1)?;
6588        Ok(Some((low, high)))
6589    }
6590
6591    /// The smallest and the largest value of one column, when every stripe wrote exact ends.
6592    ///
6593    /// A stripe's ends are allowed to be wider than the truth, because a bound that rules out a
6594    /// chunk that could not match is still correct when it rules out nothing. That is what makes
6595    /// them cheap to write for a bit packed or a dictionary column, and it is also what stops them
6596    /// answering a `MIN`. So each stripe says which of the two it wrote, and this answers only when
6597    /// all of them walked their rows.
6598    ///
6599    /// `None` for a column with no ends, for an empty table, and for a column any stripe of which
6600    /// guessed. Nulls need no special case, because the ends skip them the same way `MIN` does.
6601    ///
6602    /// One case is given up on that did not have to be. A stripe merges the ends of its sixty four
6603    /// parts, and a part with no ends at all erases the merged ones, because a part whose rows are
6604    /// not covered by the stripe's ends is a stripe that would skip rows it should keep. A part of
6605    /// nothing but nulls has no rows to cover and so did not need to erase anything, but the merge
6606    /// cannot tell that part from a part whose layout it could not read. So a column with a chunk
6607    /// of nothing but nulls in the middle of it goes and reads the rows. That is slow and right,
6608    /// and the fix is a row count per part rather than anything here.
6609    ///
6610    /// # Errors
6611    ///
6612    /// If the column is outside the schema.
6613    pub fn exact_extremes(&self, column: usize) -> Result<Option<(Bound, Bound)>> {
6614        if column >= self.table.fields.len() {
6615            return Err(invalid("extremes column index out of range"));
6616        }
6617        let mut low: Option<Bound> = None;
6618        let mut high: Option<Bound> = None;
6619        for stripe in &self.table.stripes {
6620            let range = stripe
6621                .zone
6622                .column(column)
6623                .ok_or_else(|| invalid("stripe zone is narrower than the schema"))?;
6624            if !range.exact {
6625                return Ok(None);
6626            }
6627            // A stripe of nothing but nulls has no ends and says nothing about the column's, which
6628            // is why this skips it rather than giving up on the whole column. A stripe that has
6629            // rows and still has no end is a layout whose values this cannot see, and skipping that
6630            // one would answer with an end taken from the other stripes, so it gives up instead.
6631            let (Some(small), Some(large)) = (range.low.as_ref(), range.high.as_ref()) else {
6632                if stripe.rows > range.nulls {
6633                    return Ok(None);
6634                }
6635                continue;
6636            };
6637            low = Some(low.map_or_else(|| small.clone(), |held| held.smaller(small.clone())));
6638            high = Some(high.map_or_else(|| large.clone(), |held| held.larger(large.clone())));
6639        }
6640        Ok(low.zip(high))
6641    }
6642
6643    /// The sum of one integer column and how many rows went into it, when every stripe wrote one.
6644    ///
6645    /// The count beside the sum is the non-null rows, because that is what a `SUM` adds up and what
6646    /// an `AVG` divides by, and a caller that had to work it out from the row count and the null
6647    /// count would be doing the same walk twice.
6648    ///
6649    /// `None` for anything that is not an integer column, for a file written by something that did
6650    /// not record it, and when adding the stripes together would overflow.
6651    ///
6652    /// # Errors
6653    ///
6654    /// If the column is outside the schema.
6655    pub fn exact_sum(&self, column: usize) -> Result<Option<(i128, u64)>> {
6656        if column >= self.table.fields.len() {
6657            return Err(invalid("sum column index out of range"));
6658        }
6659        let mut total = 0_i128;
6660        let mut rows = 0_u64;
6661        for stripe in &self.table.stripes {
6662            let range = stripe
6663                .zone
6664                .column(column)
6665                .ok_or_else(|| invalid("stripe zone is narrower than the schema"))?;
6666            let Some(part) = range.sum else { return Ok(None) };
6667            let Some(sum) = total.checked_add(part) else { return Ok(None) };
6668            total = sum;
6669            rows = rows.saturating_add(stripe.rows as u64 - range.nulls as u64);
6670        }
6671        Ok(Some((total, rows)))
6672    }
6673
6674    /// Certified host groups over a string column, when the caller's inclusive row-count bound
6675    /// excludes every host the synopsis omitted.
6676    pub fn host_groups(
6677        &self,
6678        column: usize,
6679        minimum_count: u64,
6680    ) -> Result<Option<Vec<host::HostEntry>>> {
6681        if column >= self.table.fields.len() {
6682            return Err(invalid("host group column index out of range"));
6683        }
6684        let Some(summary) = &self.table.host_groups else { return Ok(None) };
6685        if summary.column != column || minimum_count <= summary.omitted_max {
6686            return Ok(None);
6687        }
6688        Ok(Some(summary.entries.clone()))
6689    }
6690
6691    /// Whether the column's dictionary stopped taking values partway through the load, and so
6692    /// decodes the stripes written before that and says nothing about the column as a whole. See
6693    /// `DEMOTED`.
6694    #[must_use]
6695    pub fn demoted(&self, column: usize) -> bool {
6696        self.table.demoted.get(column).copied().unwrap_or(false)
6697    }
6698
6699    /// The global dictionary of a column, opened once however many workers ask for it at once.
6700    ///
6701    /// The unlocked look is first because it is the answer every time after the first and it costs a
6702    /// load. Everybody who misses it queues on [`Self::loading`] and looks again on the way in, so
6703    /// the one who arrived first does the reading and the rest take what it left. Waiting is the
6704    /// cheaper thing to do: the work behind the lock is a page read, a checksum and the decode of a
6705    /// dictionary that can hold half a million entries, and the alternative is every worker of the
6706    /// scan doing all of it and all but one dropping the result on the floor.
6707    fn dictionary(&self, column: usize) -> Result<Option<Arc<Vector>>> {
6708        let Some(page) = self.table.dictionaries[column] else { return Ok(None) };
6709        if let Some(dictionary) = self.dictionaries[column].get() {
6710            return Ok(Some(Arc::clone(dictionary)));
6711        }
6712        let _queued = self.loading[column].lock().map_err(|_| invalid("a poisoned dictionary"))?;
6713        if let Some(dictionary) = self.dictionaries[column].get() {
6714            return Ok(Some(Arc::clone(dictionary)));
6715        }
6716        self.opened.fetch_add(1, Atomic::Relaxed);
6717        let dictionary = Arc::new(open_global_dictionary(
6718            Arc::clone(&self.file),
6719            page,
6720            &self.table.fields[column].ty,
6721            TEXT_KEEP_BUDGET,
6722        )?);
6723        let _ = self.dictionaries[column].set(Arc::clone(&dictionary));
6724        Ok(Some(dictionary))
6725    }
6726
6727    /// Reads one section's extent table and checks it against the entry that names it.
6728    ///
6729    /// # Errors
6730    ///
6731    /// If the entry points outside the file, the table does not checksum, or it does not decode as
6732    /// a run of extents in element order.
6733    pub fn extents(&self, of: &Section) -> Result<Vec<section::Extent>> {
6734        if of.extent_bytes == 0 {
6735            return Ok(Vec::new());
6736        }
6737        let mut bytes = vec![0; of.extent_bytes as usize];
6738        read_at(&self.file, of.extent_page, &mut bytes)?;
6739        if checksum(&bytes) != of.hash {
6740            return Err(invalid("a section's extent table does not checksum"));
6741        }
6742        let extents = section::decode_extents(&bytes)?;
6743        if extents.len() != of.extents as usize {
6744            return Err(invalid("a section's extent table is not the length the entry says"));
6745        }
6746        Ok(extents)
6747    }
6748
6749    /// Reads and verifies one extent of a section.
6750    ///
6751    /// This is what section 3.2's second rule is for. A reduction that needs one extent of a two
6752    /// gigabyte forward link reads and checksums that extent and nothing else, which is the whole
6753    /// difference between a structure that works at SF100 and issue #745.
6754    ///
6755    /// # Errors
6756    ///
6757    /// If the extent points outside the file, or its bytes do not checksum.
6758    pub fn extent(&self, of: &section::Extent) -> Result<Vec<u8>> {
6759        let end = of
6760            .offset
6761            .checked_add(u64::from(of.length))
6762            .ok_or_else(|| invalid("an extent overflows the file"))?;
6763        if of.offset < HEADER || end > self.size {
6764            return Err(invalid("an extent is outside the file"));
6765        }
6766        let mut bytes = vec![0; of.length as usize];
6767        read_at(&self.file, of.offset, &mut bytes)?;
6768        if checksum(&bytes) != of.hash {
6769            return Err(invalid("an extent does not checksum"));
6770        }
6771        Ok(bytes)
6772    }
6773
6774    /// Reads a whole section's payload, every extent of it, in order.
6775    ///
6776    /// For a structure that is resident anyway, which a key map is. Anything large enough that the
6777    /// split matters should be walking [`Reader::extents`] and taking the one it needs.
6778    ///
6779    /// # Errors
6780    ///
6781    /// If the extent table or any extent fails its check.
6782    pub fn payload(&self, of: &Section) -> Result<Vec<u8>> {
6783        let extents = self.extents(of)?;
6784        let mut bytes =
6785            Vec::with_capacity(sum(extents.iter().map(|one| u64::from(one.length))) as usize);
6786        for one in &extents {
6787            if one.first != bytes.len() as u64 {
6788                return Err(invalid("a section's extents do not join up"));
6789            }
6790            bytes.extend_from_slice(&self.extent(one)?);
6791        }
6792        // The same exception `write_section` makes: a budget record has no bytes, so its
6793        // `header_bytes` is a size rather than a header and there is nothing for it to run past.
6794        if !bytes.is_empty() && of.header_bytes as usize > bytes.len() {
6795            return Err(invalid("a section's header is longer than its payload"));
6796        }
6797        Ok(bytes)
6798    }
6799
6800    /// Reads only the named columns from one part.
6801    ///
6802    /// The whole stripe page each column lives in is read and kept, because a scan asks for the
6803    /// parts of a stripe one after another and this is what turns sixty four reads into one.
6804    ///
6805    /// # Errors
6806    ///
6807    /// If a part, column, page, or checksum is invalid.
6808    pub fn read(&self, part: usize, columns: &[usize]) -> Result<Chunk> {
6809        self.read_impl(part, columns, true, None)
6810    }
6811
6812    /// Reads named columns from one part without keeping the stripe page it came out of.
6813    ///
6814    /// This is for sparse row fetches after a selective TopN or filter, which reach a few parts of
6815    /// a stripe rather than all of them. A caller that will read most of a stripe should use
6816    /// [`Self::read`] instead, because this reads and discards the page index every time.
6817    ///
6818    /// # Errors
6819    ///
6820    /// If a part, column, page, or checksum is invalid.
6821    pub fn read_sparse(&self, part: usize, columns: &[usize]) -> Result<Chunk> {
6822        self.read_impl(part, columns, false, None)
6823    }
6824
6825    /// Counts one signed integer part from its encoded row values when it uses an all-valid
6826    /// cascade. Sparse and run-length cascades are folded without expanding their rows. Other
6827    /// page forms return `None` so the caller can use the ordinary reader.
6828    ///
6829    /// # Errors
6830    ///
6831    /// If a part, column, page checksum, or encoded integer is invalid.
6832    pub fn integer_tally(&self, part: usize, column: usize) -> Result<Option<Vec<(i64, u64)>>> {
6833        let place = *self.places.get(part).ok_or_else(|| invalid("part index out of range"))?;
6834        let field =
6835            self.table.fields.get(column).ok_or_else(|| invalid("column index out of range"))?;
6836        if !matches!(
6837            field.ty,
6838            LogicalType::TinyInt
6839                | LogicalType::SmallInt
6840                | LogicalType::Integer
6841                | LogicalType::BigInt
6842        ) {
6843            return Ok(None);
6844        }
6845        let stripe_index = place.stripe as usize;
6846        let stripe = self
6847            .table
6848            .stripes
6849            .get(stripe_index)
6850            .ok_or_else(|| invalid("stripe index out of range"))?;
6851        let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
6852        let held = self.held(stripe_index, stripe, column, true)?;
6853        let span = *held
6854            .index
6855            .get(place.part as usize)
6856            .ok_or_else(|| invalid("part index out of range"))?;
6857        let owned;
6858        let bytes = match &held.page {
6859            Some(page) => part_bytes(page, span)?,
6860            None => {
6861                let offset = page
6862                    .offset
6863                    .checked_add(span.start as u64)
6864                    .ok_or_else(|| invalid("part range overflow"))?;
6865                let mut bytes = vec![0; span.length];
6866                read_at(&self.file, offset, &mut bytes)?;
6867                owned = bytes;
6868                &owned
6869            }
6870        };
6871        if checksum(bytes) != span.hash {
6872            return Err(invalid("integer part checksum differs"));
6873        }
6874        if bytes.first() != Some(&5) || bytes.get(1) != Some(&0) {
6875            return Ok(None);
6876        }
6877        let (rows, counts) = integer::tally(&bytes[2..])?;
6878        if rows != place.rows as usize {
6879            return Err(invalid("encoded integer part holds the wrong number of rows"));
6880        }
6881        for &(value, _) in &counts {
6882            let fits = match field.ty {
6883                LogicalType::TinyInt => i8::try_from(value).is_ok(),
6884                LogicalType::SmallInt => i16::try_from(value).is_ok(),
6885                LogicalType::Integer => i32::try_from(value).is_ok(),
6886                LogicalType::BigInt => true,
6887                _ => false,
6888            };
6889            if !fits {
6890                return Err(invalid("encoded integer value is outside its column type"));
6891            }
6892        }
6893        Ok(Some(counts))
6894    }
6895
6896    /// Reads named columns from one part, only at the rows `positions` names.
6897    ///
6898    /// For a scan that already knows which rows of the part it keeps, from the columns it read
6899    /// first. A compressed string page decompresses only those rows, and every other page is
6900    /// decoded whole and gathered, which is what reading it and narrowing it costs anyway. With
6901    /// `whole` the stripe's pages are kept the way [`Self::read`] keeps them, and without it they
6902    /// are not, the way [`Self::read_sparse`] does.
6903    ///
6904    /// # Errors
6905    ///
6906    /// If a part, column, page, or checksum is invalid, or the positions do not rise or run past
6907    /// the end of the part.
6908    pub fn read_rows(
6909        &self,
6910        part: usize,
6911        columns: &[usize],
6912        positions: &[u32],
6913        whole: bool,
6914    ) -> Result<Chunk> {
6915        self.read_impl(part, columns, whole, Some(positions))
6916    }
6917
6918    /// Whether an exact global-code membership index proves that the stripe holding a part cannot
6919    /// contain any of the sorted candidate codes.
6920    ///
6921    /// # Errors
6922    ///
6923    /// If the part, column, index page, checksum, or delta stream is invalid.
6924    pub fn skips_codes(&self, part: usize, column: usize, candidates: &[u32]) -> Result<bool> {
6925        // A demoted column's later stripes hold values the dictionary never coded, so no list of
6926        // codes can prove a stripe of it holds none of a value.
6927        if self.demoted(column) {
6928            return Ok(false);
6929        }
6930        if candidates.is_empty() {
6931            return Ok(true);
6932        }
6933        if candidates.windows(2).any(|pair| pair[0] >= pair[1]) {
6934            return Err(Error::internal("native code candidates are not sorted and unique"));
6935        }
6936        let stripe = self.stripe_of(part)?;
6937        let Some(page) = stripe.memberships.get(column) else {
6938            return Ok(false);
6939        };
6940        let mut bytes = vec![0; page.length as usize];
6941        read_at(&self.file, page.offset, &mut bytes)?;
6942        if checksum(&bytes) != page.hash {
6943            return Err(invalid("membership page checksum differs"));
6944        }
6945        let codes = decode_membership(&bytes)?;
6946        let mut left = 0;
6947        let mut right = 0;
6948        while left < codes.len() && right < candidates.len() {
6949            match codes[left].cmp(&candidates[right]) {
6950                Ordering::Less => left += 1,
6951                Ordering::Greater => right += 1,
6952                Ordering::Equal => return Ok(false),
6953            }
6954        }
6955        Ok(true)
6956    }
6957
6958    fn stripe_of(&self, part: usize) -> Result<&Stripe> {
6959        let place = self.places.get(part).ok_or_else(|| invalid("part index out of range"))?;
6960        self.table
6961            .stripes
6962            .get(place.stripe as usize)
6963            .ok_or_else(|| invalid("stripe index out of range"))
6964    }
6965
6966    /// The page index of one column of one stripe, and its page when the caller wants all of it.
6967    ///
6968    /// A scan hands parts out in order, so every worker on a column crosses into a new stripe within
6969    /// a few parts of the others and they all want the same page at the same moment. This used to
6970    /// let all of them read it, which cost the scan as many copies of every page as it had workers.
6971    /// On the full ClickBench file a `MIN(EventDate), MAX(EventDate)` moved 3.2 GB off the disk to
6972    /// look at 400 MB of column.
6973    ///
6974    /// A worker that finds the page it wants already being read neither waits for it nor reads it
6975    /// again. It comes back with the index alone, which sends [`Reader::read_impl`] down the path
6976    /// that reads the one part it came for, a few kilobytes against a quarter of a megabyte, and it
6977    /// picks the page up from the cache on its next part. Waiting would be the other way to avoid
6978    /// the duplicate read and it is worse: the pages that matter are the wide string ones, they take
6979    /// milliseconds to copy even warm, and every other worker would be stopped for all of it.
6980    ///
6981    /// The file is never read under the lock.
6982    fn held(&self, at: usize, stripe: &Stripe, column: usize, whole: bool) -> Result<CachedColumn> {
6983        let cache =
6984            self.cache.columns.get(column).ok_or_else(|| invalid("column index out of range"))?;
6985        let mut cached = cache.lock().map_err(|_| invalid("column page cache is poisoned"))?;
6986        let known = cached.index.get(at).and_then(Clone::clone);
6987        let page = cached.pages.get(at).and_then(Option::as_ref).map(|slot| {
6988            slot.used.store(true, Atomic::Relaxed);
6989            Arc::clone(&slot.page)
6990        });
6991        if let Some(index) = known.clone() {
6992            if !whole || page.is_some() {
6993                return Ok(CachedColumn { stripe: at, index, page });
6994            }
6995        }
6996        if cached.loading.contains(&at) {
6997            drop(cached);
6998            // The index is almost always already here, because somebody read this stripe to get
6999            // into the loading list in the first place, so this branch usually costs no read at
7000            // all and the one part read in `read_impl` is all the losing worker pays for.
7001            if let Some(index) = known {
7002                return Ok(CachedColumn { stripe: at, index, page: None });
7003            }
7004            let held = self.page_of(stripe, column, at, false, None)?;
7005            let mut cached = cache.lock().map_err(|_| invalid("column page cache is poisoned"))?;
7006            remember(&mut cached, &held);
7007            return Ok(held);
7008        }
7009        cached.loading.push(at);
7010        drop(cached);
7011
7012        let read = self.page_of(stripe, column, at, whole, known);
7013
7014        // The stripe leaves the loading list and its page enters the cache under one lock. Doing
7015        // them separately would leave a moment where another worker sees neither and reads the
7016        // page a second time, which is the whole thing this is here to stop.
7017        let mut cached = cache.lock().map_err(|_| invalid("column page cache is poisoned"))?;
7018        if let Some(position) = cached.loading.iter().position(|loading| *loading == at) {
7019            cached.loading.remove(position);
7020        }
7021        let held = read?;
7022        let taken = remember(&mut cached, &held);
7023        drop(cached);
7024        if let Some((bytes, used)) = taken {
7025            self.cache.held[column].fetch_add(1, Atomic::Relaxed);
7026            self.pool.admit(Held {
7027                shelf: Arc::downgrade(&self.cache),
7028                column,
7029                stripe: at,
7030                bytes,
7031                used,
7032            });
7033        }
7034        Ok(held)
7035    }
7036
7037    /// Reads one stripe's index for a column, and its page when the caller wants all of it.
7038    ///
7039    /// `known` is the index when the reader has already read it, which after the first worker
7040    /// through a stripe it always has, because [`remember`] keeps every index for the life of the
7041    /// reader. Without that a scan reads the index again on every part that misses the page cache.
7042    fn page_of(
7043        &self,
7044        stripe: &Stripe,
7045        column: usize,
7046        at: usize,
7047        whole: bool,
7048        known: Option<Arc<Vec<PartSpan>>>,
7049    ) -> Result<CachedColumn> {
7050        let index = match known {
7051            Some(index) => index,
7052            None => {
7053                self.indexes.fetch_add(1, Atomic::Relaxed);
7054                Arc::new(read_index(&self.file, stripe, column)?)
7055            }
7056        };
7057        let page = if whole {
7058            self.pages.fetch_add(1, Atomic::Relaxed);
7059            let span = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
7060            let mut bytes = vec![0; span.length as usize];
7061            read_at(&self.file, span.offset, &mut bytes)?;
7062            Some(Arc::new(bytes))
7063        } else {
7064            None
7065        };
7066        Ok(CachedColumn { stripe: at, index, page })
7067    }
7068
7069    fn read_impl(
7070        &self,
7071        at: usize,
7072        columns: &[usize],
7073        whole: bool,
7074        positions: Option<&[u32]>,
7075    ) -> Result<Chunk> {
7076        let place = *self.places.get(at).ok_or_else(|| invalid("part index out of range"))?;
7077        let index = place.stripe as usize;
7078        let stripe =
7079            self.table.stripes.get(index).ok_or_else(|| invalid("stripe index out of range"))?;
7080        let rows = place.rows as usize;
7081        let mut picked = Vec::with_capacity(columns.len());
7082        for &column in columns {
7083            let field = self
7084                .table
7085                .fields
7086                .get(column)
7087                .ok_or_else(|| invalid("column index out of range"))?;
7088            let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
7089            let held = self.held(index, stripe, column, whole)?;
7090            let span = *held
7091                .index
7092                .get(place.part as usize)
7093                .ok_or_else(|| invalid("part index out of range"))?;
7094            let owned;
7095            let bytes = match &held.page {
7096                Some(held) => part_bytes(held, span)?,
7097                None => {
7098                    let offset = page
7099                        .offset
7100                        .checked_add(span.start as u64)
7101                        .ok_or_else(|| invalid("part range overflow"))?;
7102                    let mut bytes = vec![0; span.length];
7103                    read_at(&self.file, offset, &mut bytes)?;
7104                    owned = bytes;
7105                    &owned
7106                }
7107            };
7108            if checksum(bytes) != span.hash {
7109                return Err(invalid(&format!(
7110                    "column page checksum differs, column {column} part {} at {}+{} of {} bytes, \
7111                     wanted {:016x} and got {:016x}",
7112                    place.part,
7113                    page.offset,
7114                    span.start,
7115                    span.length,
7116                    span.hash,
7117                    checksum(bytes),
7118                )));
7119            }
7120            let dictionary = self.dictionary(column)?;
7121            // Held as a page, because a column that came out of a file is handed out more than
7122            // once. A group by clones its key columns out of the chunk so the keys outlive it, a
7123            // projection of a bare column name does the same, and a cut of a flat run copies unless
7124            // the run is a page. One `Arc` per column per part buys all of those, and it moves the
7125            // run into the `Arc` without touching a value.
7126            let mut vector = match positions {
7127                None => decode(&field.ty, rows, bytes, dictionary)?,
7128                Some(positions) => decode_at(&field.ty, rows, bytes, dictionary, positions)?,
7129            };
7130            // A demoted column's codes are not the column's codes, only the codes of the stripes
7131            // written before the demotion, so they are not handed out as if they were. See
7132            // [`DEMOTED`].
7133            if self.demoted(column) && vector.stable_dictionary_parts().is_some() {
7134                vector = vector.flatten()?;
7135            }
7136            picked.push(vector.into_pages());
7137        }
7138        Chunk::with_rows(picked, positions.map_or(rows, <[u32]>::len))
7139    }
7140
7141    /// Whether persisted statistics prove that a part cannot match the predicates.
7142    ///
7143    /// Three of them, asked cheapest first.
7144    ///
7145    /// The stripe's bounds are in memory already, so they are free, and they are also the coarsest:
7146    /// every part of a stripe gets the same answer and a scan that skips one part that way skips all
7147    /// sixty four. Then the part's own bounds, which are a read of one page per column per stripe
7148    /// and are sixty four times finer. Then the sieves, which are per part and answer equality, the
7149    /// test bounds are worst at: a column of identifiers has every stripe and nearly every part
7150    /// covering the whole of its type, so bounds keep them all and the sieve keeps the ones that
7151    /// really hold the value.
7152    ///
7153    /// The middle one is what an ordered comparison on a column the rows are not sorted by needs. On
7154    /// ClickBench 24 the stripe bounds leave eight stripes of sixteen alive, which is half the file,
7155    /// and the part bounds leave thirty parts of nine hundred and seventy four.
7156    #[must_use]
7157    pub fn skips(&self, part: usize, probes: &[Probe]) -> bool {
7158        let Some(place) = self.places.get(part).copied() else { return false };
7159        let Some(stripe) = self.table.stripes.get(place.stripe as usize) else { return false };
7160        if stripe.zone.skips(probes) {
7161            return true;
7162        }
7163        probes.iter().any(|probe| self.outside(place, probe) || self.sifted(place, probe))
7164    }
7165
7166    /// Whether the bounds of one part rule out one probe.
7167    ///
7168    /// The part's own two ends, which are narrower than the stripe's and cost a page read the first
7169    /// time this is asked about a column. A column with no page here answers `false`, which is the
7170    /// answer a caller got before there were any.
7171    fn outside(&self, place: Place, probe: &Probe) -> bool {
7172        match self.stripe_part_ranges(place.stripe as usize, probe.column) {
7173            Some(ranges) => ranges
7174                .get(place.part as usize)
7175                .is_some_and(|range| range.excludes(probe.op, &probe.value)),
7176            None => false,
7177        }
7178    }
7179
7180    /// The per part ranges of one stripe of one column, read once and kept.
7181    ///
7182    /// `None` when the column has no page in that stripe and when the page is damaged, on the same
7183    /// reasoning as the sieves: this is an index over data that is still there, so a caller that
7184    /// cannot read one reads the rows and gets the right answer slowly.
7185    fn stripe_part_ranges(&self, stripe: usize, column: usize) -> Option<&[Range]> {
7186        let slot = self.part_ranges.get(column)?.get(stripe)?;
7187        if let Some(held) = slot.get() {
7188            return Some(held);
7189        }
7190        let page = self.table.stripes.get(stripe)?.part_ranges.get(column)?;
7191        let mut bytes = vec![0; page.length as usize];
7192        read_at(&self.file, page.offset, &mut bytes).ok()?;
7193        if checksum(&bytes) != page.hash {
7194            return None;
7195        }
7196        let ranges = Arc::new(decode_part_ranges(&bytes).ok()?);
7197        let _ = slot.set(ranges);
7198        slot.get().map(|held| held.as_slice())
7199    }
7200
7201    /// Whether persisted statistics prove that every row of a part matches the predicates.
7202    ///
7203    /// Only the bounds. The sieves say nothing here, because a sieve that holds a value is a sieve
7204    /// that may be holding somebody else's hash, so it can rule a part out and can never wave one
7205    /// through.
7206    ///
7207    /// The stripe first and the part after it, the same two steps and in the same order as
7208    /// [`Self::skips`]. The stripe's bounds are in memory already and its null count covers sixty
7209    /// four parts rather than one, so a stripe that answers is an answer for nothing, and the part's
7210    /// own bounds are only read for the probes it could not settle. Both directions are safe: a
7211    /// stretch where everything passes contains no narrower stretch where something fails, and a
7212    /// stripe with no nulls has no nulls in any of its parts.
7213    ///
7214    /// A string end a part recorded is cut down to its first few bytes, so a part's stretch can be
7215    /// wider than its rows really are as well. That is the same safe direction for the same reason,
7216    /// and it is why this asks the two ends rather than anything `exact` says.
7217    #[must_use]
7218    pub fn certain(&self, part: usize, probes: &[Probe]) -> bool {
7219        let Some(place) = self.places.get(part).copied() else { return false };
7220        let Some(stripe) = self.table.stripes.get(place.stripe as usize) else { return false };
7221        if stripe.zone.certain(probes) {
7222            return true;
7223        }
7224        probes
7225            .iter()
7226            .all(|probe| stripe.zone.certain(slice::from_ref(probe)) || self.inside(place, probe))
7227    }
7228
7229    /// Whether one part's own two ends prove that every row of it passes `probe`.
7230    ///
7231    /// The mirror of [`Self::outside`], reading the same page. `false` for a part whose stripe wrote
7232    /// no range page, which is a stripe of one part, because there the stripe's own bounds are the
7233    /// part's and the caller has already asked them.
7234    fn inside(&self, place: Place, probe: &Probe) -> bool {
7235        match self.stripe_part_ranges(place.stripe as usize, probe.column) {
7236            Some(ranges) => ranges
7237                .get(place.part as usize)
7238                .is_some_and(|range| range.certain(probe.op, &probe.value)),
7239            None => false,
7240        }
7241    }
7242
7243    /// Whether the bounds of one stripe prove that none of its parts can match the predicates.
7244    ///
7245    /// The cheap half of [`Self::skips`], asked about a whole stripe at once. The bounds live in the
7246    /// directory and are already in memory, so this answers without touching the file, and that is
7247    /// the reason it is worth having on its own: a caller that wants to know roughly where the work
7248    /// is before it starts any workers can ask this about sixteen stripes for nothing, where asking
7249    /// [`Self::skips`] about nine hundred parts would read and decode a sieve page per stripe first.
7250    ///
7251    /// It keeps stripes that [`Self::skips`] would rule out part by part, which is the right way for
7252    /// it to be wrong: the parts are still checked when they are read.
7253    #[must_use]
7254    pub fn stripe_skips(&self, stripe: usize, probes: &[Probe]) -> bool {
7255        self.table.stripes.get(stripe).is_some_and(|held| held.zone.skips(probes))
7256    }
7257
7258    /// Whether the sieve of one part rules out one probe.
7259    ///
7260    /// Only equality. An ordered comparison is what the bounds are for and a sieve says nothing
7261    /// about it, and a read that cannot answer keeps the part, which is the answer a caller with no
7262    /// sieve gets anyway.
7263    fn sifted(&self, place: Place, probe: &Probe) -> bool {
7264        if probe.op != Op::Equal {
7265            return false;
7266        }
7267        match self.stripe_sieves(place.stripe as usize, probe.column) {
7268            Some(sieves) => sieves
7269                .get(place.part as usize)
7270                .and_then(Option::as_ref)
7271                .is_some_and(|sieve| sieve.excludes(&probe.value)),
7272            None => false,
7273        }
7274    }
7275
7276    /// The sieves of one stripe of one column, read once and kept.
7277    ///
7278    /// `None` when the column has no sieves in that stripe, when the page is damaged, and when the
7279    /// bytes are not a page this version can read. A sieve is an index over data that is still there
7280    /// and a caller that cannot read one reads the rows, so this is the one place in the file where
7281    /// a bad checksum is a slow query rather than an error.
7282    fn stripe_sieves(&self, stripe: usize, column: usize) -> Option<&[Option<Sieve>]> {
7283        let slot = self.sieves.get(column)?.get(stripe)?;
7284        if let Some(held) = slot.get() {
7285            return Some(held);
7286        }
7287        let page = self.table.stripes.get(stripe)?.sieves.get(column)?;
7288        let mut bytes = vec![0; page.length as usize];
7289        read_at(&self.file, page.offset, &mut bytes).ok()?;
7290        if checksum(&bytes) != page.hash {
7291            return None;
7292        }
7293        let sieves = Arc::new(decode_sieves(&bytes).ok()?);
7294        let _ = slot.set(sieves);
7295        slot.get().map(|held| held.as_slice())
7296    }
7297}
7298
7299/// The value sitting at one position of a dictionary's sorted order.
7300fn text_at_rank(dictionary: &Vector, rank: usize) -> Result<Value> {
7301    let code = dictionary.code_at_rank(rank)? as usize;
7302    if dictionary.logical_type() == &LogicalType::Blob {
7303        let bytes = dictionary
7304            .try_bytes_at(code)?
7305            .ok_or_else(|| invalid("global dictionary order names a code it does not have"))?;
7306        return Ok(Value::Blob(bytes.to_vec()));
7307    }
7308    let text = dictionary
7309        .try_text_at(code)?
7310        .ok_or_else(|| invalid("global dictionary order names a code it does not have"))?;
7311    Ok(Value::Varchar(text.into()))
7312}
7313
7314/// Reads one span of a file at an offset, without moving a cursor anybody else can see.
7315///
7316/// Every reader of a table shares one [`File`] behind an [`Arc`], and a grouped aggregate reads its
7317/// pages from several threads at once, so this has to be positional. Seeking and then reading is
7318/// two calls with a gap in the middle, and in that gap another thread's seek lands and the read
7319/// comes back with somebody else's bytes.
7320///
7321/// The writer reads back through here too, out of the `rudb_io` file it writes through, which is
7322/// why this takes anything [`Positional`] rather than a [`File`].
7323fn read_at<F: Positional + ?Sized>(file: &F, offset: u64, bytes: &mut [u8]) -> Result<()> {
7324    file.fill_at(offset, bytes)
7325}
7326
7327/// Something a span of bytes can be read out of by offset.
7328///
7329/// There are two of these. The reader holds a `std::fs::File`, because it shares it between its
7330/// threads behind an [`Arc`] and every read it makes is on the hot path of a scan. The writer holds
7331/// an `rudb_io::File`, because everything it does to the file has to be something the simulated
7332/// filesystem can stop and crash. The few helpers both of them use, [`read_index`] and the choice
7333/// of committed slot, are written once over this rather than once for each.
7334trait Positional {
7335    /// Fills `bytes` from `offset`, or fails if the file ends first.
7336    ///
7337    /// Both kinds can come back short, so both loop. A read of zero bytes before the span is filled
7338    /// means the file stops earlier than the directory said it does.
7339    fn fill_at(&self, offset: u64, bytes: &mut [u8]) -> Result<()>;
7340}
7341
7342impl<T: Positional + ?Sized> Positional for &T {
7343    fn fill_at(&self, offset: u64, bytes: &mut [u8]) -> Result<()> {
7344        (**self).fill_at(offset, bytes)
7345    }
7346}
7347
7348impl<T: Positional + ?Sized> Positional for Arc<T> {
7349    fn fill_at(&self, offset: u64, bytes: &mut [u8]) -> Result<()> {
7350        (**self).fill_at(offset, bytes)
7351    }
7352}
7353
7354impl<T: Positional + ?Sized> Positional for Box<T> {
7355    fn fill_at(&self, offset: u64, bytes: &mut [u8]) -> Result<()> {
7356        (**self).fill_at(offset, bytes)
7357    }
7358}
7359
7360impl Positional for dyn rudb_io::File + '_ {
7361    fn fill_at(&self, mut offset: u64, mut bytes: &mut [u8]) -> Result<()> {
7362        while !bytes.is_empty() {
7363            let read = self.read_at(offset, bytes)?;
7364            if read == 0 {
7365                return Err(invalid("column page ends before its declared length"));
7366            }
7367            offset += read as u64;
7368            bytes = &mut bytes[read..];
7369        }
7370        Ok(())
7371    }
7372}
7373
7374impl Positional for File {
7375    #[cfg(unix)]
7376    fn fill_at(&self, mut offset: u64, mut bytes: &mut [u8]) -> Result<()> {
7377        use std::os::unix::fs::FileExt;
7378        while !bytes.is_empty() {
7379            let read = self.read_at(bytes, offset).map_err(io)?;
7380            if read == 0 {
7381                return Err(invalid("column page ends before its declared length"));
7382            }
7383            offset += read as u64;
7384            bytes = &mut bytes[read..];
7385        }
7386        Ok(())
7387    }
7388
7389    /// The same read, on the call Windows spells differently.
7390    ///
7391    /// `seek_read` is one `ReadFile` carrying the offset with it, so two of them cannot interleave
7392    /// the way a seek and a read can. It does leave the shared cursor somewhere afterwards, which is
7393    /// why nothing in this file may read that cursor.
7394    #[cfg(windows)]
7395    fn fill_at(&self, mut offset: u64, mut bytes: &mut [u8]) -> Result<()> {
7396        use std::os::windows::fs::FileExt;
7397        while !bytes.is_empty() {
7398            let read = self.seek_read(bytes, offset).map_err(io)?;
7399            if read == 0 {
7400                return Err(invalid("column page ends before its declared length"));
7401            }
7402            offset += read as u64;
7403            bytes = &mut bytes[read..];
7404        }
7405        Ok(())
7406    }
7407
7408    /// Somewhere that is neither, where the cursor is all there is.
7409    ///
7410    /// This one does race, and there is no way to write it so it does not. Nothing we build for
7411    /// runs here, so it exists to keep the crate compiling rather than to be correct under threads.
7412    #[cfg(not(any(unix, windows)))]
7413    fn fill_at(&self, offset: u64, bytes: &mut [u8]) -> Result<()> {
7414        use std::io::{Read, Seek, SeekFrom};
7415        let mut file = self.try_clone().map_err(io)?;
7416        file.seek(SeekFrom::Start(offset)).map_err(io)?;
7417        file.read_exact(bytes).map_err(io)
7418    }
7419}
7420
7421/// Overwrites one span of a file in place, which is how the tests damage a file on purpose.
7422///
7423/// The writer does not come through here. It writes through `rudb_io`, and this is a
7424/// `std::fs::File` opened by a test beside it.
7425#[cfg(test)]
7426fn write_at(file: &File, offset: u64, bytes: &[u8]) -> Result<()> {
7427    use std::io::{Seek, SeekFrom, Write};
7428    let mut file = file;
7429    file.seek(SeekFrom::Start(offset)).map_err(io)?;
7430    file.write_all(bytes).map_err(io)
7431}
7432
7433/// What a column type is called in the directory.
7434///
7435/// A tag is a number in a file somebody else wrote, so a tag that has been used is used forever and
7436/// the only thing that may happen to this list is that it grows. 1 to 13 are the tags the format
7437/// had when it could store thirteen types, and 14 to 27 are the rest, in the order they were added
7438/// rather than in an order that means anything.
7439fn type_tag(ty: &LogicalType) -> Result<u8> {
7440    match ty {
7441        LogicalType::SmallInt => Ok(1),
7442        LogicalType::Integer => Ok(2),
7443        LogicalType::BigInt => Ok(3),
7444        LogicalType::Varchar => Ok(4),
7445        LogicalType::Date => Ok(5),
7446        LogicalType::Timestamp => Ok(6),
7447        LogicalType::Boolean => Ok(7),
7448        LogicalType::TinyInt => Ok(8),
7449        LogicalType::UTinyInt => Ok(9),
7450        LogicalType::USmallInt => Ok(10),
7451        LogicalType::UInteger => Ok(11),
7452        LogicalType::UBigInt => Ok(12),
7453        LogicalType::Decimal { .. } => Ok(13),
7454        LogicalType::Float => Ok(14),
7455        LogicalType::Double => Ok(15),
7456        LogicalType::HugeInt => Ok(16),
7457        LogicalType::UHugeInt => Ok(17),
7458        LogicalType::Time => Ok(18),
7459        LogicalType::TimeTz => Ok(19),
7460        LogicalType::TimestampTz => Ok(20),
7461        LogicalType::Interval => Ok(21),
7462        LogicalType::Uuid => Ok(22),
7463        LogicalType::Blob => Ok(23),
7464        LogicalType::Bit => Ok(24),
7465        LogicalType::TimestampS => Ok(25),
7466        LogicalType::TimestampMs => Ok(26),
7467        LogicalType::TimestampNs => Ok(27),
7468        _ => Err(Error::not_implemented(format!("native storage for {ty}"))),
7469    }
7470}
7471
7472/// The tag of a column type, and the parameters of the ones that have any.
7473///
7474/// Only `DECIMAL` has parameters today. Width and scale go after the tag rather than into it
7475/// because they are what says how wide a value is on disk, and a reader that guessed would read the
7476/// wrong number of bytes per row rather than the wrong number of digits.
7477fn put_type(out: &mut Vec<u8>, ty: &LogicalType) -> Result<()> {
7478    out.push(type_tag(ty)?);
7479    if let LogicalType::Decimal { width, scale } = ty {
7480        out.push(*width);
7481        out.push(*scale);
7482    }
7483    Ok(())
7484}
7485
7486/// The other half of [`put_type`], reading the parameters the tag says are there.
7487fn read_type(cur: &mut Cursor<'_>) -> Result<LogicalType> {
7488    let tag = cur.u8()?;
7489    if tag == 13 {
7490        let width = cur.u8()?;
7491        let scale = cur.u8()?;
7492        return LogicalType::decimal(width, scale)
7493            .map_err(|_| invalid("decimal column width and scale are not a decimal"));
7494    }
7495    tag_type(tag)
7496}
7497
7498fn tag_type(tag: u8) -> Result<LogicalType> {
7499    match tag {
7500        1 => Ok(LogicalType::SmallInt),
7501        2 => Ok(LogicalType::Integer),
7502        3 => Ok(LogicalType::BigInt),
7503        4 => Ok(LogicalType::Varchar),
7504        5 => Ok(LogicalType::Date),
7505        6 => Ok(LogicalType::Timestamp),
7506        7 => Ok(LogicalType::Boolean),
7507        8 => Ok(LogicalType::TinyInt),
7508        9 => Ok(LogicalType::UTinyInt),
7509        10 => Ok(LogicalType::USmallInt),
7510        11 => Ok(LogicalType::UInteger),
7511        12 => Ok(LogicalType::UBigInt),
7512        14 => Ok(LogicalType::Float),
7513        15 => Ok(LogicalType::Double),
7514        16 => Ok(LogicalType::HugeInt),
7515        17 => Ok(LogicalType::UHugeInt),
7516        18 => Ok(LogicalType::Time),
7517        19 => Ok(LogicalType::TimeTz),
7518        20 => Ok(LogicalType::TimestampTz),
7519        21 => Ok(LogicalType::Interval),
7520        22 => Ok(LogicalType::Uuid),
7521        23 => Ok(LogicalType::Blob),
7522        24 => Ok(LogicalType::Bit),
7523        25 => Ok(LogicalType::TimestampS),
7524        26 => Ok(LogicalType::TimestampMs),
7525        27 => Ok(LogicalType::TimestampNs),
7526        _ => Err(invalid("column type tag is unknown")),
7527    }
7528}
7529
7530fn put_u16(out: &mut Vec<u8>, value: u16) {
7531    out.extend_from_slice(&value.to_le_bytes());
7532}
7533fn put_u32(out: &mut Vec<u8>, value: u32) {
7534    out.extend_from_slice(&value.to_le_bytes());
7535}
7536fn put_u64(out: &mut Vec<u8>, value: u64) {
7537    out.extend_from_slice(&value.to_le_bytes());
7538}
7539fn put_var_u64(out: &mut Vec<u8>, mut value: u64) {
7540    while value >= 0x80 {
7541        out.push((value as u8 & 0x7f) | 0x80);
7542        value >>= 7;
7543    }
7544    out.push(value as u8);
7545}
7546
7547fn frequency_order(left: FrequencyValue, right: FrequencyValue) -> Ordering {
7548    match (left, right) {
7549        (FrequencyValue::Null, FrequencyValue::Null) => Ordering::Equal,
7550        (FrequencyValue::Null, _) => Ordering::Less,
7551        (_, FrequencyValue::Null) => Ordering::Greater,
7552        (FrequencyValue::Integer(left), FrequencyValue::Integer(right)) => left.cmp(&right),
7553        (FrequencyValue::Code(left), FrequencyValue::Code(right)) => left.cmp(&right),
7554        (FrequencyValue::Integer(_), FrequencyValue::Code(_)) => Ordering::Less,
7555        (FrequencyValue::Code(_), FrequencyValue::Integer(_)) => Ordering::Greater,
7556    }
7557}
7558
7559/// Leaves the [`FREQUENCY_ENTRIES`] commonest entries in order and says what the next one counted.
7560///
7561/// There is one entry a distinct value, so on `URL` this is handed two and a quarter million of
7562/// them and keeps five hundred and twelve. Sorting all of them to throw almost all of them away is
7563/// the whole of what counting a dictionary column used to cost, 2.13 seconds of it on `URL` at eight
7564/// million rows against 11.93 for compressing the same column's values.
7565///
7566/// Partitioning answers both questions instead. It puts the five hundred and thirteenth entry where
7567/// it belongs and everything commoner in front of it, which is the entries to keep and the count to
7568/// report as the largest one omitted, and then only the part that survives is sorted. The order that
7569/// comes out is the order the sort gave, because the tie break makes the comparison total: two
7570/// entries never hold the same value.
7571fn keep_most_frequent(entries: &mut Vec<FrequencyEntry>) -> u64 {
7572    let order = |left: &FrequencyEntry, right: &FrequencyEntry| {
7573        right.count.cmp(&left.count).then_with(|| frequency_order(left.value, right.value))
7574    };
7575    let omitted_max = if entries.len() > FREQUENCY_ENTRIES {
7576        let (_, next, _) = entries.select_nth_unstable_by(FREQUENCY_ENTRIES, order);
7577        let omitted_max = next.count;
7578        entries.truncate(FREQUENCY_ENTRIES);
7579        omitted_max
7580    } else {
7581        0
7582    };
7583    entries.sort_unstable_by(order);
7584    omitted_max
7585}
7586
7587fn code_frequency(
7588    dictionary: &GlobalDictionary,
7589    flat: &[u8],
7590    bases: &[u64],
7591) -> Result<(FrequencySummary, Vec<Option<Vec<u8>>>)> {
7592    let mut entries = dictionary
7593        .counts
7594        .iter()
7595        .enumerate()
7596        .filter(|(_, count)| **count != 0)
7597        .map(|(code, &count)| FrequencyEntry { value: FrequencyValue::Code(code as u32), count })
7598        .collect::<Vec<_>>();
7599    if dictionary.nulls != 0 {
7600        entries.push(FrequencyEntry { value: FrequencyValue::Null, count: dictionary.nulls });
7601    }
7602    let omitted_max = keep_most_frequent(&mut entries);
7603    let mut spans = Vec::with_capacity(entries.len());
7604    let mut text_bytes = 0_usize;
7605    for entry in &entries {
7606        let span = match entry.value {
7607            FrequencyValue::Code(code) => {
7608                let span = GlobalDictionary::value_span(&dictionary.ends, bases, code as usize);
7609                let bytes = flat
7610                    .get(span.0..span.1)
7611                    .ok_or_else(|| invalid("a frequency code is outside its dictionary"))?;
7612                text_bytes = text_bytes.saturating_add(bytes.len());
7613                Some(span)
7614            }
7615            FrequencyValue::Null | FrequencyValue::Integer(_) => None,
7616        };
7617        spans.push(span);
7618    }
7619    let texts = if text_bytes > FREQUENCY_TEXT_BUDGET {
7620        Vec::new()
7621    } else {
7622        spans.into_iter().map(|span| span.map(|(from, to)| flat[from..to].to_vec())).collect()
7623    };
7624    Ok((
7625        FrequencySummary {
7626            entries,
7627            omitted_max,
7628            ordinals: Vec::new(),
7629            ordinal_entries: Vec::new(),
7630        },
7631        texts,
7632    ))
7633}
7634
7635fn encode_directory(table: &Table) -> Result<Vec<u8>> {
7636    let mut out = DIRECTORY.to_vec();
7637    let name = table.name.as_bytes();
7638    put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("table name too long"))?);
7639    out.extend_from_slice(name);
7640    put_u16(&mut out, u16::try_from(table.fields.len()).map_err(|_| invalid("too many columns"))?);
7641    for field in &table.fields {
7642        let name = field.name.as_bytes();
7643        put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("column name too long"))?);
7644        out.extend_from_slice(name);
7645        put_type(&mut out, &field.ty)?;
7646        out.push(u8::from(field.not_null));
7647    }
7648    for (field, dictionary) in table.fields.iter().zip(&table.dictionaries) {
7649        match dictionary {
7650            None => out.push(0),
7651            Some(page) => {
7652                out.push(dictionary_tag(&field.ty));
7653                put_u64(&mut out, page.offset);
7654                put_u32(&mut out, page.length);
7655                put_u64(&mut out, page.hash);
7656            }
7657        }
7658    }
7659    for distinct in &table.distincts {
7660        match distinct {
7661            None => out.push(0),
7662            Some(count) => {
7663                out.push(1);
7664                put_u64(&mut out, *count);
7665            }
7666        }
7667    }
7668    put_u64(&mut out, u64::try_from(table.rows).map_err(|_| invalid("row count overflow"))?);
7669    put_u32(&mut out, u32::try_from(table.stripes.len()).map_err(|_| invalid("too many stripes"))?);
7670    for stripe in &table.stripes {
7671        put_u32(
7672            &mut out,
7673            u32::try_from(stripe.parts.len()).map_err(|_| invalid("too many parts in a stripe"))?,
7674        );
7675        for &rows in &stripe.parts {
7676            put_u32(&mut out, rows);
7677        }
7678        put_u64(&mut out, stripe.index.offset);
7679        put_u32(&mut out, stripe.index.length);
7680        for page in &stripe.pages {
7681            put_u64(&mut out, page.offset);
7682            put_u32(&mut out, page.length);
7683        }
7684        // A membership index says which of a dictionary's codes a part holds, so a column the writer
7685        // decided against giving a dictionary has nothing for it to be about and writes none. Every
7686        // file written before that decision existed has a dictionary on every varchar column, so
7687        // this reads those files byte for byte the way it always did.
7688        for (column, ((field, dictionary), membership)) in
7689            table.fields.iter().zip(&table.dictionaries).zip(stripe.memberships.slots()).enumerate()
7690        {
7691            if !coded_type(&field.ty) || dictionary.is_none() {
7692                continue;
7693            }
7694            let page = match membership {
7695                Some(page) => page,
7696                None if table.demoted.get(column).copied().unwrap_or(false) => {
7697                    Page { offset: HEADER, length: 0, hash: 0 }
7698                }
7699                None => return Err(invalid("string page has no code membership index")),
7700            };
7701            put_u64(&mut out, page.offset);
7702            put_u32(&mut out, page.length);
7703            put_u64(&mut out, page.hash);
7704        }
7705        for sieve in stripe.sieves.slots() {
7706            match sieve {
7707                None => out.push(0),
7708                Some(page) => {
7709                    out.push(1);
7710                    put_u64(&mut out, page.offset);
7711                    put_u32(&mut out, page.length);
7712                    put_u64(&mut out, page.hash);
7713                }
7714            }
7715        }
7716        for held in stripe.part_ranges.slots() {
7717            match held {
7718                None => out.push(0),
7719                Some(page) => {
7720                    out.push(1);
7721                    put_u64(&mut out, page.offset);
7722                    put_u32(&mut out, page.length);
7723                    put_u64(&mut out, page.hash);
7724                }
7725            }
7726        }
7727        for range in stripe.zone.columns() {
7728            put_bound(&mut out, range.low.as_ref())?;
7729            put_bound(&mut out, range.high.as_ref())?;
7730            put_u32(
7731                &mut out,
7732                u32::try_from(range.nulls).map_err(|_| invalid("null count overflow"))?,
7733            );
7734            out.push(u8::from(range.exact));
7735            match range.sum {
7736                None => out.push(0),
7737                Some(total) => {
7738                    out.push(1);
7739                    out.extend_from_slice(&total.to_le_bytes());
7740                }
7741            }
7742        }
7743    }
7744    out.extend_from_slice(FREQUENCIES);
7745    put_u16(
7746        &mut out,
7747        u16::try_from(table.frequencies.len())
7748            .map_err(|_| invalid("too many frequency columns"))?,
7749    );
7750    for summary in &table.frequencies {
7751        let summary = match summary {
7752            None => {
7753                out.push(0);
7754                continue;
7755            }
7756            Some(Frequencies::Held(summary)) => summary,
7757            // Only a reader leaves a synopsis in the file, and nothing writes a reader's table back.
7758            Some(Frequencies::Stored { .. }) => {
7759                return Err(invalid("a synopsis left in the file cannot be written back"));
7760            }
7761        };
7762        out.push(1);
7763        put_u64(&mut out, summary.omitted_max);
7764        put_u32(
7765            &mut out,
7766            u32::try_from(summary.entries.len())
7767                .map_err(|_| invalid("too many frequency entries"))?,
7768        );
7769        for entry in &summary.entries {
7770            match entry.value {
7771                FrequencyValue::Null => out.push(0),
7772                FrequencyValue::Integer(value) => {
7773                    out.push(1);
7774                    out.extend_from_slice(&value.to_le_bytes());
7775                }
7776                FrequencyValue::Code(value) => {
7777                    out.push(2);
7778                    put_u32(&mut out, value);
7779                }
7780            }
7781            put_u64(&mut out, entry.count);
7782        }
7783        put_u32(
7784            &mut out,
7785            u32::try_from(summary.ordinals.len())
7786                .map_err(|_| invalid("too many frequency ordinals"))?,
7787        );
7788        let mut previous = 0_u64;
7789        for (at, &ordinal) in summary.ordinals.iter().enumerate() {
7790            let delta = if at == 0 {
7791                ordinal
7792            } else {
7793                ordinal
7794                    .checked_sub(previous)
7795                    .ok_or_else(|| invalid("frequency ordinals are not ordered"))?
7796            };
7797            if at != 0 && delta == 0 {
7798                return Err(invalid("frequency ordinals are not unique"));
7799            }
7800            put_var_u64(&mut out, delta);
7801            previous = ordinal;
7802        }
7803        if summary.ordinal_entries.len() != summary.ordinals.len() {
7804            return Err(invalid("frequency ordinal values have a different length"));
7805        }
7806        for &entry in &summary.ordinal_entries {
7807            if entry as usize >= summary.entries.len() {
7808                return Err(invalid("frequency ordinal value is outside its entries"));
7809            }
7810            put_u16(&mut out, entry);
7811        }
7812    }
7813    if !table.pair_frequencies.is_empty() {
7814        out.extend_from_slice(PAIR_FREQUENCIES);
7815        put_u16(
7816            &mut out,
7817            u16::try_from(table.pair_frequencies.len())
7818                .map_err(|_| invalid("too many pair frequency summaries"))?,
7819        );
7820        for summary in &table.pair_frequencies {
7821            put_u16(&mut out, summary.first);
7822            put_u16(&mut out, summary.second);
7823            put_u64(&mut out, summary.omitted_max);
7824            put_u16(
7825                &mut out,
7826                u16::try_from(summary.entries.len())
7827                    .map_err(|_| invalid("too many pair frequency entries"))?,
7828            );
7829            for entry in &summary.entries {
7830                put_u16(&mut out, entry.first_entry);
7831                match entry.second {
7832                    None => out.push(0),
7833                    Some(code) => {
7834                        out.push(1);
7835                        put_u32(&mut out, code);
7836                    }
7837                }
7838                put_u64(&mut out, entry.count);
7839            }
7840        }
7841    }
7842    let text_columns = table.frequency_texts.iter().filter(|texts| !texts.is_empty()).count();
7843    if text_columns != 0 {
7844        out.extend_from_slice(FREQUENCY_TEXTS);
7845        put_u16(
7846            &mut out,
7847            u16::try_from(text_columns)
7848                .map_err(|_| invalid("too many string frequency columns"))?,
7849        );
7850        for (column, texts) in table.frequency_texts.iter().enumerate() {
7851            if texts.is_empty() {
7852                continue;
7853            }
7854            put_u16(
7855                &mut out,
7856                u16::try_from(column).map_err(|_| invalid("frequency text column overflows"))?,
7857            );
7858            put_u16(
7859                &mut out,
7860                u16::try_from(texts.len())
7861                    .map_err(|_| invalid("too many frequency text entries"))?,
7862            );
7863            for text in texts {
7864                match text {
7865                    None => out.push(0),
7866                    Some(text) => {
7867                        out.push(1);
7868                        put_u32(
7869                            &mut out,
7870                            u32::try_from(text.len())
7871                                .map_err(|_| invalid("frequency text is too long"))?,
7872                        );
7873                        out.extend_from_slice(text);
7874                    }
7875                }
7876            }
7877        }
7878    }
7879    if let Some(summary) = &table.host_groups {
7880        out.extend_from_slice(HOST_GROUPS);
7881        put_u16(
7882            &mut out,
7883            u16::try_from(summary.column).map_err(|_| invalid("host column overflows"))?,
7884        );
7885        put_u64(&mut out, summary.omitted_max);
7886        put_u16(
7887            &mut out,
7888            u16::try_from(summary.entries.len()).map_err(|_| invalid("too many host groups"))?,
7889        );
7890        for entry in &summary.entries {
7891            put_u32(
7892                &mut out,
7893                u32::try_from(entry.host.len()).map_err(|_| invalid("host name is too long"))?,
7894            );
7895            out.extend_from_slice(entry.host.as_bytes());
7896            put_u64(&mut out, entry.count);
7897            out.extend_from_slice(&entry.bytes_sum.to_le_bytes());
7898            put_u32(
7899                &mut out,
7900                u32::try_from(entry.minimum.len())
7901                    .map_err(|_| invalid("host minimum is too long"))?,
7902            );
7903            out.extend_from_slice(entry.minimum.as_bytes());
7904        }
7905    }
7906    // Written only when there is a declaration, so that the common file is the same bytes it was
7907    // and the section is not a byte of zero on every table in the world that never asked for one.
7908    if let Some(clustering) = &table.clustering {
7909        out.extend_from_slice(CLUSTERING);
7910        out.push(clustering.width().tag());
7911        put_u16(
7912            &mut out,
7913            u16::try_from(clustering.columns().len())
7914                .map_err(|_| invalid("too many clustering columns"))?,
7915        );
7916        for &column in clustering.columns() {
7917            put_u16(
7918                &mut out,
7919                u16::try_from(column).map_err(|_| invalid("clustering column index overflow"))?,
7920            );
7921        }
7922    }
7923    let demoted = (0..table.fields.len())
7924        .filter(|&column| table.demoted.get(column).copied().unwrap_or(false))
7925        .collect::<Vec<_>>();
7926    if !demoted.is_empty() {
7927        out.extend_from_slice(DEMOTED);
7928        put_u16(
7929            &mut out,
7930            u16::try_from(demoted.len()).map_err(|_| invalid("too many demoted columns"))?,
7931        );
7932        for column in demoted {
7933            put_u16(
7934                &mut out,
7935                u16::try_from(column).map_err(|_| invalid("demoted column index overflow"))?,
7936            );
7937        }
7938    }
7939    // The section table, last, behind its own magic, for the same reason the frequency block is
7940    // behind its own: a reader that stops before it gets a table with no sections, and a table with
7941    // no sections is a correct table. The one difference from the blocks before it is that this one
7942    // is written even when it is empty, so that a file written by this build always says which
7943    // sections it has rather than leaving a reader to infer it from where the bytes ran out.
7944    out.extend_from_slice(SECTIONS);
7945    put_u64(&mut out, table.generation);
7946    put_u16(
7947        &mut out,
7948        u16::try_from(table.sections.len()).map_err(|_| invalid("too many sections"))?,
7949    );
7950    for held in &table.sections {
7951        held.encode(&mut out)?;
7952    }
7953    if table.dictionary_payloads.iter().any(|&bytes| bytes != 0) {
7954        out.extend_from_slice(DICTIONARY_PAYLOADS);
7955        put_u16(
7956            &mut out,
7957            u16::try_from(table.fields.len()).map_err(|_| invalid("too many columns"))?,
7958        );
7959        for at in 0..table.fields.len() {
7960            put_u64(&mut out, table.dictionary_payloads.get(at).copied().unwrap_or(0));
7961        }
7962    }
7963    Ok(out)
7964}
7965
7966/// The small level of the directory, naming every table in the file.
7967///
7968/// This is what a footer slot points at. Each entry carries its own checksum over its table
7969/// directory, so a table whose directory is torn is found when that table is first touched rather
7970/// than being trusted because the catalog around it checksummed.
7971///
7972/// The views go after the tables and are whole here, since a view is text and a column list and has
7973/// no pages for a second level to point at.
7974fn signed_integer(ty: &LogicalType) -> bool {
7975    matches!(
7976        ty,
7977        LogicalType::TinyInt | LogicalType::SmallInt | LogicalType::Integer | LogicalType::BigInt
7978    )
7979}
7980
7981fn integer_or_date(ty: &LogicalType) -> bool {
7982    matches!(
7983        ty,
7984        LogicalType::TinyInt
7985            | LogicalType::SmallInt
7986            | LogicalType::Integer
7987            | LogicalType::BigInt
7988            | LogicalType::UTinyInt
7989            | LogicalType::USmallInt
7990            | LogicalType::UInteger
7991            | LogicalType::UBigInt
7992            | LogicalType::Date
7993    )
7994}
7995
7996fn table_integer_extremes(table: &Table) -> Vec<StoredIntegerExtremes> {
7997    table
7998        .fields
7999        .iter()
8000        .enumerate()
8001        .map(|(column, field)| {
8002            if !integer_or_date(&field.ty) {
8003                return None;
8004            }
8005            let mut low: Option<i128> = None;
8006            let mut high: Option<i128> = None;
8007            for stripe in &table.stripes {
8008                let range = stripe.zone.column(column)?;
8009                if !range.exact {
8010                    return None;
8011                }
8012                match (range.low.as_ref(), range.high.as_ref()) {
8013                    (Some(Bound::Int(small)), Some(Bound::Int(large))) => {
8014                        low = Some(low.map_or(*small, |held| held.min(*small)));
8015                        high = Some(high.map_or(*large, |held| held.max(*large)));
8016                    }
8017                    (None, None) if stripe.rows == range.nulls => {}
8018                    _ => return None,
8019                }
8020            }
8021            Some(low.zip(high))
8022        })
8023        .collect()
8024}
8025
8026fn reader_integer_extremes(reader: &Reader) -> Result<Vec<StoredIntegerExtremes>> {
8027    reader
8028        .table
8029        .fields
8030        .iter()
8031        .enumerate()
8032        .map(|(column, field)| {
8033            if !integer_or_date(&field.ty) {
8034                return Ok(None);
8035            }
8036            match reader.exact_extremes(column)? {
8037                Some((Bound::Int(low), Bound::Int(high))) => Ok(Some(Some((low, high)))),
8038                None if reader.null_count(column)? == reader.table.rows as u64 => Ok(Some(None)),
8039                _ => Ok(None),
8040            }
8041        })
8042        .collect()
8043}
8044
8045fn table_complete_numeric_frequencies(table: &Table) -> Vec<StoredNumericFrequencies> {
8046    table
8047        .fields
8048        .iter()
8049        .enumerate()
8050        .map(|(column, field)| {
8051            if !integer_or_date(&field.ty) {
8052                return None;
8053            }
8054            let Some(Frequencies::Held(summary)) = table.frequencies.get(column)?.as_ref() else {
8055                return None;
8056            };
8057            if summary.omitted_max != 0 || summary.entries.len() > MAX_CATALOG_FREQUENCIES {
8058                return None;
8059            }
8060            let entries = summary
8061                .entries
8062                .iter()
8063                .map(|entry| {
8064                    let value = match entry.value {
8065                        FrequencyValue::Null => None,
8066                        FrequencyValue::Integer(value) => Some(value),
8067                        FrequencyValue::Code(_) => return None,
8068                    };
8069                    Some((value, entry.count))
8070                })
8071                .collect::<Option<Vec<_>>>()?;
8072            let rows = entries.iter().try_fold(0_u64, |sum, (_, count)| sum.checked_add(*count))?;
8073            (rows == table.rows as u64).then_some(entries)
8074        })
8075        .collect()
8076}
8077
8078/// The sixty four bits the close keys a numeric column's frequencies by, for a value the writer's
8079/// tally held.
8080///
8081/// The same bits [`Writer::visit_numeric`] hands over: a signed value sign extended to `i64`, and an
8082/// unsigned one as it is.
8083fn frequency_bits(value: &Value) -> Option<u64> {
8084    Some(match value {
8085        Value::TinyInt(value) => i64::from(*value) as u64,
8086        Value::SmallInt(value) => i64::from(*value) as u64,
8087        Value::Integer(value) | Value::Date(value) => i64::from(*value) as u64,
8088        Value::BigInt(value) | Value::Timestamp(value) => *value as u64,
8089        Value::UTinyInt(value) => u64::from(*value),
8090        Value::USmallInt(value) => u64::from(*value),
8091        Value::UInteger(value) => u64::from(*value),
8092        Value::UBigInt(value) => *value,
8093        _ => return None,
8094    })
8095}
8096
8097fn numeric_frequency_value(value: &Value) -> Option<Option<i128>> {
8098    Some(match value {
8099        Value::Null => None,
8100        Value::TinyInt(value) => Some(i128::from(*value)),
8101        Value::SmallInt(value) => Some(i128::from(*value)),
8102        Value::Integer(value) | Value::Date(value) => Some(i128::from(*value)),
8103        Value::BigInt(value) => Some(i128::from(*value)),
8104        Value::UTinyInt(value) => Some(i128::from(*value)),
8105        Value::USmallInt(value) => Some(i128::from(*value)),
8106        Value::UInteger(value) => Some(i128::from(*value)),
8107        Value::UBigInt(value) => Some(i128::from(*value)),
8108        _ => return None,
8109    })
8110}
8111
8112fn reader_complete_numeric_frequencies(reader: &Reader) -> Result<Vec<StoredNumericFrequencies>> {
8113    reader
8114        .table
8115        .fields
8116        .iter()
8117        .enumerate()
8118        .map(|(column, field)| {
8119            if !integer_or_date(&field.ty) {
8120                return Ok(None);
8121            }
8122            let Some(summary) = reader.frequency_summary(column)? else { return Ok(None) };
8123            if summary.omitted_max != 0 || summary.entries.len() > MAX_CATALOG_FREQUENCIES {
8124                return Ok(None);
8125            }
8126            let entries = reader.decode_frequencies(column, &field.ty, &summary.entries)?;
8127            let Some(entries) = entries
8128                .iter()
8129                .map(|(value, count)| Some((numeric_frequency_value(value)?, *count)))
8130                .collect::<Option<Vec<_>>>()
8131            else {
8132                return Ok(None);
8133            };
8134            let rows = entries.iter().try_fold(0_u64, |sum, (_, count)| sum.checked_add(*count));
8135            Ok((rows == Some(reader.table.rows as u64)).then_some(entries))
8136        })
8137        .collect()
8138}
8139
8140fn table_exact_sum(table: &Table, column: usize) -> Option<(i128, u64)> {
8141    table.stripes.iter().try_fold((0_i128, 0_u64), |(sum, count), stripe| {
8142        let range = stripe.zone.column(column)?;
8143        let sum = sum.checked_add(range.sum?)?;
8144        let nonnull = (stripe.rows as u64).checked_sub(range.nulls as u64)?;
8145        Some((sum, count.checked_add(nonnull)?))
8146    })
8147}
8148
8149fn table_aggregate_sums(table: &Table) -> Vec<Option<(i128, u64)>> {
8150    table
8151        .fields
8152        .iter()
8153        .enumerate()
8154        .map(|(column, field)| {
8155            signed_integer(&field.ty).then(|| table_exact_sum(table, column)).flatten()
8156        })
8157        .collect()
8158}
8159
8160fn reader_aggregate_sums(reader: &Reader) -> Result<Vec<Option<(i128, u64)>>> {
8161    reader
8162        .table
8163        .fields
8164        .iter()
8165        .enumerate()
8166        .map(
8167            |(column, field)| {
8168                if signed_integer(&field.ty) { reader.exact_sum(column) } else { Ok(None) }
8169            },
8170        )
8171        .collect()
8172}
8173
8174fn encode_catalog(entries: &[Entry], views: &[ViewEntry]) -> Result<Vec<u8>> {
8175    let mut out = CATALOG.to_vec();
8176    put_u32(&mut out, u32::try_from(entries.len()).map_err(|_| invalid("too many tables"))?);
8177    for entry in entries {
8178        let name = entry.name.as_bytes();
8179        put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("table name too long"))?);
8180        out.extend_from_slice(name);
8181        put_u64(&mut out, u64::try_from(entry.rows).map_err(|_| invalid("row count overflow"))?);
8182        put_u16(
8183            &mut out,
8184            u16::try_from(entry.fields.len()).map_err(|_| invalid("too many columns"))?,
8185        );
8186        for field in &entry.fields {
8187            let name = field.name.as_bytes();
8188            put_u16(
8189                &mut out,
8190                u16::try_from(name.len()).map_err(|_| invalid("column name too long"))?,
8191            );
8192            out.extend_from_slice(name);
8193            put_type(&mut out, &field.ty)?;
8194            out.push(u8::from(field.not_null));
8195        }
8196        put_u64(&mut out, entry.directory.offset);
8197        put_u32(&mut out, entry.directory.length);
8198        put_u64(&mut out, entry.directory.hash);
8199    }
8200    put_u32(&mut out, u32::try_from(views.len()).map_err(|_| invalid("too many views"))?);
8201    for view in views {
8202        let name = view.name.as_bytes();
8203        put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("view name too long"))?);
8204        out.extend_from_slice(name);
8205        put_long_text(&mut out, &view.sql, "view body")?;
8206        put_long_text(&mut out, &view.statement, "view statement")?;
8207        put_u16(
8208            &mut out,
8209            u16::try_from(view.aliases.len()).map_err(|_| invalid("too many aliases"))?,
8210        );
8211        for alias in &view.aliases {
8212            let alias = alias.as_bytes();
8213            put_u16(
8214                &mut out,
8215                u16::try_from(alias.len()).map_err(|_| invalid("alias name too long"))?,
8216            );
8217            out.extend_from_slice(alias);
8218        }
8219        put_u16(
8220            &mut out,
8221            u16::try_from(view.columns.len()).map_err(|_| invalid("too many columns"))?,
8222        );
8223        for field in &view.columns {
8224            let name = field.name.as_bytes();
8225            put_u16(
8226                &mut out,
8227                u16::try_from(name.len()).map_err(|_| invalid("column name too long"))?,
8228            );
8229            out.extend_from_slice(name);
8230            put_type(&mut out, &field.ty)?;
8231            out.push(u8::from(field.not_null));
8232        }
8233    }
8234    out.extend_from_slice(NONZERO_COUNTS);
8235    for entry in entries {
8236        if entry.nonzero.len() != entry.fields.len() {
8237            return Err(invalid("nonzero count width differs from schema"));
8238        }
8239        for count in &entry.nonzero {
8240            match count {
8241                None => out.push(0),
8242                Some(count) => {
8243                    out.push(1);
8244                    put_u64(&mut out, *count);
8245                }
8246            }
8247        }
8248    }
8249    out.extend_from_slice(AGGREGATE_SUMS);
8250    for entry in entries {
8251        if entry.aggregates.len() != entry.fields.len() {
8252            return Err(invalid("aggregate sum width differs from schema"));
8253        }
8254        for summary in &entry.aggregates {
8255            match summary {
8256                None => out.push(0),
8257                Some((sum, count)) => {
8258                    out.push(1);
8259                    out.extend_from_slice(&sum.to_le_bytes());
8260                    put_u64(&mut out, *count);
8261                }
8262            }
8263        }
8264    }
8265    out.extend_from_slice(DISTINCT_COUNTS);
8266    for entry in entries {
8267        if entry.distincts.len() != entry.fields.len() {
8268            return Err(invalid("distinct count width differs from schema"));
8269        }
8270        for count in &entry.distincts {
8271            match count {
8272                None => out.push(0),
8273                Some(count) => {
8274                    if *count > entry.rows as u64 {
8275                        return Err(invalid("distinct count exceeds table rows"));
8276                    }
8277                    out.push(1);
8278                    put_u64(&mut out, *count);
8279                }
8280            }
8281        }
8282    }
8283    out.extend_from_slice(INTEGER_EXTREMES);
8284    for entry in entries {
8285        if entry.extremes.len() != entry.fields.len() {
8286            return Err(invalid("integer extremes width differs from schema"));
8287        }
8288        for (field, extremes) in entry.fields.iter().zip(&entry.extremes) {
8289            match extremes {
8290                None => out.push(0),
8291                Some(None) if integer_or_date(&field.ty) => out.push(1),
8292                Some(Some((low, high))) if integer_or_date(&field.ty) && low <= high => {
8293                    out.push(2);
8294                    out.extend_from_slice(&low.to_le_bytes());
8295                    out.extend_from_slice(&high.to_le_bytes());
8296                }
8297                _ => return Err(invalid("integer extremes type or range differs")),
8298            }
8299        }
8300    }
8301    out.extend_from_slice(COMPLETE_FREQUENCIES);
8302    for entry in entries {
8303        if entry.frequencies.len() != entry.fields.len() {
8304            return Err(invalid("numeric frequency width differs from schema"));
8305        }
8306        for (field, frequencies) in entry.fields.iter().zip(&entry.frequencies) {
8307            match frequencies {
8308                None => out.push(0),
8309                Some(entries)
8310                    if integer_or_date(&field.ty) && entries.len() <= MAX_CATALOG_FREQUENCIES =>
8311                {
8312                    let mut total = 0_u64;
8313                    for (at, (value, count)) in entries.iter().enumerate() {
8314                        if entries[..at].iter().any(|(held, _)| held == value) {
8315                            return Err(invalid("numeric frequency value repeats"));
8316                        }
8317                        total = total
8318                            .checked_add(*count)
8319                            .ok_or_else(|| invalid("numeric frequency count overflows"))?;
8320                    }
8321                    if total != entry.rows as u64 {
8322                        return Err(invalid("numeric frequencies do not cover table rows"));
8323                    }
8324                    out.push(1);
8325                    out.push(entries.len() as u8);
8326                    for (value, count) in entries {
8327                        match value {
8328                            None => out.push(0),
8329                            Some(value) => {
8330                                out.push(1);
8331                                out.extend_from_slice(&value.to_le_bytes());
8332                            }
8333                        }
8334                        put_u64(&mut out, *count);
8335                    }
8336                }
8337                _ => return Err(invalid("numeric frequency type or width differs")),
8338            }
8339        }
8340    }
8341    Ok(out)
8342}
8343
8344/// A length and that many bytes, for text that is allowed to be longer than a name.
8345fn put_long_text(out: &mut Vec<u8>, text: &str, what: &str) -> Result<()> {
8346    let bytes = text.as_bytes();
8347    put_u32(out, u32::try_from(bytes.len()).map_err(|_| invalid(&format!("{what} too long")))?);
8348    out.extend_from_slice(bytes);
8349    Ok(())
8350}
8351
8352/// Reads the catalog directory back, checking every span against the file before anything is
8353/// allocated for it.
8354fn decode_catalog(bytes: &[u8], size: u64) -> Result<(Vec<Entry>, Vec<ViewEntry>)> {
8355    let mut cur = Cursor::new(bytes);
8356    if cur.take(8)? != CATALOG {
8357        return Err(invalid("catalog magic differs"));
8358    }
8359    let count = cur.u32()? as usize;
8360    let mut entries: Vec<Entry> = Vec::with_capacity(count.min(1024));
8361    for _ in 0..count {
8362        let name = cur.text()?;
8363        let rows = usize::try_from(cur.u64()?).map_err(|_| invalid("row count does not fit"))?;
8364        let width = cur.u16()? as usize;
8365        let mut fields = Vec::with_capacity(width);
8366        for _ in 0..width {
8367            let name = cur.text()?;
8368            let ty = read_type(&mut cur)?;
8369            let not_null = match cur.u8()? {
8370                0 => false,
8371                1 => true,
8372                _ => return Err(invalid("nullability flag differs")),
8373            };
8374            fields.push(Field { name, ty, not_null });
8375        }
8376        let directory = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
8377        let end = directory
8378            .offset
8379            .checked_add(u64::from(directory.length))
8380            .ok_or_else(|| invalid("table directory offset overflow"))?;
8381        if directory.offset < HEADER
8382            || end > size
8383            || directory.length as usize > MAX_DIRECTORY
8384            || directory.length == 0
8385        {
8386            return Err(invalid("table directory range is outside the file"));
8387        }
8388        if entries.iter().any(|held| held.name == name) {
8389            return Err(invalid("two tables in the catalog have the same name"));
8390        }
8391        let nonzero = vec![None; fields.len()];
8392        let aggregates = vec![None; fields.len()];
8393        let distincts = vec![None; fields.len()];
8394        let extremes = vec![None; fields.len()];
8395        let frequencies = vec![None; fields.len()];
8396        entries.push(Entry {
8397            name,
8398            fields,
8399            rows,
8400            directory,
8401            nonzero,
8402            aggregates,
8403            distincts,
8404            extremes,
8405            frequencies,
8406        });
8407    }
8408    // A catalog that ends where the tables end is a catalog with no views in it, which is every
8409    // file written before format 25. That is why the count is allowed to be missing rather than
8410    // read as a zero that has to be there: an older file has nothing after the last table entry at
8411    // all, and [`READABLE`] says those files still open.
8412    let count = if cur.done() { 0 } else { cur.u32()? as usize };
8413    let mut views: Vec<ViewEntry> = Vec::with_capacity(count.min(1024));
8414    for _ in 0..count {
8415        let name = cur.text()?;
8416        let sql = cur.long_text()?;
8417        let statement = cur.long_text()?;
8418        let width = cur.u16()? as usize;
8419        let mut aliases = Vec::with_capacity(width);
8420        for _ in 0..width {
8421            aliases.push(cur.text()?);
8422        }
8423        let width = cur.u16()? as usize;
8424        let mut columns = Vec::with_capacity(width);
8425        for _ in 0..width {
8426            let name = cur.text()?;
8427            let ty = read_type(&mut cur)?;
8428            let not_null = match cur.u8()? {
8429                0 => false,
8430                1 => true,
8431                _ => return Err(invalid("nullability flag differs")),
8432            };
8433            columns.push(Field { name, ty, not_null });
8434        }
8435        // The same rule the tables above get, and for the same reason. Two entries under one name
8436        // is a catalog nothing can answer a lookup from, and finding that out here is better than
8437        // finding it out from whichever of the two a search happened to reach first.
8438        if views.iter().any(|held| held.name == name) {
8439            return Err(invalid("two views in the catalog have the same name"));
8440        }
8441        if entries.iter().any(|held| held.name == name) {
8442            return Err(invalid("a table and a view in the catalog have the same name"));
8443        }
8444        views.push(ViewEntry { name, sql, statement, aliases, columns });
8445    }
8446    if !cur.done() {
8447        if cur.take(8)? != NONZERO_COUNTS {
8448            return Err(invalid("catalog extension magic differs"));
8449        }
8450        for entry in &mut entries {
8451            for (field, count) in entry.fields.iter().zip(&mut entry.nonzero) {
8452                *count = match cur.u8()? {
8453                    0 => None,
8454                    1 if matches!(
8455                        field.ty,
8456                        LogicalType::TinyInt
8457                            | LogicalType::SmallInt
8458                            | LogicalType::Integer
8459                            | LogicalType::BigInt
8460                            | LogicalType::UTinyInt
8461                            | LogicalType::USmallInt
8462                            | LogicalType::UInteger
8463                            | LogicalType::UBigInt
8464                    ) =>
8465                    {
8466                        let value = cur.u64()?;
8467                        if value > entry.rows as u64 {
8468                            return Err(invalid("nonzero count exceeds rows"));
8469                        }
8470                        Some(value)
8471                    }
8472                    _ => return Err(invalid("nonzero count tag or column type differs")),
8473                };
8474            }
8475        }
8476    }
8477    if !cur.done() {
8478        if cur.take(8)? != AGGREGATE_SUMS {
8479            return Err(invalid("aggregate catalog extension magic differs"));
8480        }
8481        for entry in &mut entries {
8482            for (field, summary) in entry.fields.iter().zip(&mut entry.aggregates) {
8483                *summary = match cur.u8()? {
8484                    0 => None,
8485                    1 if signed_integer(&field.ty) => {
8486                        let sum = i128::from_le_bytes(
8487                            cur.take(16)?
8488                                .try_into()
8489                                .map_err(|_| invalid("aggregate sum is truncated"))?,
8490                        );
8491                        let count = cur.u64()?;
8492                        if count > entry.rows as u64 {
8493                            return Err(invalid("aggregate count exceeds table rows"));
8494                        }
8495                        Some((sum, count))
8496                    }
8497                    _ => return Err(invalid("aggregate sum tag or column type differs")),
8498                };
8499            }
8500        }
8501    }
8502    if !cur.done() {
8503        if cur.take(8)? != DISTINCT_COUNTS {
8504            return Err(invalid("distinct catalog extension magic differs"));
8505        }
8506        for entry in &mut entries {
8507            for count in &mut entry.distincts {
8508                *count = match cur.u8()? {
8509                    0 => None,
8510                    1 => {
8511                        let value = cur.u64()?;
8512                        if value > entry.rows as u64 {
8513                            return Err(invalid("distinct count exceeds table rows"));
8514                        }
8515                        Some(value)
8516                    }
8517                    _ => return Err(invalid("distinct count tag differs")),
8518                };
8519            }
8520        }
8521    }
8522    if !cur.done() {
8523        if cur.take(8)? != INTEGER_EXTREMES {
8524            return Err(invalid("integer extremes catalog extension magic differs"));
8525        }
8526        for entry in &mut entries {
8527            for (field, extremes) in entry.fields.iter().zip(&mut entry.extremes) {
8528                *extremes = match cur.u8()? {
8529                    0 => None,
8530                    1 if integer_or_date(&field.ty) => Some(None),
8531                    2 if integer_or_date(&field.ty) => {
8532                        let low = i128::from_le_bytes(
8533                            cur.take(16)?
8534                                .try_into()
8535                                .map_err(|_| invalid("minimum is truncated"))?,
8536                        );
8537                        let high = i128::from_le_bytes(
8538                            cur.take(16)?
8539                                .try_into()
8540                                .map_err(|_| invalid("maximum is truncated"))?,
8541                        );
8542                        if low > high {
8543                            return Err(invalid("integer extremes are reversed"));
8544                        }
8545                        Some(Some((low, high)))
8546                    }
8547                    _ => return Err(invalid("integer extremes tag or type differs")),
8548                };
8549            }
8550        }
8551    }
8552    if !cur.done() {
8553        if cur.take(8)? != COMPLETE_FREQUENCIES {
8554            return Err(invalid("numeric frequency catalog extension magic differs"));
8555        }
8556        for entry in &mut entries {
8557            for (field, frequencies) in entry.fields.iter().zip(&mut entry.frequencies) {
8558                *frequencies = match cur.u8()? {
8559                    0 => None,
8560                    1 if integer_or_date(&field.ty) => {
8561                        let len = cur.u8()? as usize;
8562                        if len > MAX_CATALOG_FREQUENCIES {
8563                            return Err(invalid("too many catalog numeric frequencies"));
8564                        }
8565                        let mut values = Vec::with_capacity(len);
8566                        let mut total = 0_u64;
8567                        for _ in 0..len {
8568                            let value = match cur.u8()? {
8569                                0 => None,
8570                                1 => Some(i128::from_le_bytes(cur.take(16)?.try_into().map_err(
8571                                    |_| invalid("numeric frequency value is truncated"),
8572                                )?)),
8573                                _ => return Err(invalid("numeric frequency value tag differs")),
8574                            };
8575                            if values.iter().any(|(held, _)| *held == value) {
8576                                return Err(invalid("numeric frequency value repeats"));
8577                            }
8578                            let count = cur.u64()?;
8579                            total = total
8580                                .checked_add(count)
8581                                .ok_or_else(|| invalid("numeric frequency count overflows"))?;
8582                            values.push((value, count));
8583                        }
8584                        if total != entry.rows as u64 {
8585                            return Err(invalid("numeric frequencies do not cover table rows"));
8586                        }
8587                        Some(values)
8588                    }
8589                    _ => return Err(invalid("numeric frequency tag or type differs")),
8590                };
8591            }
8592        }
8593    }
8594    if !cur.done() {
8595        return Err(invalid("catalog has trailing bytes"));
8596    }
8597    Ok((entries, views))
8598}
8599
8600/// Reads the fields of a directory or a catalog in order, off bytes in memory or out of the file.
8601///
8602/// A catalog is small and is read whole. A table directory is not: at ten million rows of `hits` it
8603/// is nearly a megabyte, and holding that buffer while the table it describes is built out of it
8604/// put both at the peak of every query. Out of the file, the cursor holds one window of
8605/// [`DIRECTORY_WINDOW`] bytes and moves it forward as the fields are read, so what a directory
8606/// costs at open is what it decodes into and not that plus its own bytes.
8607struct Cursor<'a> {
8608    bytes: &'a [u8],
8609    at: usize,
8610    window: Option<Window<'a>>,
8611}
8612
8613/// The part of a directory in the file that a [`Cursor`] has read in.
8614struct Window<'a> {
8615    file: &'a File,
8616    offset: u64,
8617    length: usize,
8618    /// Where `held` starts, counted from the start of the directory.
8619    start: usize,
8620    held: Vec<u8>,
8621    /// How much to read at once, which is [`DIRECTORY_WINDOW`] outside the tests.
8622    size: usize,
8623}
8624
8625/// How much of a directory a cursor reading one out of the file holds at once.
8626const DIRECTORY_WINDOW: usize = 64 << 10;
8627
8628impl<'a> Cursor<'a> {
8629    fn new(bytes: &'a [u8]) -> Self {
8630        Self { bytes, at: 0, window: None }
8631    }
8632
8633    /// A cursor over `length` bytes of `file` from `offset`, which it reads a window at a time.
8634    fn over(file: &'a File, offset: u64, length: usize) -> Self {
8635        let window =
8636            Window { file, offset, length, start: 0, held: Vec::new(), size: DIRECTORY_WINDOW };
8637        Self { bytes: &[], at: 0, window: Some(window) }
8638    }
8639
8640    /// How many bytes the cursor walks in all.
8641    fn len(&self) -> usize {
8642        self.window.as_ref().map_or(self.bytes.len(), |window| window.length)
8643    }
8644
8645    /// Makes sure the next `len` bytes are in memory.
8646    fn ensure(&mut self, len: usize) -> Result<()> {
8647        let end = self.at.checked_add(len).ok_or_else(|| invalid("directory offset overflow"))?;
8648        if end > self.len() {
8649            return Err(invalid("directory is truncated"));
8650        }
8651        let Some(window) = &mut self.window else { return Ok(()) };
8652        if self.at < window.start || end > window.start + window.held.len() {
8653            let want = len.max(window.size).min(window.length - self.at);
8654            window.start = self.at;
8655            window.held.resize(want, 0);
8656            read_at(window.file, window.offset + self.at as u64, &mut window.held)?;
8657        }
8658        Ok(())
8659    }
8660
8661    /// `len` bytes from `at`, which [`Self::ensure`] has already brought in.
8662    fn held(&self, at: usize, len: usize) -> &[u8] {
8663        match &self.window {
8664            Some(window) => &window.held[at - window.start..at - window.start + len],
8665            None => &self.bytes[at..at + len],
8666        }
8667    }
8668
8669    /// The next `len` bytes, without moving past them.
8670    #[inline]
8671    fn peek(&mut self, len: usize) -> Result<&[u8]> {
8672        if self.window.is_none() {
8673            let bytes = self.bytes;
8674            return Ok(&bytes[self.at..self.end(len)?]);
8675        }
8676        self.ensure(len)?;
8677        Ok(self.held(self.at, len))
8678    }
8679
8680    /// The next `len` bytes, moving past them.
8681    ///
8682    /// Every data page is decoded through this, a byte or a word at a time, so a cursor over bytes
8683    /// already in memory takes them here and never reaches [`Self::ensure`]. With the window check
8684    /// on every call, q06 on TPC-H spent a seventh of its instructions in it.
8685    #[inline]
8686    fn take(&mut self, len: usize) -> Result<&[u8]> {
8687        if self.window.is_none() {
8688            let bytes = self.bytes;
8689            let (at, end) = (self.at, self.end(len)?);
8690            self.at = end;
8691            return Ok(&bytes[at..end]);
8692        }
8693        self.take_windowed(len)
8694    }
8695
8696    /// Moves over a checked field without reading its payload from a windowed directory.
8697    fn skip(&mut self, len: usize) -> Result<()> {
8698        let end = self.at.checked_add(len).ok_or_else(|| invalid("directory offset overflow"))?;
8699        if end > self.len() {
8700            return Err(invalid("directory is truncated"));
8701        }
8702        self.at = end;
8703        Ok(())
8704    }
8705
8706    fn skip_bound(&mut self) -> Result<()> {
8707        match self.u8()? {
8708            0 => Ok(()),
8709            1 => self.skip(16),
8710            2 => self.skip(8),
8711            3 => {
8712                let length = self.u32()? as usize;
8713                self.skip(length)
8714            }
8715            4 => self.skip(17),
8716            _ => Err(invalid("a stored bound has an unknown tag")),
8717        }
8718    }
8719
8720    /// Where `len` bytes from here end, when they end inside the bytes.
8721    #[inline]
8722    fn end(&self, len: usize) -> Result<usize> {
8723        let end = self.at.checked_add(len).ok_or_else(|| invalid("directory offset overflow"))?;
8724        if end > self.bytes.len() {
8725            return Err(invalid("directory is truncated"));
8726        }
8727        Ok(end)
8728    }
8729
8730    /// [`Self::take`] out of the file, a window at a time.
8731    #[inline(never)]
8732    fn take_windowed(&mut self, len: usize) -> Result<&[u8]> {
8733        self.ensure(len)?;
8734        self.at += len;
8735        Ok(self.held(self.at - len, len))
8736    }
8737    #[inline]
8738    fn u8(&mut self) -> Result<u8> {
8739        Ok(self.take(1)?[0])
8740    }
8741    #[inline]
8742    fn u16(&mut self) -> Result<u16> {
8743        Ok(u16::from_le_bytes(self.take(2)?.try_into().expect("two bytes")))
8744    }
8745    #[inline]
8746    fn u32(&mut self) -> Result<u32> {
8747        Ok(u32::from_le_bytes(self.take(4)?.try_into().expect("four bytes")))
8748    }
8749    #[inline]
8750    fn u64(&mut self) -> Result<u64> {
8751        Ok(u64::from_le_bytes(self.take(8)?.try_into().expect("eight bytes")))
8752    }
8753    fn var_u64(&mut self) -> Result<u64> {
8754        let mut value = 0_u64;
8755        for shift in (0..=63).step_by(7) {
8756            let byte = self.u8()?;
8757            let part = u64::from(byte & 0x7f);
8758            if shift == 63 && part > 1 {
8759                return Err(invalid("frequency ordinal varint overflows"));
8760            }
8761            value |= part << shift;
8762            if byte & 0x80 == 0 {
8763                return Ok(value);
8764            }
8765        }
8766        Err(invalid("frequency ordinal varint is too long"))
8767    }
8768    /// A zone map's end, in the layout `rudb_common::bounds` defines.
8769    ///
8770    /// The bytes are the ones this directory has written since format 10 and the codec moved to
8771    /// rank zero rather than being copied, because a column summary now writes the same two ends
8772    /// and two encodings of one type is how the two quietly stop agreeing.
8773    ///
8774    /// A bound's length is in the bound, so out of the file the cursor offers the codec a few bytes
8775    /// and offers it twice as many whenever it runs out before the directory does.
8776    fn bound(&mut self) -> Result<Option<Bound>> {
8777        let rest = self.len().saturating_sub(self.at);
8778        let mut want = 32;
8779        loop {
8780            let offered = self.peek(want.min(rest))?;
8781            let mut used = 0;
8782            match bounds::get(offered, &mut used) {
8783                Ok(bound) => {
8784                    self.at += used;
8785                    return Ok(bound);
8786                }
8787                Err(_) if want < rest => want *= 2,
8788                Err(error) => return Err(error),
8789            }
8790        }
8791    }
8792    fn text(&mut self) -> Result<String> {
8793        let len = self.u16()? as usize;
8794        String::from_utf8(self.take(len)?.to_vec()).map_err(|_| invalid("name is not UTF-8"))
8795    }
8796    /// Whether everything has been read, which is how a section that an older file does not have at
8797    /// all is told from one that is there and empty.
8798    fn done(&self) -> bool {
8799        self.at >= self.len()
8800    }
8801    /// The same, for text that is a query rather than a name.
8802    ///
8803    /// A name fits in sixteen bits of length and a view body does not have to. Nobody writes a 64
8804    /// kilobyte identifier by accident and people do write generated queries that long, and a view
8805    /// that could not be written down because its body was too big would be a limit invented here
8806    /// rather than one anything else in the engine has.
8807    fn long_text(&mut self) -> Result<String> {
8808        let len = self.u32()? as usize;
8809        String::from_utf8(self.take(len)?.to_vec()).map_err(|_| invalid("text is not UTF-8"))
8810    }
8811}
8812
8813/// One column's frequency synopsis, or `None` for a column that has none, checked against the column.
8814fn decode_summary(
8815    cur: &mut Cursor<'_>,
8816    field: &Field,
8817    rows: usize,
8818    values: bool,
8819) -> Result<Option<FrequencySummary>> {
8820    Ok(match cur.u8()? {
8821        0 => None,
8822        1 => {
8823            let omitted_max = cur.u64()?;
8824            let count = cur.u32()? as usize;
8825            if count > FREQUENCY_ENTRIES {
8826                return Err(invalid("frequency entry count exceeds its bound"));
8827            }
8828            let mut entries = Vec::with_capacity(count);
8829            // row at a time: directory decoding validates each persisted bounded frequency entry.
8830            for _ in 0..count {
8831                let value = match cur.u8()? {
8832                    0 => FrequencyValue::Null,
8833                    1 => FrequencyValue::Integer(i128::from_le_bytes(
8834                        cur.take(16)?.try_into().expect("sixteen bytes"),
8835                    )),
8836                    2 => FrequencyValue::Code(cur.u32()?),
8837                    _ => return Err(invalid("frequency value tag differs")),
8838                };
8839                let valid = matches!(
8840                    (&field.ty, value),
8841                    (_, FrequencyValue::Null)
8842                        | (LogicalType::Varchar | LogicalType::Blob, FrequencyValue::Code(_))
8843                        | (
8844                            LogicalType::TinyInt
8845                                | LogicalType::SmallInt
8846                                | LogicalType::Integer
8847                                | LogicalType::BigInt
8848                                | LogicalType::UTinyInt
8849                                | LogicalType::USmallInt
8850                                | LogicalType::UInteger
8851                                | LogicalType::UBigInt
8852                                | LogicalType::Date
8853                                | LogicalType::Timestamp,
8854                            FrequencyValue::Integer(_),
8855                        )
8856                );
8857                if !valid {
8858                    return Err(invalid("frequency value does not match its column"));
8859                }
8860                let count = cur.u64()?;
8861                if count == 0 || count > rows as u64 {
8862                    return Err(invalid("frequency count is outside the table"));
8863                }
8864                entries.push(FrequencyEntry { value, count });
8865            }
8866            if entries.windows(2).any(|pair| pair[0].count < pair[1].count) {
8867                return Err(invalid("frequency entries are not descending"));
8868            }
8869            let ordinals = {
8870                let ordinal_count = cur.u32()? as usize;
8871                if ordinal_count > FREQUENCY_ORDINALS || ordinal_count > rows {
8872                    return Err(invalid("frequency ordinal count exceeds its bound"));
8873                }
8874                let mut ordinals = Vec::with_capacity(ordinal_count);
8875                let mut previous = 0_u64;
8876                for at in 0..ordinal_count {
8877                    let delta = cur.var_u64()?;
8878                    if at != 0 && delta == 0 {
8879                        return Err(invalid("frequency ordinals are not increasing"));
8880                    }
8881                    let ordinal = if at == 0 {
8882                        delta
8883                    } else {
8884                        previous
8885                            .checked_add(delta)
8886                            .ok_or_else(|| invalid("frequency ordinal overflows"))?
8887                    };
8888                    if ordinal >= rows as u64 {
8889                        return Err(invalid("frequency ordinal is outside the table"));
8890                    }
8891                    ordinals.push(ordinal);
8892                    previous = ordinal;
8893                }
8894                ordinals
8895            };
8896            let ordinal_entries = if values {
8897                let mut ordinal_entries = Vec::with_capacity(ordinals.len());
8898                for _ in 0..ordinals.len() {
8899                    let entry = cur.u16()?;
8900                    if entry as usize >= entries.len() {
8901                        return Err(invalid("frequency ordinal value is outside its entries"));
8902                    }
8903                    ordinal_entries.push(entry);
8904                }
8905                ordinal_entries
8906            } else {
8907                Vec::new()
8908            };
8909            Some(FrequencySummary { entries, omitted_max, ordinals, ordinal_entries })
8910        }
8911        _ => return Err(invalid("frequency summary tag differs")),
8912    })
8913}
8914
8915/// Skips a synopsis whose column the caller does not need. The directory checksum was checked
8916/// before this walk, and the fields still need their lengths and tags checked to find the next one.
8917fn skip_summary(cur: &mut Cursor<'_>, values: bool, rows: usize) -> Result<()> {
8918    match cur.u8()? {
8919        0 => Ok(()),
8920        1 => {
8921            cur.skip(8)?;
8922            let entries = cur.u32()? as usize;
8923            if entries > FREQUENCY_ENTRIES {
8924                return Err(invalid("frequency entry count exceeds its bound"));
8925            }
8926            for _ in 0..entries {
8927                match cur.u8()? {
8928                    0 => {}
8929                    1 => cur.skip(16)?,
8930                    2 => cur.skip(4)?,
8931                    _ => return Err(invalid("frequency value tag differs")),
8932                }
8933                cur.skip(8)?;
8934            }
8935            let ordinals = cur.u32()? as usize;
8936            if ordinals > FREQUENCY_ORDINALS || ordinals > rows {
8937                return Err(invalid("frequency ordinal count exceeds its bound"));
8938            }
8939            for _ in 0..ordinals {
8940                cur.var_u64()?;
8941            }
8942            if values {
8943                cur.skip(ordinals * 2)?;
8944            }
8945            Ok(())
8946        }
8947        _ => Err(invalid("frequency summary tag differs")),
8948    }
8949}
8950
8951/// Reads only the catalog, stripe null counts, and one frequency synopsis. This is the cold path
8952/// for a summary-backed count; constructing every page descriptor and zone map would make it cost
8953/// the size of the table directory even when no row is read.
8954fn quick_nonzero(
8955    mut cur: Cursor<'_>,
8956    name: &str,
8957    fields: &[Field],
8958    rows: usize,
8959    wanted: usize,
8960) -> Result<Option<u64>> {
8961    if cur.take(8)? != DIRECTORY || cur.text()? != name {
8962        return Err(invalid("table directory differs from the catalog"));
8963    }
8964    let width = cur.u16()? as usize;
8965    if width != fields.len() {
8966        return Err(invalid("table directory width differs from the catalog"));
8967    }
8968    for field in fields {
8969        let stored =
8970            Field { name: cur.text()?, ty: read_type(&mut cur)?, not_null: cur.u8()? != 0 };
8971        if &stored != field {
8972            return Err(invalid("table directory schema differs from the catalog"));
8973        }
8974    }
8975    let mut dictionaries = Vec::with_capacity(width);
8976    for field in fields {
8977        let held = match cur.u8()? {
8978            0 => false,
8979            tag if coded_type(&field.ty) && tag == dictionary_tag(&field.ty) => {
8980                cur.skip(20)?;
8981                true
8982            }
8983            _ => return Err(invalid("dictionary page tag differs")),
8984        };
8985        dictionaries.push(held);
8986    }
8987    for _ in 0..width {
8988        match cur.u8()? {
8989            0 => {}
8990            1 => cur.skip(8)?,
8991            _ => return Err(invalid("distinct count tag differs")),
8992        }
8993    }
8994    if cur.u64()? != rows as u64 {
8995        return Err(invalid("table row count differs from the catalog"));
8996    }
8997    let stripes = cur.u32()? as usize;
8998    let mut total = 0_usize;
8999    let mut nulls = 0_u64;
9000    for _ in 0..stripes {
9001        let parts = cur.u32()? as usize;
9002        if parts == 0 || parts > STRIPE_PARTS {
9003            return Err(invalid("stripe part count is outside its bound"));
9004        }
9005        let mut stripe_rows = 0_usize;
9006        for _ in 0..parts {
9007            stripe_rows = stripe_rows
9008                .checked_add(cur.u32()? as usize)
9009                .ok_or_else(|| invalid("stripe row count overflow"))?;
9010        }
9011        total =
9012            total.checked_add(stripe_rows).ok_or_else(|| invalid("stripe row count overflow"))?;
9013        cur.skip(12 + width * 12)?;
9014        for (field, held) in fields.iter().zip(&dictionaries) {
9015            if coded_type(&field.ty) && *held {
9016                cur.skip(20)?;
9017            }
9018        }
9019        for _ in 0..width * 2 {
9020            match cur.u8()? {
9021                0 => {}
9022                1 => cur.skip(20)?,
9023                _ => return Err(invalid("stripe page tag differs")),
9024            }
9025        }
9026        for column in 0..width {
9027            cur.skip_bound()?;
9028            cur.skip_bound()?;
9029            let count = cur.u32()? as u64;
9030            if count > stripe_rows as u64 {
9031                return Err(invalid("null count exceeds stripe rows"));
9032            }
9033            if column == wanted {
9034                nulls = nulls.checked_add(count).ok_or_else(|| invalid("null count overflow"))?;
9035            }
9036            cur.skip(1)?;
9037            match cur.u8()? {
9038                0 => {}
9039                1 => cur.skip(16)?,
9040                _ => return Err(invalid("a stripe sum has an unknown tag")),
9041            }
9042        }
9043    }
9044    if total != rows {
9045        return Err(invalid("table row count differs from stripes"));
9046    }
9047    if cur.done() {
9048        return Ok(None);
9049    }
9050    let magic = cur.take(8)?;
9051    let values = magic == FREQUENCIES;
9052    if !values && magic != FREQUENCIES_V2 {
9053        return Err(invalid("directory extension magic differs"));
9054    }
9055    if cur.u16()? as usize != width {
9056        return Err(invalid("frequency column count differs"));
9057    }
9058    for _ in 0..wanted {
9059        skip_summary(&mut cur, values, rows)?;
9060    }
9061    let Some(summary) = decode_summary(&mut cur, &fields[wanted], rows, values)? else {
9062        return Ok(None);
9063    };
9064    let zero = summary
9065        .entries
9066        .iter()
9067        .find(|entry| entry.value == FrequencyValue::Integer(0))
9068        .map(|entry| entry.count)
9069        .or_else(|| (summary.omitted_max == 0).then_some(0));
9070    Ok(zero.and_then(|zero| (rows as u64).checked_sub(nulls)?.checked_sub(zero)))
9071}
9072
9073/// Walks the row-oriented directory while retaining only one column's index and page spans.
9074/// The catalog supplies the schema and the caller checks the complete directory checksum first.
9075fn quick_integer_fold(
9076    file: &File,
9077    mut cur: Cursor<'_>,
9078    entry: &Entry,
9079    size: u64,
9080    wanted: usize,
9081    emit: &mut impl FnMut(i64, u64) -> Result<()>,
9082) -> Result<()> {
9083    let name = &entry.name;
9084    let fields = &entry.fields;
9085    let rows = entry.rows;
9086    if cur.take(8)? != DIRECTORY || cur.text()? != name.as_str() {
9087        return Err(invalid("table directory differs from the catalog"));
9088    }
9089    let width = cur.u16()? as usize;
9090    if width != fields.len() {
9091        return Err(invalid("table directory width differs from the catalog"));
9092    }
9093    for field in fields {
9094        let stored =
9095            Field { name: cur.text()?, ty: read_type(&mut cur)?, not_null: cur.u8()? != 0 };
9096        if &stored != field {
9097            return Err(invalid("table directory schema differs from the catalog"));
9098        }
9099    }
9100    let mut dictionaries = Vec::with_capacity(width);
9101    for field in fields {
9102        dictionaries.push(match cur.u8()? {
9103            0 => false,
9104            tag if coded_type(&field.ty) && tag == dictionary_tag(&field.ty) => {
9105                cur.skip(20)?;
9106                true
9107            }
9108            _ => return Err(invalid("dictionary page tag differs")),
9109        });
9110    }
9111    for _ in 0..width {
9112        match cur.u8()? {
9113            0 => {}
9114            1 => cur.skip(8)?,
9115            _ => return Err(invalid("distinct count tag differs")),
9116        }
9117    }
9118    if cur.u64()? != rows as u64 {
9119        return Err(invalid("table row count differs from the catalog"));
9120    }
9121    let stripes = cur.u32()? as usize;
9122    let mut total = 0_usize;
9123    let mut bytes = Vec::new();
9124    for _ in 0..stripes {
9125        let parts = cur.u32()? as usize;
9126        if parts == 0 || parts > STRIPE_PARTS {
9127            return Err(invalid("stripe part count is outside its bound"));
9128        }
9129        let mut part_rows = Vec::with_capacity(parts);
9130        for _ in 0..parts {
9131            let count = cur.u32()? as usize;
9132            if count == 0 {
9133                return Err(invalid("empty part"));
9134            }
9135            total = total.checked_add(count).ok_or_else(|| invalid("stripe row count overflow"))?;
9136            part_rows.push(count);
9137        }
9138        let index = Span { offset: cur.u64()?, length: cur.u32()? };
9139        let section = index_section(parts)?;
9140        let index_length =
9141            section.checked_mul(width).ok_or_else(|| invalid("index page length overflow"))?;
9142        if index.offset < HEADER
9143            || index.offset.checked_add(u64::from(index.length)).is_none_or(|end| end > size)
9144            || index.length as usize != index_length
9145        {
9146            return Err(invalid("index page range is outside the file"));
9147        }
9148        cur.skip(wanted * 12)?;
9149        let page = Span { offset: cur.u64()?, length: cur.u32()? };
9150        if page.offset < HEADER
9151            || page.offset.checked_add(u64::from(page.length)).is_none_or(|end| end > size)
9152            || page.length as usize > MAX_PAGE
9153        {
9154            return Err(invalid("column page range is outside the file"));
9155        }
9156        cur.skip((width - wanted - 1) * 12)?;
9157        for (field, held) in fields.iter().zip(&dictionaries) {
9158            if coded_type(&field.ty) && *held {
9159                cur.skip(20)?;
9160            }
9161        }
9162        for _ in 0..width * 2 {
9163            match cur.u8()? {
9164                0 => {}
9165                1 => cur.skip(20)?,
9166                _ => return Err(invalid("stripe page tag differs")),
9167            }
9168        }
9169        for _ in 0..width {
9170            cur.skip_bound()?;
9171            cur.skip_bound()?;
9172            cur.skip(5)?;
9173            match cur.u8()? {
9174                0 => {}
9175                1 => cur.skip(16)?,
9176                _ => return Err(invalid("a stripe sum has an unknown tag")),
9177            }
9178        }
9179        let spans = read_index_span(file, index, page, parts, wanted)?;
9180        for (span, expected_rows) in spans.into_iter().zip(part_rows) {
9181            bytes.resize(span.length, 0);
9182            let at = page
9183                .offset
9184                .checked_add(span.start as u64)
9185                .ok_or_else(|| invalid("part range overflow"))?;
9186            read_at(file, at, &mut bytes)?;
9187            if checksum(&bytes) != span.hash {
9188                return Err(invalid("integer part checksum differs"));
9189            }
9190            if bytes.first() == Some(&5) && bytes.get(1) == Some(&0) {
9191                let decoded_rows = integer::fold(&bytes[2..], |value, count| {
9192                    check_integer_tally_value(value, &fields[wanted].ty)?;
9193                    emit(value, count)
9194                })?;
9195                if decoded_rows != expected_rows {
9196                    return Err(invalid("encoded integer part holds the wrong number of rows"));
9197                }
9198            } else {
9199                let column = decode(&fields[wanted].ty, expected_rows, &bytes, None)?;
9200                if let Some(packed) = column.packed_parts() {
9201                    let validity = column.validity();
9202                    let all_valid = column.none_null();
9203                    let base = packed.base();
9204                    let mut codes = [0_u64; 64];
9205                    for from in (0..expected_rows).step_by(codes.len()) {
9206                        let count = (expected_rows - from).min(codes.len());
9207                        packed.unpack(from, &mut codes[..count]);
9208                        for (offset, &code) in codes[..count].iter().enumerate() {
9209                            if all_valid || validity.is_valid(from + offset) {
9210                                // Vector::packed checked that this entire range fits the type.
9211                                emit((base + i128::from(code)) as i64, 1)?;
9212                            }
9213                        }
9214                    }
9215                    continue;
9216                }
9217                let column = column.into_flat()?;
9218                let validity = column.validity();
9219                macro_rules! count_decoded {
9220                    ($values:expr) => {
9221                        for (row, &value) in $values.as_slice().iter().enumerate() {
9222                            if validity.is_valid(row) {
9223                                emit(i64::from(value), 1)?;
9224                            }
9225                        }
9226                    };
9227                }
9228                match column.data() {
9229                    Some(Data::Int8(values)) => count_decoded!(values),
9230                    Some(Data::Int16(values)) => count_decoded!(values),
9231                    Some(Data::Int32(values)) => count_decoded!(values),
9232                    Some(Data::Int64(values)) => count_decoded!(values),
9233                    _ => return Err(invalid("decoded integer part has the wrong type")),
9234                }
9235            }
9236        }
9237    }
9238    if total != rows {
9239        return Err(invalid("table row count differs from stripes"));
9240    }
9241    Ok(())
9242}
9243
9244fn check_integer_tally_value(value: i64, ty: &LogicalType) -> Result<()> {
9245    let fits = match ty {
9246        LogicalType::TinyInt => i8::try_from(value).is_ok(),
9247        LogicalType::SmallInt => i16::try_from(value).is_ok(),
9248        LogicalType::Integer => i32::try_from(value).is_ok(),
9249        LogicalType::BigInt => true,
9250        _ => false,
9251    };
9252    if fits { Ok(()) } else { Err(invalid("encoded integer value is outside its column type")) }
9253}
9254
9255fn decode_directory(bytes: &[u8], size: u64) -> Result<Table> {
9256    read_directory(Cursor::new(bytes), size, None)
9257}
9258
9259/// A directory out of `cur`, which is a whole one in memory or one being read out of the file.
9260///
9261/// `stored_at` is where the directory starts in the file when it is being read out of it, and then
9262/// every frequency synopsis is checked and left there, as [`Frequencies::Stored`].
9263fn read_directory(mut cur: Cursor<'_>, size: u64, stored_at: Option<u64>) -> Result<Table> {
9264    if cur.take(8)? != DIRECTORY {
9265        return Err(invalid("directory magic differs"));
9266    }
9267    let name = cur.text()?;
9268    let width = cur.u16()? as usize;
9269    let mut fields = Vec::with_capacity(width);
9270    for _ in 0..width {
9271        let name = cur.text()?;
9272        let ty = read_type(&mut cur)?;
9273        let not_null = match cur.u8()? {
9274            0 => false,
9275            1 => true,
9276            _ => return Err(invalid("nullability flag differs")),
9277        };
9278        fields.push(Field { name, ty, not_null });
9279    }
9280    let mut dictionaries = Vec::with_capacity(width);
9281    for field in &fields {
9282        dictionaries.push(match cur.u8()? {
9283            0 => None,
9284            tag if tag == dictionary_tag(&field.ty) => {
9285                let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
9286                let end = page
9287                    .offset
9288                    .checked_add(u64::from(page.length))
9289                    .ok_or_else(|| invalid("dictionary page offset overflow"))?;
9290                // A global dictionary covers a whole column, not one bounded stripe. Its lazy
9291                // payload is intentionally allowed to grow past `MAX_PAGE`; only ordinary column
9292                // pages are capped there. `Writer::finish` has already bounded this length by the
9293                // on-disk `u32`, and the range check below keeps it inside the file.
9294                if page.offset < HEADER || end > size {
9295                    return Err(invalid("dictionary page range is outside the file"));
9296                }
9297                Some(page)
9298            }
9299            _ => return Err(invalid("dictionary page tag differs")),
9300        });
9301    }
9302    let mut distincts = Vec::with_capacity(width);
9303    for _ in 0..width {
9304        distincts.push(match cur.u8()? {
9305            0 => None,
9306            1 => Some(cur.u64()?),
9307            _ => return Err(invalid("distinct count tag differs")),
9308        });
9309    }
9310    let rows = usize::try_from(cur.u64()?).map_err(|_| invalid("row count does not fit"))?;
9311    let count = cur.u32()? as usize;
9312    let mut stripes = Vec::with_capacity(count);
9313    let mut total = 0_usize;
9314    for _ in 0..count {
9315        let count = cur.u32()? as usize;
9316        if count == 0 || count > STRIPE_PARTS {
9317            return Err(invalid("stripe part count is outside its bound"));
9318        }
9319        let mut parts = Vec::with_capacity(count);
9320        let mut stripe_rows = 0_usize;
9321        for _ in 0..count {
9322            let rows = cur.u32()?;
9323            if rows == 0 {
9324                return Err(invalid("empty part"));
9325            }
9326            parts.push(rows);
9327            stripe_rows = stripe_rows
9328                .checked_add(rows as usize)
9329                .ok_or_else(|| invalid("stripe row count overflow"))?;
9330        }
9331        total =
9332            total.checked_add(stripe_rows).ok_or_else(|| invalid("stripe row count overflow"))?;
9333        let index = Span { offset: cur.u64()?, length: cur.u32()? };
9334        let section = index_section(count)?;
9335        let wanted = section
9336            .checked_mul(width)
9337            .and_then(|bytes| u32::try_from(bytes).ok())
9338            .ok_or_else(|| invalid("index page length overflow"))?;
9339        let end = index
9340            .offset
9341            .checked_add(u64::from(index.length))
9342            .ok_or_else(|| invalid("index page offset overflow"))?;
9343        if index.offset < HEADER || end > size || index.length != wanted {
9344            return Err(invalid("index page range is outside the file"));
9345        }
9346        let mut pages = Vec::with_capacity(width);
9347        for _ in 0..width {
9348            let offset = cur.u64()?;
9349            let length = cur.u32()?;
9350            let end = offset
9351                .checked_add(u64::from(length))
9352                .ok_or_else(|| invalid("page offset overflow"))?;
9353            if offset < HEADER || end > size || length as usize > MAX_PAGE {
9354                return Err(invalid("page range is outside the file"));
9355            }
9356            pages.push(Span { offset, length });
9357        }
9358        let mut memberships = vec![None; width];
9359        for (column, field) in fields.iter().enumerate() {
9360            if !coded_type(&field.ty) || dictionaries[column].is_none() {
9361                continue;
9362            }
9363            let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
9364            let end = page
9365                .offset
9366                .checked_add(u64::from(page.length))
9367                .ok_or_else(|| invalid("membership page offset overflow"))?;
9368            if page.offset < HEADER || end > size || page.length as usize > MAX_PAGE {
9369                return Err(invalid("membership page range is outside the file"));
9370            }
9371            // No bytes is a stripe written after the column's dictionary was demoted, see
9372            // [`DEMOTED`], which is checked once the block that says so has been read.
9373            if page.length != 0 {
9374                memberships[column] = Some(page);
9375            }
9376        }
9377        let mut sieves = vec![None; width];
9378        for sieve in sieves.iter_mut().take(width) {
9379            match cur.u8()? {
9380                0 => continue,
9381                1 => {}
9382                _ => return Err(invalid("a sieve page has an unknown tag")),
9383            }
9384            let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
9385            let end = page
9386                .offset
9387                .checked_add(u64::from(page.length))
9388                .ok_or_else(|| invalid("sieve page offset overflow"))?;
9389            if page.offset < HEADER || end > size || page.length as usize > MAX_PAGE {
9390                return Err(invalid("sieve page range is outside the file"));
9391            }
9392            *sieve = Some(page);
9393        }
9394        let mut part_ranges = vec![None; width];
9395        for held in part_ranges.iter_mut().take(width) {
9396            match cur.u8()? {
9397                0 => continue,
9398                1 => {}
9399                _ => return Err(invalid("a part range page has an unknown tag")),
9400            }
9401            let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
9402            let end = page
9403                .offset
9404                .checked_add(u64::from(page.length))
9405                .ok_or_else(|| invalid("part range page offset overflow"))?;
9406            if page.offset < HEADER || end > size || page.length as usize > MAX_PAGE {
9407                return Err(invalid("part range page range is outside the file"));
9408            }
9409            *held = Some(page);
9410        }
9411        let mut ranges = Vec::with_capacity(width);
9412        for column in 0..width {
9413            let low = cur.bound()?;
9414            let high = cur.bound()?;
9415            let nulls = cur.u32()? as usize;
9416            if nulls > stripe_rows {
9417                return Err(invalid("null count exceeds stripe rows"));
9418            }
9419            let exact = cur.u8()? != 0;
9420            let sum = match cur.u8()? {
9421                0 => None,
9422                1 => Some(i128::from_le_bytes(
9423                    cur.take(16)?.try_into().map_err(|_| invalid("a stripe sum is truncated"))?,
9424                )),
9425                _ => return Err(invalid("a stripe sum has an unknown tag")),
9426            };
9427            // Files written before the ends of a decimal or a timestamp column carried their power
9428            // of ten hold a bare integer here, and that integer is the one the column holds, which
9429            // is what the power is over. So the type puts it back on the way in and an old file
9430            // prunes as well as a new one. A file that already wrote the power keeps it, because
9431            // this leaves anything that is not an integer alone.
9432            let ty = &fields.get(column).ok_or_else(|| invalid("a stripe range has no column"))?.ty;
9433            let low = low.map(|bound| scaled_as(bound, ty));
9434            let high = high.map(|bound| scaled_as(bound, ty));
9435            ranges.push(Range { low, high, nulls, exact, sum });
9436        }
9437        stripes.push(Stripe {
9438            rows: stripe_rows,
9439            parts,
9440            index,
9441            pages,
9442            memberships: Pages::from_slots(memberships)?,
9443            sieves: Pages::from_slots(sieves)?,
9444            part_ranges: Pages::from_slots(part_ranges)?,
9445            zone: Zone::from_ranges(ranges),
9446        });
9447    }
9448    if total != rows {
9449        return Err(invalid("table row count differs from stripes"));
9450    }
9451    // How many entries each column's synopsis lists, which is all a pair summary is checked against,
9452    // kept apart because the synopses themselves may be left in the file.
9453    let mut entry_counts = vec![0; width];
9454    let frequencies = if cur.done() {
9455        vec![None; width]
9456    } else {
9457        let frequency_magic = cur.take(8)?;
9458        let frequency_values = frequency_magic == FREQUENCIES;
9459        if !frequency_values && frequency_magic != FREQUENCIES_V2 {
9460            return Err(invalid("directory extension magic differs"));
9461        }
9462        if cur.u16()? as usize != width {
9463            return Err(invalid("frequency column count differs"));
9464        }
9465        let mut frequencies = Vec::with_capacity(width);
9466        for (field, entry_count) in fields.iter().zip(&mut entry_counts) {
9467            let start = cur.at;
9468            let summary = decode_summary(&mut cur, field, rows, frequency_values)?;
9469            *entry_count = summary.as_ref().map_or(0, |summary| summary.entries.len());
9470            frequencies.push(match (summary, stored_at) {
9471                (None, _) => None,
9472                (Some(summary), None) => Some(Frequencies::Held(summary)),
9473                (Some(_), Some(offset)) => Some(Frequencies::Stored {
9474                    span: Span {
9475                        offset: offset + start as u64,
9476                        length: u32::try_from(cur.at - start)
9477                            .map_err(|_| invalid("a frequency synopsis is too long"))?,
9478                    },
9479                    values: frequency_values,
9480                }),
9481            });
9482        }
9483        frequencies
9484    };
9485    // There are two optional trailing blocks now rather than one, so the reader dispatches on the
9486    // magic it finds rather than on where the bytes ran out. That is what lets the two arrive
9487    // independently: a format 22 directory ends here and has neither, a directory written before
9488    // the section table has only the clustering declaration, and each one still opens without a
9489    // rewrite. It is the G1 exit criterion, which is that a reader that knows about sections opens
9490    // a file that predates them and answers every query, only without the graph path.
9491    //
9492    // A repeated block is refused rather than allowed to win, because two clustering declarations
9493    // in one directory is a torn directory and the only question is which of them is the lie.
9494    let mut clustering = None;
9495    let mut sections = Vec::new();
9496    let mut pair_frequencies = Vec::new();
9497    let mut seen_pair_frequencies = false;
9498    let mut frequency_texts = vec![Vec::new(); width];
9499    let mut seen_frequency_texts = false;
9500    let mut host_groups = None;
9501    let mut demoted = Vec::new();
9502    let mut seen_sections = false;
9503    let mut dictionary_payloads = Vec::new();
9504    let mut seen_payloads = false;
9505    // Zero until a section table says otherwise, which is what a format 22 table gets and what
9506    // makes every section stamp fail to match on one, because real generations start at one.
9507    let mut generation = 0;
9508    while !cur.done() {
9509        let mut tag = [0u8; 8];
9510        tag.copy_from_slice(cur.take(8)?);
9511        if &tag == PAIR_FREQUENCIES {
9512            if seen_pair_frequencies {
9513                return Err(invalid("directory names two pair frequency blocks"));
9514            }
9515            seen_pair_frequencies = true;
9516            let count = cur.u16()? as usize;
9517            if count > MAX_PAIR_FREQUENCIES {
9518                return Err(invalid("pair frequency count exceeds its bound"));
9519            }
9520            pair_frequencies = Vec::with_capacity(count);
9521            for _ in 0..count {
9522                let first = cur.u16()?;
9523                let second = cur.u16()?;
9524                let first_at = first as usize;
9525                let second_at = second as usize;
9526                if frequencies.get(first_at).and_then(Option::as_ref).is_none() {
9527                    return Err(invalid("pair frequency first column has no synopsis"));
9528                }
9529                let first_entries = entry_counts[first_at];
9530                if !matches!(fields.get(second_at), Some(field) if field.ty == LogicalType::Varchar)
9531                    || dictionaries.get(second_at).copied().flatten().is_none()
9532                {
9533                    return Err(invalid("pair frequency second column has no stable dictionary"));
9534                }
9535                if pair_frequencies
9536                    .iter()
9537                    .any(|held: &PairFrequencySummary| held.first == first && held.second == second)
9538                {
9539                    return Err(invalid("directory repeats a pair frequency summary"));
9540                }
9541                let omitted_max = cur.u64()?;
9542                if omitted_max > rows as u64 {
9543                    return Err(invalid("pair frequency omitted count exceeds the table"));
9544                }
9545                let entries_count = cur.u16()? as usize;
9546                if entries_count > FREQUENCY_ENTRIES {
9547                    return Err(invalid("pair frequency entry count exceeds its bound"));
9548                }
9549                let mut entries = Vec::with_capacity(entries_count);
9550                for _ in 0..entries_count {
9551                    let first_entry = cur.u16()?;
9552                    if first_entry as usize >= first_entries {
9553                        return Err(invalid("pair frequency anchor is outside its synopsis"));
9554                    }
9555                    let second = match cur.u8()? {
9556                        0 => None,
9557                        1 => Some(cur.u32()?),
9558                        _ => return Err(invalid("pair frequency string tag differs")),
9559                    };
9560                    let count = cur.u64()?;
9561                    if count == 0 || count > rows as u64 {
9562                        return Err(invalid("pair frequency count is outside the table"));
9563                    }
9564                    entries.push(PairFrequencyEntry { first_entry, second, count });
9565                }
9566                if entries.windows(2).any(|pair| pair[0].count < pair[1].count) {
9567                    return Err(invalid("pair frequency entries are not descending"));
9568                }
9569                pair_frequencies.push(PairFrequencySummary { first, second, entries, omitted_max });
9570            }
9571        } else if &tag == FREQUENCY_TEXTS {
9572            if seen_frequency_texts {
9573                return Err(invalid("directory names two frequency text blocks"));
9574            }
9575            seen_frequency_texts = true;
9576            let columns = cur.u16()? as usize;
9577            if columns > width {
9578                return Err(invalid("frequency text column count exceeds the schema"));
9579            }
9580            for _ in 0..columns {
9581                let column = cur.u16()? as usize;
9582                if !frequency_texts.get(column).is_some_and(Vec::is_empty) {
9583                    return Err(invalid("frequency text column is repeated or out of range"));
9584                }
9585                if !matches!(fields.get(column), Some(field) if coded_type(&field.ty))
9586                    || dictionaries.get(column).copied().flatten().is_none()
9587                    || frequencies.get(column).and_then(Option::as_ref).is_none()
9588                {
9589                    return Err(invalid("frequency texts belong to a non-string synopsis"));
9590                }
9591                let count = cur.u16()? as usize;
9592                if count == 0 || count != entry_counts[column] {
9593                    return Err(invalid("frequency text count differs from its synopsis"));
9594                }
9595                let mut texts = Vec::with_capacity(count);
9596                for _ in 0..count {
9597                    texts.push(match cur.u8()? {
9598                        0 => None,
9599                        1 => {
9600                            let length = cur.u32()? as usize;
9601                            let bytes = cur.take(length)?.to_vec();
9602                            if fields[column].ty == LogicalType::Varchar {
9603                                std::str::from_utf8(&bytes)
9604                                    .map_err(|_| invalid("frequency text is not UTF-8"))?;
9605                            }
9606                            Some(bytes)
9607                        }
9608                        _ => return Err(invalid("frequency text tag differs")),
9609                    });
9610                }
9611                frequency_texts[column] = texts;
9612            }
9613        } else if &tag == HOST_GROUPS {
9614            if host_groups.is_some() {
9615                return Err(invalid("directory names two host group blocks"));
9616            }
9617            let column = cur.u16()? as usize;
9618            if !matches!(fields.get(column), Some(field) if field.ty == LogicalType::Varchar)
9619                || dictionaries.get(column).copied().flatten().is_none()
9620            {
9621                return Err(invalid("host groups belong to a non-string dictionary"));
9622            }
9623            let omitted_max = cur.u64()?;
9624            if omitted_max > rows as u64 {
9625                return Err(invalid("host group bound exceeds the table"));
9626            }
9627            let count = cur.u16()? as usize;
9628            if count > host::CAPACITY {
9629                return Err(invalid("host group count exceeds its bound"));
9630            }
9631            let mut entries = Vec::with_capacity(count);
9632            let mut bytes = 0_usize;
9633            for _ in 0..count {
9634                let host_len = cur.u32()? as usize;
9635                bytes =
9636                    bytes.checked_add(host_len).ok_or_else(|| invalid("host bytes overflow"))?;
9637                if bytes > host::BYTE_BUDGET {
9638                    return Err(invalid("host groups exceed their byte budget"));
9639                }
9640                let host = std::str::from_utf8(cur.take(host_len)?)
9641                    .map_err(|_| invalid("host is not UTF-8"))?
9642                    .to_owned();
9643                let count = cur.u64()?;
9644                if count == 0 || count > rows as u64 {
9645                    return Err(invalid("host group count exceeds the table"));
9646                }
9647                let bytes_sum = i128::from_le_bytes(
9648                    cur.take(16)?
9649                        .try_into()
9650                        .map_err(|_| invalid("host length sum is truncated"))?,
9651                );
9652                if bytes_sum < 0 {
9653                    return Err(invalid("host length sum is negative"));
9654                }
9655                let minimum_len = cur.u32()? as usize;
9656                bytes =
9657                    bytes.checked_add(minimum_len).ok_or_else(|| invalid("host bytes overflow"))?;
9658                if bytes > host::BYTE_BUDGET {
9659                    return Err(invalid("host groups exceed their byte budget"));
9660                }
9661                let minimum = std::str::from_utf8(cur.take(minimum_len)?)
9662                    .map_err(|_| invalid("host minimum is not UTF-8"))?
9663                    .to_owned();
9664                entries.push(host::HostEntry { host, count, bytes_sum, minimum });
9665            }
9666            if entries.windows(2).any(|pair| pair[0].count < pair[1].count)
9667                || entries.iter().any(|entry| entry.host.is_empty() || entry.minimum.is_empty())
9668            {
9669                return Err(invalid("host groups are not in certified order"));
9670            }
9671            host_groups = Some(host::HostSummary { column, omitted_max, entries });
9672        } else if &tag == CLUSTERING {
9673            if clustering.is_some() {
9674                return Err(invalid("directory names two clustering declarations"));
9675            }
9676            let bucket = Width::from_tag(cur.u8()?)
9677                .ok_or_else(|| invalid("clustering width tag differs"))?;
9678            let count = cur.u16()? as usize;
9679            let mut columns = Vec::with_capacity(count.min(fields.len()));
9680            for _ in 0..count {
9681                columns.push(u32::from(cur.u16()?));
9682            }
9683            // Through the constructor and not built by hand, so that a file claiming a column the
9684            // table does not have is caught at open rather than at the first scan that trusted it.
9685            clustering = Some(Clustering::new(columns, bucket, &fields).map_err(|_| {
9686                invalid("stored clustering declaration does not match the table it is on")
9687            })?);
9688        } else if &tag == DEMOTED {
9689            if !demoted.is_empty() {
9690                return Err(invalid("directory names two demoted column blocks"));
9691            }
9692            let count = cur.u16()? as usize;
9693            if count == 0 || count > width {
9694                return Err(invalid("demoted column count is outside the schema"));
9695            }
9696            demoted = vec![false; width];
9697            for _ in 0..count {
9698                let column = cur.u16()? as usize;
9699                if dictionaries.get(column).copied().flatten().is_none() || demoted[column] {
9700                    return Err(invalid("a demoted column is repeated or has no dictionary"));
9701                }
9702                demoted[column] = true;
9703            }
9704        } else if &tag == SECTIONS {
9705            if seen_sections {
9706                return Err(invalid("directory names two section tables"));
9707            }
9708            seen_sections = true;
9709            generation = cur.u64()?;
9710            let count = cur.u16()? as usize;
9711            if count > MAX_SECTIONS {
9712                return Err(invalid("section count exceeds its bound"));
9713            }
9714            sections = Vec::with_capacity(count);
9715            // entry at a time: a malformed section entry is refused rather than turned into an
9716            // offset.
9717            for _ in 0..count {
9718                sections.push(Section::decode(cur.take(section::ENTRY_BYTES)?)?);
9719            }
9720            for held in &sections {
9721                let Some(end) = held.extent_page.checked_add(u64::from(held.extent_bytes)) else {
9722                    return Err(invalid("a section's extent table overflows the file"));
9723                };
9724                // The bound check is here and not in `section`, because only the caller knows how
9725                // big the file is. A section pointing past the end is a torn directory, and reading
9726                // the payload it names would be reading whatever else is at that offset.
9727                if held.extent_bytes != 0 && (held.extent_page < HEADER || end > size) {
9728                    return Err(invalid("a section's extent table is outside the file"));
9729                }
9730                if held.extents == 0 && held.extent_bytes != 0 {
9731                    return Err(invalid("a section with no extents names an extent table"));
9732                }
9733            }
9734        } else if &tag == DICTIONARY_PAYLOADS {
9735            if seen_payloads {
9736                return Err(invalid("directory names two dictionary payload blocks"));
9737            }
9738            seen_payloads = true;
9739            let count = cur.u16()? as usize;
9740            if count != fields.len() {
9741                return Err(invalid("dictionary payload block does not match the table's columns"));
9742            }
9743            dictionary_payloads = Vec::with_capacity(count);
9744            for _ in 0..count {
9745                let bytes = cur.u64()?;
9746                if bytes > size {
9747                    return Err(invalid("a dictionary payload is larger than the file"));
9748                }
9749                dictionary_payloads.push(bytes);
9750            }
9751        } else {
9752            return Err(invalid("directory extension magic differs"));
9753        }
9754    }
9755    if !cur.done() {
9756        return Err(invalid("directory has trailing bytes"));
9757    }
9758    for stripe in &stripes {
9759        for (column, field) in fields.iter().enumerate() {
9760            if coded_type(&field.ty)
9761                && dictionaries[column].is_some()
9762                && stripe.memberships.get(column).is_none()
9763                && !demoted.get(column).copied().unwrap_or(false)
9764            {
9765                return Err(invalid("string page has no code membership index"));
9766            }
9767        }
9768    }
9769    Ok(Table {
9770        name,
9771        fields,
9772        stripes,
9773        rows,
9774        dictionaries,
9775        dictionary_payloads,
9776        demoted,
9777        distincts,
9778        frequencies,
9779        pair_frequencies,
9780        frequency_texts,
9781        host_groups,
9782        clustering,
9783        generation,
9784        sections,
9785    })
9786}
9787
9788/// A zone map's end, in the layout `rudb_common::bounds` defines. See [`Cursor::bound`].
9789fn put_bound(out: &mut Vec<u8>, bound: Option<&Bound>) -> Result<()> {
9790    bounds::put(out, bound)
9791}
9792
9793/// Which cascades are worth trying on a run of dictionary codes.
9794///
9795/// The exhaustive chooser encodes every candidate at every level of a cascade three deep and keeps
9796/// the smallest, which on a part of 1024 codes is around a hundred full encodes to decide something
9797/// three candidates were always going to win. It is the right default for a crate that does not
9798/// know what it is looking at. Here we do know. Codes are counted from zero in the order the values
9799/// were first seen, so a part of them is one value, or a narrow band, or a few long runs, and those
9800/// are constant, frame of reference and run length. Nothing else has ever come first on this data.
9801///
9802/// A dictionary of dictionary codes is the one candidate that can never pay, because the codes are
9803/// already the dictionary, and it is also the most expensive one to try. Below the top level the
9804/// streams are an RLE's run values and run lengths, which are integers in their own right with no
9805/// runs left in them, so only the two flat candidates go down there.
9806///
9807/// This is size given up for time on purpose, and the ablation is this chooser against
9808/// [`chooser::EXHAUSTIVE`] on the same file.
9809#[derive(Debug)]
9810struct Codes;
9811
9812impl chooser::Chooser for Codes {
9813    fn name(&self) -> &'static str {
9814        "codes"
9815    }
9816
9817    fn narrow_strings(
9818        &self,
9819        _values: &[&[u8]],
9820        offered: &[string::Kind],
9821        _depth: u8,
9822    ) -> Vec<string::Kind> {
9823        // Never reached, because nothing here encodes strings through the cascade. The trait asks
9824        // for it and the honest answer to a question we have no opinion on is the whole list.
9825        offered.to_vec()
9826    }
9827
9828    fn narrow_integers(
9829        &self,
9830        _values: &[i64],
9831        offered: &[integer::Kind],
9832        depth: u8,
9833    ) -> Vec<integer::Kind> {
9834        // The contract is a non empty subset, and a chunk that offers none of the three is a chunk
9835        // this has no opinion about rather than one that cannot be written.
9836        narrowed_to(Codes::keep(depth), offered)
9837    }
9838
9839    fn considers_integer(&self, kind: integer::Kind, depth: u8) -> bool {
9840        Codes::keep(depth).contains(&kind)
9841    }
9842}
9843
9844impl Codes {
9845    fn keep(depth: u8) -> &'static [integer::Kind] {
9846        if depth == 0 {
9847            &[integer::Kind::Constant, integer::Kind::Packed, integer::Kind::Rle]
9848        } else {
9849            &[integer::Kind::Constant, integer::Kind::Packed]
9850        }
9851    }
9852}
9853
9854/// The kinds of `offered` that are in `keep`, or all of `offered` when none of them are.
9855///
9856/// `Packed` applies to every chunk and both choosers keep it, so the fallback is never taken on a
9857/// chunk the cascade offers. It is there because the contract is a non empty subset and a chooser
9858/// that returned nothing would be a chunk that cannot be written. It is also why saying no to a kind
9859/// in `considers_integer` is safe: a kind that is never offered could only have been kept through
9860/// this fallback, and the fallback is never reached.
9861fn narrowed_to(keep: &[integer::Kind], offered: &[integer::Kind]) -> Vec<integer::Kind> {
9862    let narrowed: Vec<integer::Kind> =
9863        offered.iter().copied().filter(|kind| keep.contains(kind)).collect();
9864    if narrowed.is_empty() { offered.to_vec() } else { narrowed }
9865}
9866
9867/// Which cascades are worth trying on a part of plain integers.
9868///
9869/// Wider than [`Codes`] because the values are not codes and carry whatever shape the column has.
9870/// A timestamp column climbs, so delta is the one that matters and is the reason this exists at
9871/// all: three timestamp columns in ClickBench were coming out at exactly eight bytes a row with
9872/// nothing asked of them. The same three columns are why the stride is here, since a timestamp
9873/// loaded from a source that recorded whole seconds is microseconds with twenty zero bits under
9874/// every value. A column that is one value with a handful of exceptions is sparse. What is still
9875/// left out is the dictionary, for the same reason as in [`Codes`]: it is the most
9876/// expensive candidate to try and this file already puts the columns that want one through a
9877/// dictionary of their own before they ever reach here.
9878#[derive(Debug)]
9879struct Fixed;
9880
9881impl chooser::Chooser for Fixed {
9882    fn name(&self) -> &'static str {
9883        "fixed"
9884    }
9885
9886    fn narrow_strings(
9887        &self,
9888        _values: &[&[u8]],
9889        offered: &[string::Kind],
9890        _depth: u8,
9891    ) -> Vec<string::Kind> {
9892        offered.to_vec()
9893    }
9894
9895    fn narrow_integers(
9896        &self,
9897        _values: &[i64],
9898        offered: &[integer::Kind],
9899        depth: u8,
9900    ) -> Vec<integer::Kind> {
9901        narrowed_to(Fixed::keep(depth), offered)
9902    }
9903
9904    fn considers_integer(&self, kind: integer::Kind, depth: u8) -> bool {
9905        Fixed::keep(depth).contains(&kind)
9906    }
9907}
9908
9909impl Fixed {
9910    fn keep(depth: u8) -> &'static [integer::Kind] {
9911        if depth == 0 {
9912            &[
9913                integer::Kind::Constant,
9914                integer::Kind::Packed,
9915                integer::Kind::Delta,
9916                integer::Kind::Rle,
9917                integer::Kind::Sparse,
9918                integer::Kind::Strided,
9919            ]
9920        } else {
9921            &[integer::Kind::Constant, integer::Kind::Packed, integer::Kind::Delta]
9922        }
9923    }
9924}
9925
9926/// Every value of an integer part as an `i64`, or `None` for a part this cannot widen without
9927/// losing one.
9928///
9929/// `UBIGINT` is the only integer type left out, because half its range does not fit and a page that
9930/// silently wrapped would be worse than a page that stays plain. Booleans and strings are not
9931/// integers and have their own ways of being small.
9932fn widened(data: &Data) -> Option<Vec<i64>> {
9933    match data {
9934        Data::Int8(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
9935        Data::UInt8(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
9936        Data::Int16(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
9937        Data::UInt16(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
9938        Data::Int32(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
9939        Data::UInt32(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
9940        Data::Int64(values) => Some(values.to_vec()),
9941        _ => None,
9942    }
9943}
9944
9945/// An integer type a cascaded page can be read back into, and how to tell whether a value fits.
9946///
9947/// This exists so that the check and the conversion can be two loops instead of one. `TryFrom` puts
9948/// them together, which is the right shape for one value and the wrong one for a page: a fallible
9949/// conversion a value at a time is a branch a value at a time, the branch decides whether the loop
9950/// keeps going, and a loop like that is one no compiler will widen.
9951trait Narrow: Copy {
9952    /// How wide this type is, and what to add to a value to put its range at the bottom of a `u64`.
9953    ///
9954    /// Half the width for a signed type, which is what moves its smallest value to zero, and nothing
9955    /// for an unsigned one, whose smallest value is already there.
9956    const BIASED: (u32, u64);
9957
9958    /// The value narrowed, which the caller has already shown fits.
9959    fn narrow(value: i64) -> Self;
9960}
9961
9962/// The bits of `value` a `T` cannot hold, and zero when the value fits.
9963///
9964/// The question is asked this way round because the answers or together. A page fits when every
9965/// residue in it is zero, so the loop is an or into an accumulator and the decision is one test
9966/// after it, where asking whether each value is between a floor and a ceiling gives an answer that
9967/// does not combine and turns into a running minimum and maximum.
9968///
9969/// Biasing and shifting is what the answer is made of, rather than anything that reads more like the
9970/// question, because those are the operations a machine has four of. A 64 bit integer minimum is
9971/// AVX-512. So is a 64 bit arithmetic shift right, which is how the sign extension this could be
9972/// written as would have to be done. An add and a logical shift right are AVX2 and are on every
9973/// machine this runs on, so this is the form that gets four values a cycle instead of one.
9974///
9975/// Adding the bias moves the type's range to `0..=2^bits`, wrapping, so everything in range shifts
9976/// away to nothing and everything outside it leaves something behind. A negative value under an
9977/// unsigned type is caught by the same shift, because a negative `i64` read as a `u64` is enormous.
9978#[allow(clippy::cast_sign_loss, reason = "a residue is a bit pattern and not a number")]
9979fn residue<T: Narrow>(value: i64) -> u64 {
9980    let (bits, bias) = T::BIASED;
9981    (value as u64).wrapping_add(bias) >> bits
9982}
9983
9984/// Says a primitive integer narrows with `as`, and where the bottom of its range is.
9985///
9986/// `as` is a truncation and is the right operation here only because [`fit`] has already found every
9987/// residue zero, and it is what makes the second loop a narrowing store with no branch in it.
9988macro_rules! narrows {
9989    ($($ty:ty => $bias:expr),* $(,)?) => {$(
9990        impl Narrow for $ty {
9991            const BIASED: (u32, u64) = (<$ty>::BITS, $bias);
9992
9993            #[allow(
9994                clippy::cast_possible_truncation,
9995                clippy::cast_sign_loss,
9996                reason = "the caller has checked the bits this truncates away"
9997            )]
9998            fn narrow(value: i64) -> Self {
9999                value as Self
10000            }
10001        }
10002    )*};
10003}
10004
10005narrows! {
10006    i8 => 1 << 7,
10007    u8 => 0,
10008    i16 => 1 << 15,
10009    u16 => 0,
10010    i32 => 1 << 31,
10011    u32 => 0,
10012}
10013
10014/// Narrows a page's values, refusing the page if any of them does not fit.
10015///
10016/// The check first and the conversion second, rather than a fallible conversion a value at a time.
10017/// Both loops here are ones a compiler widens: [`residue`] is three instructions a lane and a
10018/// narrowing store is one. The version before this was a `TryFrom` and a `collect` into a `Result`,
10019/// which is a compare, a branch and a short circuit a value at a time, and on ClickBench 39 it was
10020/// seven percent of the query. The version after that kept a running minimum and maximum, which is
10021/// the obvious way to ask and needs a 64 bit integer minimum that AVX2 does not have, so it stayed
10022/// a value at a time and was still ten percent of the same query.
10023///
10024/// An empty page has nothing to refuse, which falls out of the accumulator starting at zero rather
10025/// than needing a case of its own.
10026fn fit<T: Narrow>(values: &[i64]) -> Result<Vec<T>> {
10027    let mut spilled = 0u64;
10028    for value in values {
10029        spilled |= residue::<T>(*value);
10030    }
10031    if spilled != 0 {
10032        return Err(invalid("page value is not of its type"));
10033    }
10034    Ok(values.iter().map(|value| T::narrow(*value)).collect())
10035}
10036
10037/// The same values back in the width the column is declared at.
10038///
10039/// A value that does not fit is a page that disagrees with the directory about what the column is,
10040/// which is a damaged file rather than a caller error, so it is refused rather than truncated.
10041fn narrowed(ty: &LogicalType, values: Vec<i64>) -> Result<Data> {
10042    Ok(match ty {
10043        LogicalType::TinyInt => Data::Int8(fit::<i8>(&values)?.into()),
10044        LogicalType::UTinyInt => Data::UInt8(fit::<u8>(&values)?.into()),
10045        LogicalType::SmallInt => Data::Int16(fit::<i16>(&values)?.into()),
10046        LogicalType::USmallInt => Data::UInt16(fit::<u16>(&values)?.into()),
10047        LogicalType::Integer | LogicalType::Date => Data::Int32(fit::<i32>(&values)?.into()),
10048        LogicalType::UInteger => Data::UInt32(fit::<u32>(&values)?.into()),
10049        LogicalType::BigInt
10050        | LogicalType::Timestamp
10051        | LogicalType::Time
10052        | LogicalType::TimeTz
10053        | LogicalType::TimestampTz
10054        | LogicalType::TimestampS
10055        | LogicalType::TimestampMs
10056        | LogicalType::TimestampNs => Data::Int64(values.into()),
10057        // A decimal is an integer of unscaled units, so the cascade reads back into whichever
10058        // integer the declared width says the column is stored as.
10059        LogicalType::Decimal { .. } => match ty.physical() {
10060            PhysicalType::Int16 => Data::Int16(fit::<i16>(&values)?.into()),
10061            PhysicalType::Int32 => Data::Int32(fit::<i32>(&values)?.into()),
10062            PhysicalType::Int64 => Data::Int64(values.into()),
10063            _ => return Err(invalid("cascade codec belongs to a decimal that is not an integer")),
10064        },
10065        _ => return Err(invalid("cascade codec belongs to a page that is not integers")),
10066    })
10067}
10068
10069/// How many bytes a part of this type costs written out plainly, which is what the cascade has to
10070/// beat before it is worth the decode.
10071fn plain_width(ty: &LogicalType) -> Option<usize> {
10072    Some(match ty {
10073        LogicalType::TinyInt | LogicalType::UTinyInt => 1,
10074        LogicalType::SmallInt | LogicalType::USmallInt => 2,
10075        LogicalType::Integer | LogicalType::UInteger | LogicalType::Date => 4,
10076        LogicalType::BigInt
10077        | LogicalType::Timestamp
10078        | LogicalType::Time
10079        | LogicalType::TimeTz
10080        | LogicalType::TimestampTz
10081        | LogicalType::TimestampS
10082        | LogicalType::TimestampMs
10083        | LogicalType::TimestampNs => 8,
10084        LogicalType::Decimal { .. } => match ty.physical() {
10085            PhysicalType::Int16 => 2,
10086            PhysicalType::Int32 => 4,
10087            PhysicalType::Int64 => 8,
10088            // The widest decimals are stored as `i128`, which the cascade does not widen into, so
10089            // they take the plain path and there is nothing here to compare against.
10090            _ => return None,
10091        },
10092        _ => return None,
10093    })
10094}
10095
10096/// A part's plain integers through the cascade, or `None` when nothing it offers is worth it.
10097///
10098/// What it has to beat is whatever the page would otherwise have cost, which is the bit packed form
10099/// where there is one and the plain width where there is not. Both are cheaper to decode than a
10100/// cascade, so a tie goes to them.
10101fn cascaded(
10102    flat: &Vector,
10103    ty: &LogicalType,
10104    packed: Option<&Packed<'_>>,
10105    settling: &mut Settling,
10106) -> Result<Option<Vec<u8>>> {
10107    let (Some(width), Some(data)) = (plain_width(ty), flat.data()) else { return Ok(None) };
10108    let Some(values) = widened(data) else { return Ok(None) };
10109    let plain = values.len().saturating_mul(width);
10110    let best = match packed {
10111        // The tag, the base, the word count and the words, which is what the codec 2 branch writes.
10112        Some(packed) => plain.min(21 + size_of_val(packed.words())),
10113        None => plain,
10114    };
10115    let out = settling.encode(&values)?;
10116    Ok((out.len() < best).then_some(out))
10117}
10118
10119/// How often the parts of one column in one stripe search the cascade again, in parts.
10120///
10121/// A stripe is 64 parts, so this is four searches a stripe where there were 64. The search is
10122/// what the cascade costs: on ClickBench `hits` the integer cascade was about a tenth of the load's
10123/// CPU and nearly all of it under `encode_pages`, trying six trees on every part to keep the one
10124/// the part before had kept.
10125const SEARCH_EVERY: usize = 16;
10126
10127/// What the parts of one column in one stripe have settled on in the integer cascade.
10128///
10129/// One of these per column per stripe, used in part order, so what a part comes out as depends on
10130/// the stripe and not on which thread wrote it or on how many there were.
10131#[derive(Debug, Default)]
10132struct Settling {
10133    /// The shape of the last part that was searched, with what its top level offered, its length
10134    /// and its row count, which is the size a replay is held to.
10135    shape: Option<Shape>,
10136    /// Parts replayed since that search.
10137    since: usize,
10138}
10139
10140impl Settling {
10141    /// A part's integers through the cascade, replaying the settled shape where there is one.
10142    ///
10143    /// The replay is kept when it held and came out no more than a quarter bigger a row than the
10144    /// part the shape was searched on. Past that the column has changed under it and the part is
10145    /// searched. A replay that stopped fitting partway has already searched from where it stopped,
10146    /// so its shape is taken as the new one rather than searched a second time.
10147    fn encode(&mut self, values: &[i64]) -> Result<Vec<u8>> {
10148        if let Some(shape) = self.shape.as_ref().filter(|_| self.since < SEARCH_EVERY) {
10149            let replay = chooser::Replay::new(&shape.kinds, &Fixed).expecting(&shape.offered);
10150            let out = integer::encode_with(values, &replay)?;
10151            if !replay.held() {
10152                self.settle(&out, values.len(), replay.first_offered())?;
10153                return Ok(out);
10154            }
10155            let grown = (out.len() as u128) * (shape.rows as u128) * 4;
10156            if grown <= (shape.len as u128) * (values.len() as u128) * 5 {
10157                self.since += 1;
10158                return Ok(out);
10159            }
10160        }
10161        // A replay of nothing is the search, and says what the top level offered on the way.
10162        let search = chooser::Replay::new(&[], &Fixed);
10163        let out = integer::encode_with(values, &search)?;
10164        self.settle(&out, values.len(), search.first_offered())?;
10165        Ok(out)
10166    }
10167
10168    fn settle(&mut self, out: &[u8], rows: usize, offered: Vec<integer::Kind>) -> Result<()> {
10169        let kinds = integer::shape(out)?;
10170        self.shape = Some(Shape { kinds, offered, len: out.len().max(1), rows: rows.max(1) });
10171        self.since = 0;
10172        Ok(())
10173    }
10174}
10175
10176/// A searched part's cascade, what its top level was offered, and what it came to.
10177#[derive(Debug)]
10178struct Shape {
10179    kinds: Vec<integer::Kind>,
10180    offered: Vec<integer::Kind>,
10181    len: usize,
10182    rows: usize,
10183}
10184
10185/// A part's dictionary codes through the integer cascade, or `None` when the cascade did not pay.
10186///
10187/// Until now this stream was a `u32` a row with nothing asked of it, and on ClickBench that was
10188/// 400,185,326 bytes for every one of the 28 varchar columns, the same count for `URL` as for a
10189/// column holding the empty string in nearly every row. Codes are dense integers counted from zero
10190/// and a part holds 1024 of them, which is the shape frame of reference is best at, and a column
10191/// with one value everywhere comes back a constant costing nothing per row rather than four bytes.
10192///
10193/// The result is taken only when it is smaller than the plain form. A cascade is allowed to come
10194/// out larger on a part whose codes are genuinely wide, `URL` has about sixty million distinct
10195/// values, and there is no reason to pay for the decode when it does.
10196/// A varchar page as one FSST layer, or `None` when it did not pay.
10197///
10198/// Until now a varchar page that neither the global dictionary nor the per page dictionary claimed
10199/// was written out raw: four bytes of offset a row and then the bytes. That is the right answer for
10200/// a page of values with nothing in common and the wrong one for a page of English, and a column of
10201/// comments is the case this exists for.
10202///
10203/// One layer and not the full string cascade, which is what the payload blocks of a global
10204/// dictionary go through. The cascade is a search: it encodes the page under every candidate it has
10205/// and recurses into the integer cascade for the lengths of each one, and on TPC-H `orders` that
10206/// took the write from 6.9 s to 48.3 s. It reads back no faster than the dictionary it replaced
10207/// either, 1.807 G instructions against 1.810 G for `select o_comment from orders`, because
10208/// unpicking a nest of layers a value at a time costs what the dictionary's payload block decode
10209/// cost. Raw pages of the same column read in 0.686 G, which says the whole of the difference is
10210/// what the page has to be put back together from.
10211///
10212/// FSST alone keeps most of what the cascade found and gives all of that back. Decoding it is one
10213/// pass over the payload into one buffer, the values are laid end to end in it the way the raw form
10214/// already lays them out, and what the reader hands a chunk is views over that buffer.
10215///
10216/// The page dictionary gets first refusal because it is cheaper still, and it wins on a page whose
10217/// values repeat. What is left for this is the page whose values mostly do not, which is exactly the
10218/// page that was being written raw.
10219///
10220/// Taken only when it comes out smaller than the raw form, so a page of incompressible values pays
10221/// nothing at read time for having been offered.
10222fn text_compressed(flat: &Vector) -> Result<Option<Vec<u8>>> {
10223    let mut values: Vec<&[u8]> = Vec::with_capacity(flat.len());
10224    let mut payload = 0_usize;
10225    for row in 0..flat.len() {
10226        // bytes_at: the rows were checked for UTF-8 on the way in, and checking them again here
10227        // was most of what the loop cost.
10228        let text = flat.bytes_at(row).unwrap_or(b"");
10229        payload = payload.saturating_add(text.len());
10230        values.push(text);
10231    }
10232    // What codec 0 writes for a varchar page: an offset a row and one more, then the payload.
10233    let plain = (flat.len() + 1).saturating_mul(4).saturating_add(payload);
10234    let Some(out) = string::encode_only(string::Kind::Fsst, &values)? else {
10235        return Ok(None);
10236    };
10237    Ok((out.len() < plain).then_some(out))
10238}
10239
10240fn encoded_codes(codes: &[u32]) -> Result<Option<Vec<u8>>> {
10241    let wide: Vec<i64> = codes.iter().map(|code| i64::from(*code)).collect();
10242    let coded = integer::encode_with(&wide, &Codes)?;
10243    let plain = codes.len().saturating_mul(size_of::<u32>());
10244    Ok((coded.len() < plain).then_some(coded))
10245}
10246
10247/// The validity of a page, which is a flag and then, when some rows are null and some are not, a
10248/// bit a row with the valid ones set.
10249fn push_validity(out: &mut Vec<u8>, flat: &Vector) {
10250    let flag = match flat.validity() {
10251        Validity::AllValid => 0,
10252        Validity::AllInvalid => 1,
10253        Validity::Mask(_) => 2,
10254    };
10255    out.push(flag);
10256    if flag == 2 {
10257        for group in (0..flat.len()).step_by(8) {
10258            let mut bits = 0_u8;
10259            for bit in 0..8 {
10260                if group + bit < flat.len() && !flat.is_null_at(group + bit) {
10261                    bits |= 1 << bit;
10262                }
10263            }
10264            out.push(bits);
10265        }
10266    }
10267}
10268
10269/// One part of a column coded against its global dictionary as a page, from the codes and the
10270/// validity [`push_validity`] wrote for it.
10271///
10272/// The codes go through the integer cascade when that comes out smaller than four bytes a code,
10273/// which on a column that repeats itself it nearly always does, and are written as they are when it
10274/// does not.
10275fn coded_page(codes: &[u32], validity: &[u8]) -> Result<Vec<u8>> {
10276    let coded = encoded_codes(codes)?;
10277    let mut out = Vec::with_capacity(
10278        1 + validity.len() + coded.as_ref().map_or(size_of_val(codes), Vec::len),
10279    );
10280    out.push(if coded.is_some() { 4 } else { 3 });
10281    out.extend_from_slice(validity);
10282    match coded {
10283        Some(coded) => out.extend_from_slice(&coded),
10284        None => {
10285            for &code in codes {
10286                put_u32(&mut out, code);
10287            }
10288        }
10289    }
10290    Ok(out)
10291}
10292
10293/// One part of one column as a page, for every column that is not coded against a global
10294/// dictionary. Those are built by [`coded_page`] from codes [`prepare`] handed out.
10295fn encode(vector: &Vector, settling: &mut Settling) -> Result<Vec<u8>> {
10296    let ty = vector.logical_type();
10297    // flatten: the file writer needs a uniform scalar page and does it once per loaded chunk.
10298    let flat = vector.flatten()?;
10299    let mut out = Vec::new();
10300    let dictionary = if coded_type(ty) { string_dictionary(&flat)? } else { None };
10301    let compressed_text =
10302        if dictionary.is_none() && coded_type(ty) { text_compressed(&flat)? } else { None };
10303    let packed_vector = if dictionary.is_none() { Some(flat.bit_packed()?) } else { None };
10304    let packed = packed_vector.as_ref().and_then(Vector::packed_parts);
10305    // Only where nothing else has claimed the page, which is the plain integer case. A packed part
10306    // is still on the table because the cascade has to beat it too: the bit pack takes a part only
10307    // when it halves it, so a column that shrinks by a third was coming out whole.
10308    let cascade =
10309        if dictionary.is_none() { cascaded(&flat, ty, packed.as_ref(), settling)? } else { None };
10310    out.push(if cascade.is_some() {
10311        5
10312    } else if dictionary.is_some() {
10313        1
10314    } else if compressed_text.is_some() {
10315        6
10316    } else if packed.is_some() {
10317        2
10318    } else {
10319        0
10320    });
10321    push_validity(&mut out, &flat);
10322    if let Some(cascade) = cascade {
10323        out.extend_from_slice(&cascade);
10324        return Ok(out);
10325    }
10326    if let Some(dictionary) = dictionary {
10327        out.extend_from_slice(&dictionary);
10328        return Ok(out);
10329    }
10330    if let Some(compressed_text) = compressed_text {
10331        out.extend_from_slice(&compressed_text);
10332        return Ok(out);
10333    }
10334    if let Some(packed) = packed {
10335        if packed.offset() != 0 {
10336            return Err(invalid("writer received a sliced packed vector"));
10337        }
10338        out.push(u8::try_from(packed.width()).map_err(|_| invalid("packed width overflow"))?);
10339        out.extend_from_slice(&packed.base().to_le_bytes());
10340        put_u32(
10341            &mut out,
10342            u32::try_from(packed.words().len()).map_err(|_| invalid("too many packed words"))?,
10343        );
10344        for word in packed.words() {
10345            put_u64(&mut out, *word);
10346        }
10347        return Ok(out);
10348    }
10349    let data = flat.data().ok_or_else(|| invalid("scalar column did not flatten"))?;
10350    match (ty, data) {
10351        (LogicalType::TinyInt, Data::Int8(values)) => {
10352            for value in &**values {
10353                out.extend_from_slice(&value.to_le_bytes());
10354            }
10355        }
10356        (LogicalType::UTinyInt, Data::UInt8(values)) => {
10357            for value in &**values {
10358                out.extend_from_slice(&value.to_le_bytes());
10359            }
10360        }
10361        (LogicalType::SmallInt, Data::Int16(values)) => {
10362            for value in &**values {
10363                out.extend_from_slice(&value.to_le_bytes());
10364            }
10365        }
10366        (LogicalType::USmallInt, Data::UInt16(values)) => {
10367            for value in &**values {
10368                out.extend_from_slice(&value.to_le_bytes());
10369            }
10370        }
10371        (LogicalType::UInteger, Data::UInt32(values)) => {
10372            for value in &**values {
10373                out.extend_from_slice(&value.to_le_bytes());
10374            }
10375        }
10376        (LogicalType::UBigInt, Data::UInt64(values)) => {
10377            for value in &**values {
10378                out.extend_from_slice(&value.to_le_bytes());
10379            }
10380        }
10381        (LogicalType::Integer | LogicalType::Date, Data::Int32(values)) => {
10382            for value in &**values {
10383                out.extend_from_slice(&value.to_le_bytes());
10384            }
10385        }
10386        (
10387            LogicalType::BigInt
10388            | LogicalType::Timestamp
10389            | LogicalType::Time
10390            | LogicalType::TimeTz
10391            | LogicalType::TimestampTz
10392            | LogicalType::TimestampS
10393            | LogicalType::TimestampMs
10394            | LogicalType::TimestampNs,
10395            Data::Int64(values),
10396        ) => {
10397            for value in &**values {
10398                out.extend_from_slice(&value.to_le_bytes());
10399            }
10400        }
10401        // A hugeint and a uuid are both the 128 bit lane, and a uuid's bits are the ones the rest of
10402        // the engine already carries it in, so nothing about the value changes on the way down.
10403        (LogicalType::HugeInt | LogicalType::Uuid, Data::Int128(values)) => {
10404            for value in &**values {
10405                out.extend_from_slice(&value.to_le_bytes());
10406            }
10407        }
10408        (LogicalType::UHugeInt, Data::UInt128(values)) => {
10409            for value in &**values {
10410                out.extend_from_slice(&value.to_le_bytes());
10411            }
10412        }
10413        // Plainly, in the IEEE bytes. The integer encodings do not apply to a float and none of the
10414        // float codecs is worth having before somebody has measured a corpus of them.
10415        (LogicalType::Float, Data::Float32(values)) => {
10416            for value in &**values {
10417                out.extend_from_slice(&value.to_le_bytes());
10418            }
10419        }
10420        (LogicalType::Double, Data::Float64(values)) => {
10421            for value in &**values {
10422                out.extend_from_slice(&value.to_le_bytes());
10423            }
10424        }
10425        // Three counts and not one number. Months, days and microseconds stay apart on disk because
10426        // they are apart in the value: a month is not a fixed number of days and a day is not a
10427        // fixed number of microseconds, which is the whole reason the type has three fields.
10428        (LogicalType::Interval, Data::Interval(values)) => {
10429            for (months, days, micros) in &**values {
10430                out.extend_from_slice(&months.to_le_bytes());
10431                out.extend_from_slice(&days.to_le_bytes());
10432                out.extend_from_slice(&micros.to_le_bytes());
10433            }
10434        }
10435        (LogicalType::Boolean, Data::Bool(values)) => {
10436            for value in &**values {
10437                out.push(u8::from(*value));
10438            }
10439        }
10440        // The unscaled integer and nothing else. Scale is a property of the column and it is in the
10441        // directory already, so writing it a value at a time would be paying for it twice.
10442        (LogicalType::Decimal { .. }, Data::Int16(values)) => {
10443            for value in &**values {
10444                out.extend_from_slice(&value.to_le_bytes());
10445            }
10446        }
10447        (LogicalType::Decimal { .. }, Data::Int32(values)) => {
10448            for value in &**values {
10449                out.extend_from_slice(&value.to_le_bytes());
10450            }
10451        }
10452        (LogicalType::Decimal { .. }, Data::Int64(values)) => {
10453            for value in &**values {
10454                out.extend_from_slice(&value.to_le_bytes());
10455            }
10456        }
10457        (LogicalType::Decimal { .. }, Data::Int128(values)) => {
10458            for value in &**values {
10459                out.extend_from_slice(&value.to_le_bytes());
10460            }
10461        }
10462        // A blob and a bit string go down the way a varchar does, because the layout is the same
10463        // one: an offset a value and then the bytes. What is not the same is that nothing here may
10464        // read the payload as text, which is why this arm asks the column for bytes rather than for
10465        // a string, and why the codecs above that do read text are all asked of a varchar by name.
10466        (LogicalType::Varchar | LogicalType::Blob | LogicalType::Bit, Data::Varlen(values)) => {
10467            let mut bytes = Vec::new();
10468            put_u32(&mut out, 0);
10469            for row in 0..vector.len() {
10470                let value = values.bytes(row).ok_or_else(|| invalid("string view is invalid"))?;
10471                bytes.extend_from_slice(value);
10472                put_u32(
10473                    &mut out,
10474                    u32::try_from(bytes.len())
10475                        .map_err(|_| invalid("string payload exceeds 4GiB"))?,
10476                );
10477            }
10478            out.extend_from_slice(&bytes);
10479        }
10480        _ => return Err(Error::not_implemented(format!("native page for {ty}"))),
10481    }
10482    Ok(out)
10483}
10484
10485fn put_varint(out: &mut Vec<u8>, mut value: u32) {
10486    while value >= 0x80 {
10487        out.push((value as u8 & 0x7f) | 0x80);
10488        value >>= 7;
10489    }
10490    out.push(value as u8);
10491}
10492
10493/// The distinct codes of one part, which is what a stripe's membership index is merged from.
10494fn unique_codes(codes: &[u32]) -> Vec<u32> {
10495    let mut unique = codes.to_vec();
10496    unique.sort_unstable();
10497    unique.dedup();
10498    unique
10499}
10500
10501/// The union of the sorted distinct codes of every part in a stripe.
10502///
10503/// Pairwise up a tree rather than one long list concatenated and sorted. Both are the same order of
10504/// work on paper and the tree is the one that does not sort what is already in order: sixty four
10505/// sorted lists become one in six passes over the values.
10506fn merged_codes(lists: Vec<Vec<u32>>) -> Vec<u32> {
10507    let mut lists = lists;
10508    while lists.len() > 1 {
10509        let mut next = Vec::with_capacity(lists.len().div_ceil(2));
10510        for pair in lists.chunks(2) {
10511            match pair {
10512                [left, right] => next.push(merged_pair(left, right)),
10513                [only] => next.push(only.clone()),
10514                _ => {}
10515            }
10516        }
10517        lists = next;
10518    }
10519    lists.pop().unwrap_or_default()
10520}
10521
10522fn merged_pair(left: &[u32], right: &[u32]) -> Vec<u32> {
10523    let mut out = Vec::with_capacity(left.len().saturating_add(right.len()));
10524    let mut at = 0;
10525    let mut to = 0;
10526    while at < left.len() && to < right.len() {
10527        match left[at].cmp(&right[to]) {
10528            Ordering::Less => {
10529                out.push(left[at]);
10530                at += 1;
10531            }
10532            Ordering::Greater => {
10533                out.push(right[to]);
10534                to += 1;
10535            }
10536            Ordering::Equal => {
10537                out.push(left[at]);
10538                at += 1;
10539                to += 1;
10540            }
10541        }
10542    }
10543    out.extend_from_slice(&left[at..]);
10544    out.extend_from_slice(&right[to..]);
10545    out
10546}
10547
10548/// The widest bounds and the total null count of a stripe, from the bounds of its parts.
10549///
10550/// A bound that is missing from any part is missing from the stripe, because a missing bound means
10551/// nothing is known and a stripe that holds an unknown cannot claim one.
10552fn merged_range(ranges: impl Iterator<Item = Range>) -> Range {
10553    let mut merged = Range::default();
10554    let mut first = true;
10555    for range in ranges {
10556        merged.nulls = merged.nulls.saturating_add(range.nulls);
10557        // Both of these have to survive every part, so one part that could not say anything makes
10558        // the stripe unable to say it either. A sum is dropped on overflow rather than wrapped,
10559        // which leaves the stripe with exact ends and no total, which is a true thing to say.
10560        merged.sum = match (merged.sum.take(), range.sum) {
10561            (Some(held), Some(next)) if !first => held.checked_add(next),
10562            (_, next) if first => next,
10563            _ => None,
10564        };
10565        merged.exact = if first { range.exact } else { merged.exact && range.exact };
10566        if first {
10567            merged.low = range.low;
10568            merged.high = range.high;
10569            first = false;
10570            continue;
10571        }
10572        merged.low = match (merged.low.take(), range.low) {
10573            (Some(held), Some(next)) => Some(held.smaller(next)),
10574            _ => None,
10575        };
10576        merged.high = match (merged.high.take(), range.high) {
10577            (Some(held), Some(next)) => Some(held.larger(next)),
10578            _ => None,
10579        };
10580    }
10581    merged
10582}
10583
10584/// One stripe's sieves for one column: the part count, a length for each part, then their bytes.
10585///
10586/// One page for the whole stripe rather than one per part, because a part's sieve is a few hundred
10587/// bytes and sixty four of those are sixty four directory entries and sixty four reads for something
10588/// a scan walks straight through. A part with no sieve writes a length of zero and costs four bytes.
10589/// `bound` cut down to [`PART_BOUND_BYTES`], still a bound of the side it was.
10590///
10591/// A prefix of a string sorts at or before the string, so cutting one down leaves a low end that is
10592/// still a low end. A high end has to go the other way, so the cut prefix is stepped up at the last
10593/// byte that can carry it, and a prefix of nothing but `0xFF` has no such byte and gives up the
10594/// bound rather than claiming one that is too small. Anything that is not a string is already a
10595/// fixed width and is left alone.
10596fn shortened(bound: Option<Bound>, high: bool) -> Option<Bound> {
10597    match bound {
10598        Some(Bound::Bytes(mut value)) if value.len() > PART_BOUND_BYTES => {
10599            value.truncate(PART_BOUND_BYTES);
10600            if !high {
10601                return Some(Bound::Bytes(value));
10602            }
10603            while let Some(last) = value.pop() {
10604                if last < u8::MAX {
10605                    value.push(last + 1);
10606                    return Some(Bound::Bytes(value));
10607                }
10608            }
10609            None
10610        }
10611        other => other,
10612    }
10613}
10614
10615/// The ranges of one column's parts of one stripe, as a page.
10616///
10617/// The two ends and the null count, and not `exact` or the total. Those two answer a `MIN` or a
10618/// `SUM` out of the directory, and the directory already answers those per stripe, where the same
10619/// number costs sixty times less to keep. What a part range is for is skipping the part, and
10620/// skipping needs the ends. So a range read back from here says it is not exact, which is true of a
10621/// string end that was cut down anyway.
10622fn encode_part_ranges(ranges: &[Range]) -> Result<Vec<u8>> {
10623    let mut out = Vec::new();
10624    put_u32(
10625        &mut out,
10626        u32::try_from(ranges.len()).map_err(|_| invalid("too many parts in a stripe"))?,
10627    );
10628    for range in ranges {
10629        put_bound(&mut out, shortened(range.low.clone(), false).as_ref())?;
10630        put_bound(&mut out, shortened(range.high.clone(), true).as_ref())?;
10631        put_u32(&mut out, u32::try_from(range.nulls).map_err(|_| invalid("null count overflow"))?);
10632    }
10633    Ok(out)
10634}
10635
10636/// The ranges one encoded page holds, one entry per part of the stripe.
10637fn decode_part_ranges(bytes: &[u8]) -> Result<Vec<Range>> {
10638    let mut cur = Cursor::new(bytes);
10639    let parts = cur.u32()? as usize;
10640    let mut out = Vec::new();
10641    for _ in 0..parts {
10642        let low = cur.bound()?;
10643        let high = cur.bound()?;
10644        let nulls = cur.u32()? as usize;
10645        out.push(Range { low, high, nulls, exact: false, sum: None });
10646    }
10647    Ok(out)
10648}
10649
10650fn encode_sieves<'a>(sieves: impl Iterator<Item = &'a Option<Sieve>>) -> Result<Vec<u8>> {
10651    let held: Vec<&Option<Sieve>> = sieves.collect();
10652    let mut out = Vec::new();
10653    put_u32(
10654        &mut out,
10655        u32::try_from(held.len()).map_err(|_| invalid("too many parts in a stripe"))?,
10656    );
10657    for sieve in &held {
10658        let length = sieve.as_ref().map_or(0, Sieve::len);
10659        put_u32(&mut out, u32::try_from(length).map_err(|_| invalid("sieve length overflow"))?);
10660    }
10661    // flatten: a part with no sieve wrote a length of zero above and contributes no bytes here.
10662    for sieve in held.into_iter().flatten() {
10663        out.extend_from_slice(&sieve.to_bytes());
10664    }
10665    Ok(out)
10666}
10667
10668/// The sieves one encoded page holds, one entry per part of the stripe.
10669///
10670/// A part whose bytes are not a sieve this version understands comes back as `None`, which is a part
10671/// that gets read. That is how a file written by a later version of the sieve stays readable rather
10672/// than being a corrupt page.
10673fn decode_sieves(bytes: &[u8]) -> Result<Vec<Option<Sieve>>> {
10674    let parts = u32::from_le_bytes(
10675        bytes
10676            .get(..4)
10677            .ok_or_else(|| invalid("sieve page is truncated"))?
10678            .try_into()
10679            .map_err(|_| invalid("sieve page is truncated"))?,
10680    ) as usize;
10681    let mut lengths = Vec::with_capacity(parts);
10682    for part in 0..parts {
10683        let at = 4 + part * 4;
10684        let field = bytes.get(at..at + 4).ok_or_else(|| invalid("sieve page is truncated"))?;
10685        lengths.push(u32::from_le_bytes(
10686            field.try_into().map_err(|_| invalid("sieve page is truncated"))?,
10687        ) as usize);
10688    }
10689    let mut at = 4 + parts * 4;
10690    let mut out = Vec::with_capacity(parts);
10691    for length in lengths {
10692        if length == 0 {
10693            out.push(None);
10694            continue;
10695        }
10696        let end = at.checked_add(length).ok_or_else(|| invalid("sieve page is truncated"))?;
10697        let field = bytes.get(at..end).ok_or_else(|| invalid("sieve page is truncated"))?;
10698        out.push(Sieve::from_bytes(field));
10699        at = end;
10700    }
10701    if at != bytes.len() {
10702        return Err(invalid("sieve page has trailing bytes"));
10703    }
10704    Ok(out)
10705}
10706
10707/// One stripe's membership index: the code count and then the codes as ascending deltas.
10708///
10709/// The codes have to be sorted and distinct already, which is what [`unique_codes`] and
10710/// [`merged_codes`] hand over. Anything else decodes as different codes, so neither of those two is
10711/// a step a caller can skip.
10712fn encode_membership(unique: &[u32]) -> Vec<u8> {
10713    let mut out = Vec::with_capacity(unique.len().saturating_mul(2).saturating_add(5));
10714    put_varint(&mut out, u32::try_from(unique.len()).unwrap_or(u32::MAX));
10715    let mut previous = 0;
10716    for (at, &code) in unique.iter().enumerate() {
10717        put_varint(&mut out, if at == 0 { code } else { code - previous });
10718        previous = code;
10719    }
10720    out
10721}
10722
10723fn take_varint(bytes: &[u8], at: &mut usize) -> Result<u32> {
10724    let mut value = 0_u32;
10725    for shift in (0..35).step_by(7) {
10726        let byte = *bytes.get(*at).ok_or_else(|| invalid("membership varint is truncated"))?;
10727        *at += 1;
10728        let part = u32::from(byte & 0x7f);
10729        if shift == 28 && part > 0x0f {
10730            return Err(invalid("membership varint overflow"));
10731        }
10732        value = value
10733            .checked_add(
10734                part.checked_shl(shift).ok_or_else(|| invalid("membership varint overflow"))?,
10735            )
10736            .ok_or_else(|| invalid("membership varint overflow"))?;
10737        if byte & 0x80 == 0 {
10738            return Ok(value);
10739        }
10740    }
10741    Err(invalid("membership varint is too long"))
10742}
10743
10744fn decode_membership(bytes: &[u8]) -> Result<Vec<u32>> {
10745    let mut at = 0;
10746    let count = take_varint(bytes, &mut at)? as usize;
10747    let mut codes = Vec::with_capacity(count);
10748    let mut previous = 0_u32;
10749    for index in 0..count {
10750        let delta = take_varint(bytes, &mut at)?;
10751        let code = if index == 0 {
10752            delta
10753        } else {
10754            previous.checked_add(delta).ok_or_else(|| invalid("membership code overflow"))?
10755        };
10756        if index > 0 && code <= previous {
10757            return Err(invalid("membership codes are not increasing"));
10758        }
10759        codes.push(code);
10760        previous = code;
10761    }
10762    if at != bytes.len() {
10763        return Err(invalid("membership page has trailing bytes"));
10764    }
10765    Ok(codes)
10766}
10767
10768fn string_dictionary(vector: &Vector) -> Result<Option<Vec<u8>>> {
10769    let mut by_text = HashMap::new();
10770    let mut values = Vec::new();
10771    let mut codes = Vec::with_capacity(vector.len());
10772    let mut plain_bytes = 0_usize;
10773    for row in 0..vector.len() {
10774        let text = vector.bytes_at(row).unwrap_or(b"");
10775        plain_bytes = plain_bytes.saturating_add(text.len());
10776        let code = match by_text.get(text) {
10777            Some(&code) => code,
10778            None => {
10779                let code = u32::try_from(values.len())
10780                    .map_err(|_| invalid("too many dictionary values"))?;
10781                by_text.insert(text, code);
10782                values.push(text);
10783                code
10784            }
10785        };
10786        codes.push(code);
10787    }
10788    let dictionary_bytes = values.iter().map(|value| value.len()).sum::<usize>();
10789    let encoded = 8_usize
10790        .saturating_add((values.len() + 1).saturating_mul(4))
10791        .saturating_add(dictionary_bytes)
10792        .saturating_add(codes.len().saturating_mul(4));
10793    let plain = (vector.len() + 1).saturating_mul(4).saturating_add(plain_bytes);
10794    if encoded >= plain {
10795        return Ok(None);
10796    }
10797    let mut out = Vec::with_capacity(encoded);
10798    put_u32(
10799        &mut out,
10800        u32::try_from(values.len()).map_err(|_| invalid("too many dictionary values"))?,
10801    );
10802    put_u32(
10803        &mut out,
10804        u32::try_from(dictionary_bytes).map_err(|_| invalid("dictionary payload exceeds 4GiB"))?,
10805    );
10806    let mut offset = 0_u32;
10807    put_u32(&mut out, offset);
10808    for value in &values {
10809        offset = offset
10810            .checked_add(
10811                u32::try_from(value.len()).map_err(|_| invalid("dictionary value is too long"))?,
10812            )
10813            .ok_or_else(|| invalid("dictionary payload exceeds 4GiB"))?;
10814        put_u32(&mut out, offset);
10815    }
10816    for value in values {
10817        out.extend_from_slice(value);
10818    }
10819    for code in codes {
10820        put_u32(&mut out, code);
10821    }
10822    Ok(Some(out))
10823}
10824
10825/// The room one closing column takes under [`CLOSE_BYTES`], given back when dropped.
10826struct Room<'a, T> {
10827    state: &'a Mutex<(T, usize)>,
10828    finished: &'a Condvar,
10829    bytes: usize,
10830}
10831
10832impl<T> Drop for Room<'_, T> {
10833    fn drop(&mut self) {
10834        let mut held = self.state.lock().unwrap_or_else(PoisonError::into_inner);
10835        held.1 -= self.bytes;
10836        drop(held);
10837        self.finished.notify_all();
10838    }
10839}
10840
10841/// One column's work at the end of a load, as [`Writer::close_columns`] schedules it.
10842enum Closing<'a> {
10843    /// A numeric column's frequencies, and whether to count its distinct values exactly.
10844    Numeric {
10845        column: usize,
10846        counted: bool,
10847    },
10848    Dictionary {
10849        index: usize,
10850        dictionary: &'a GlobalDictionary,
10851    },
10852}
10853
10854/// What one [`Closing`] came back with, by column.
10855enum Closed {
10856    Numeric(usize, (Option<FrequencySummary>, Option<u64>)),
10857    Dictionary(usize, ClosedDictionary),
10858}
10859
10860/// What [`Writer::close_dictionary`] builds for one column and [`Writer::close`] writes.
10861struct ClosedDictionary {
10862    /// `None` for a demoted dictionary, which holds only some of the column. See [`DEMOTED`].
10863    distinct: Option<u64>,
10864    frequencies: Option<FrequencySummary>,
10865    texts: Vec<Option<Vec<u8>>>,
10866    hosts: Option<host::HostSummary>,
10867    encoded: EncodedDictionary,
10868    /// The bytes of the column's payload blocks, which are already in the file.
10869    payload: u64,
10870}
10871
10872struct EncodedDictionary {
10873    index: Vec<u8>,
10874    ranks: Vec<u8>,
10875    grams: Vec<u8>,
10876}
10877
10878/// Sorts codes into the byte order of the values they name, eight bytes of depth at a time.
10879///
10880/// # What the shape of the data does to a comparison sort
10881///
10882/// Distinct values against distinct prefixes, on the eight million row `hits`:
10883///
10884/// ```text
10885///   distinct   first 8   first 16   first 32   column
10886///  2,266,417        50      8,892    232,630   URL
10887///  2,346,025        49      8,534    204,060   Referer
10888///  1,357,764    81,362    348,340    861,579   Title
10889/// ```
10890///
10891/// Two and a quarter million URLs have fifty distinct first eight bytes between them, because they
10892/// all begin `http://` and then a host and there are not many hosts. So a sort that leads with
10893/// those eight bytes settles almost nothing on `URL` and `Referer`, whatever the comment on it used
10894/// to say, and almost every pair falls through to a comparison of whole values that agree for most
10895/// of their length. `Title` is free text and separates at eight bytes, which is why the design
10896/// looked right when it was written.
10897///
10898/// # What is done about it
10899///
10900/// Sort on eight bytes of the value at the current depth, held beside the code, and then take each
10901/// run that those eight bytes leave tied and sort it again on the next eight. A value is fetched
10902/// from the payload once per eight bytes of depth rather than once per comparison, and the sort
10903/// itself runs over an array of integers that is in cache rather than over pointers into a payload
10904/// that is hundreds of megabytes.
10905///
10906/// That is the whole trick, and it matters because the payload touch is the expensive part. The
10907/// bytes themselves are nearly free once the line is in cache, so reading eight at a time and
10908/// throwing away the ones that were not needed beats going back for each one.
10909///
10910/// # Why the length has to be carried
10911///
10912/// The eight bytes are padded with zero when the value has fewer than eight left, and a zero byte
10913/// can appear in a value, so equal keys do not mean equal bytes. What is true is that a value which
10914/// ran out inside the window is a prefix of any other value with the same key, and a prefix sorts
10915/// first, so how many of the eight bytes were real is the tie break and nothing further is needed.
10916/// A run is only worth another pass when all eight were real, because otherwise the run is one
10917/// value: a dictionary holds a value once.
10918fn sort_by_value<'a>(codes: &mut [u32], values: impl Fn(u32) -> &'a [u8]) {
10919    let mut work = vec![(0, codes.len(), 0)];
10920    let mut keyed: Vec<(u64, u8, u32)> = Vec::new();
10921    while let Some((from, to, depth)) = work.pop() {
10922        let part = &mut codes[from..to];
10923        keyed.clear();
10924        keyed.extend(part.iter().map(|&code| {
10925            let value = values(code);
10926            let rest = value.get(depth..).unwrap_or_default();
10927            (head(rest), rest.len().min(8) as u8, code)
10928        }));
10929        keyed.sort_unstable();
10930        for (slot, entry) in part.iter_mut().zip(keyed.iter()) {
10931            *slot = entry.2;
10932        }
10933        let mut start = 0;
10934        while start < keyed.len() {
10935            let (key, taken, _) = keyed[start];
10936            let mut end = start + 1;
10937            while end < keyed.len() && keyed[end].0 == key && keyed[end].1 == taken {
10938                end += 1;
10939            }
10940            if taken == 8 && end - start > 1 {
10941                work.push((from + start, from + end, depth + 8));
10942            }
10943            start = end;
10944        }
10945    }
10946}
10947
10948/// How few codes are worth sorting on more than one thread.
10949const PARALLEL_SORT_MIN: usize = 1 << 16;
10950
10951/// How many buckets a thread gets in [`sort_by_value_across`], so that a thread that drew a slow
10952/// bucket is not what the others wait for.
10953const BUCKETS_PER_WORKER: usize = 4;
10954
10955/// How many sampled codes stand for each bucket when the splitters are picked.
10956const SAMPLES_PER_BUCKET: usize = 32;
10957
10958/// [`sort_by_value`] over `workers` threads, with the same answer.
10959///
10960/// A sample sort. A sample of the codes is sorted and cut into as many equal runs as there are
10961/// buckets, and the values at the cuts are the splitters. Every code goes to the bucket its value
10962/// falls in by a binary search of the splitters, the buckets are laid end to end in splitter order,
10963/// and each bucket is then sorted on its own by whichever thread takes it. Every value in a bucket
10964/// sorts after every value in the bucket before, so the buckets sorted one by one are the codes
10965/// sorted.
10966///
10967/// The answer is the one [`sort_by_value`] gives down to the order of equal values, not only the
10968/// order of different ones. A global dictionary holds each value once, so there are none, but the
10969/// sort does not rely on it: equal values land in the same bucket in code order, which is the order
10970/// [`sort_by_value`] leaves them in, since the code is the last thing it sorts on.
10971///
10972/// On the 10m ClickBench sample the close sorts five columns of one to three and a half million
10973/// distinct values, one column at a time, and until this each sort ran on one thread while the
10974/// other thirty one waited for it.
10975fn sort_by_value_across<'a>(
10976    codes: &mut [u32],
10977    values: impl Fn(u32) -> &'a [u8] + Sync,
10978    workers: usize,
10979) {
10980    if workers <= 1 || codes.len() < PARALLEL_SORT_MIN {
10981        sort_by_value(codes, values);
10982        return;
10983    }
10984    let buckets = workers * BUCKETS_PER_WORKER;
10985    let wanted = buckets * SAMPLES_PER_BUCKET;
10986    let mut sample = (0..wanted).map(|at| codes[at * codes.len() / wanted]).collect::<Vec<_>>();
10987    sort_by_value(&mut sample, &values);
10988    let splitters =
10989        (1..buckets).map(|cut| values(sample[cut * sample.len() / buckets])).collect::<Vec<_>>();
10990    let values = &values;
10991    let splitters = &splitters;
10992    let per = codes.len().div_ceil(workers);
10993    // Which bucket each code goes to, a run of the codes per thread.
10994    let places = std::thread::scope(|scope| {
10995        codes
10996            .chunks(per)
10997            .map(|run| {
10998                scope.spawn(move || {
10999                    run.iter()
11000                        .map(|&code| {
11001                            let value = values(code);
11002                            splitters.partition_point(|splitter| *splitter <= value) as u32
11003                        })
11004                        .collect::<Vec<_>>()
11005                })
11006            })
11007            .collect::<Vec<_>>()
11008            .into_iter()
11009            .flat_map(|handle| {
11010                handle.join().unwrap_or_else(|panic| std::panic::resume_unwind(panic))
11011            })
11012            .collect::<Vec<_>>()
11013    });
11014    let mut starts = vec![0_usize; buckets + 1];
11015    for &place in &places {
11016        starts[place as usize + 1] += 1;
11017    }
11018    for bucket in 0..buckets {
11019        starts[bucket + 1] += starts[bucket];
11020    }
11021    let mut laid = vec![0_u32; codes.len()];
11022    let mut next = starts.clone();
11023    for (&code, &place) in codes.iter().zip(&places) {
11024        laid[next[place as usize]] = code;
11025        next[place as usize] += 1;
11026    }
11027    drop(places);
11028    let mut runs = Vec::with_capacity(buckets);
11029    let mut rest = laid.as_mut_slice();
11030    for bucket in 0..buckets {
11031        let (run, after) = rest.split_at_mut(starts[bucket + 1] - starts[bucket]);
11032        runs.push(run);
11033        rest = after;
11034    }
11035    // The largest buckets first, since they are taken from the back.
11036    runs.sort_by_key(|run| run.len());
11037    let queue = Mutex::new(runs);
11038    std::thread::scope(|scope| {
11039        for _ in 0..workers {
11040            scope.spawn(|| {
11041                loop {
11042                    let taken = queue.lock().unwrap_or_else(PoisonError::into_inner).pop();
11043                    let Some(run) = taken else { break };
11044                    sort_by_value(run, values);
11045                }
11046            });
11047        }
11048    });
11049    codes.copy_from_slice(&laid);
11050}
11051
11052/// The first eight bytes of a value as an integer that sorts the way the bytes sort.
11053fn head(bytes: &[u8]) -> u64 {
11054    let mut word = [0; 8];
11055    let take = bytes.len().min(8);
11056    word[..take].copy_from_slice(&bytes[..take]);
11057    u64::from_be_bytes(word)
11058}
11059
11060/// One column's dictionary page, which is its index and its sorted order.
11061///
11062/// The payload is not in it. Its blocks are in the file already, written as each was encoded, and
11063/// `places` says where, in block order. With `scattered` set the index records each block's start
11064/// and length, so a reader can find one wherever it went.
11065///
11066/// `scattered` false lays the blocks out the way a file written before format 26 has them, one
11067/// behind the next with only the ends recorded. Nothing in the writer asks for that any more. It is
11068/// kept because [`open_global_dictionary`] still reads those files and a reading path that nothing
11069/// can produce is a reading path nothing tests.
11070fn encode_global_dictionary(
11071    dictionary: &GlobalDictionary,
11072    order: &[(u64, u32)],
11073    places: &[Placed],
11074    scattered: bool,
11075) -> Result<EncodedDictionary> {
11076    let values = dictionary.values();
11077    if order.len() != values {
11078        return Err(invalid("global dictionary order does not cover its values"));
11079    }
11080    let blocks = values.div_ceil(TEXT_PAYLOAD_VALUES);
11081    if places.len() != blocks {
11082        return Err(invalid("global dictionary payload is not the blocks it says it is"));
11083    }
11084    if dictionary.grams.len() != blocks {
11085        return Err(invalid("global dictionary signatures do not cover its blocks"));
11086    }
11087    let (ranks, rank_ends) = encode_ranks(order, code_width(values))?;
11088    let rank_blocks = values.div_ceil(TEXT_RANK_BLOCK);
11089    let offset_bits = offset_width(&dictionary.ends);
11090    let payload_words = if scattered { 3 } else { 2 };
11091    let index_len = DICTIONARY_HEADER
11092        .checked_add(offset_bytes(values, offset_bits))
11093        .and_then(|len| len.checked_add(blocks.checked_mul(payload_words * 8)?))
11094        .and_then(|len| len.checked_add(rank_blocks.checked_mul(16)?))
11095        .and_then(|len| len.checked_add(8))
11096        .ok_or_else(|| invalid("global dictionary index length overflow"))?;
11097    let mut index = Vec::with_capacity(index_len);
11098    put_u32(
11099        &mut index,
11100        u32::try_from(values).map_err(|_| invalid("global dictionary has too many values"))?,
11101    );
11102    put_u32(&mut index, TEXT_PAYLOAD_VALUES as u32);
11103    put_u32(
11104        &mut index,
11105        u32::try_from(blocks).map_err(|_| invalid("global dictionary has too many blocks"))?,
11106    );
11107    let flag = (if scattered { DICTIONARY_SCATTERED } else { 0 })
11108        | DICTIONARY_GRAMS
11109        | DICTIONARY_WIDE_GRAMS;
11110    put_u32(&mut index, offset_bits as u32 | flag);
11111    encode_offsets(&dictionary.ends, offset_bits, &mut index)?;
11112    // Where each block is and how long it is, so a reader can find one. The stored blocks are
11113    // shorter than the decoded ones and by a different amount each, so their lengths are the one
11114    // thing the offsets above no longer say, and where they start is no longer arithmetic on the
11115    // block before once a block is written the moment it is encoded.
11116    let mut end = 0_u64;
11117    for place in places {
11118        if scattered {
11119            put_u64(&mut index, place.start);
11120            put_u64(&mut index, place.length);
11121        } else {
11122            end = end
11123                .checked_add(place.length)
11124                .ok_or_else(|| invalid("global dictionary payload overflow"))?;
11125            put_u64(&mut index, end);
11126        }
11127    }
11128    for place in places {
11129        put_u64(&mut index, place.hash);
11130    }
11131    // The same two lists for the sorted order. A rank block is packed at whatever width its own
11132    // heads need, so where one ends is no longer arithmetic on the block number.
11133    if rank_ends.len() != rank_blocks {
11134        return Err(invalid("global dictionary order is not the blocks it says it is"));
11135    }
11136    for end in &rank_ends {
11137        put_u64(&mut index, *end);
11138    }
11139    let mut at = 0_usize;
11140    for end in &rank_ends {
11141        let end = usize::try_from(*end).map_err(|_| invalid("global dictionary order overflow"))?;
11142        put_u64(&mut index, checksum(&ranks[at..end]));
11143        at = end;
11144    }
11145    let gram_len = blocks
11146        .checked_mul(TEXT_GRAM_BYTES)
11147        .ok_or_else(|| invalid("global dictionary signature count overflow"))?;
11148    let mut grams = Vec::with_capacity(gram_len);
11149    for block in &dictionary.grams {
11150        grams.extend_from_slice(block);
11151    }
11152    put_u64(&mut index, checksum(&grams));
11153    if index.len() != index_len {
11154        return Err(invalid("global dictionary index is not the length it was laid out for"));
11155    }
11156    Ok(EncodedDictionary { index, ranks, grams })
11157}
11158
11159/// How many blocks of the payload the shape is settled on.
11160///
11161/// Eight blocks is 8,192 values, which is the sample `chooser::Sampled` draws and is that size for
11162/// the same reason. They are spread across the dictionary rather than taken off the front, because
11163/// a dictionary is in the order values were first seen and the front of it is the first morsel of
11164/// the load.
11165const PAYLOAD_SAMPLE_BLOCKS: usize = 8;
11166
11167/// The shapes the payload encoder picks between.
11168///
11169/// Narrow on purpose. The exhaustive search encodes every candidate at every level and runs at two
11170/// to six megabytes a second on this data, which over the twelve gigabytes of dictionary `hits`
11171/// carries is about an hour of processor time, so it cannot be what a load does. Each of these
11172/// settles the outer level and the one below it, which is where almost all of that hour goes, and
11173/// leaves the levels under them to the exhaustive search where the chunks are small enough for it
11174/// to cost nothing.
11175///
11176/// Measured on the five ClickBench columns that have a dictionary worth the name, at 1,024 values a
11177/// block, against the exhaustive search over the same blocks:
11178///
11179/// | column | exhaustive | FRONT then LZ | LZ then FSST | LZ then PLAIN |
11180/// |---|---|---|---|---|
11181/// | 2 | 2.923 at 4.3 MB/s | 2.587 at 21.2 | 2.593 at 36.1 | 2.538 at 53.6 |
11182/// | 13 | 3.093 at 3.1 | 3.029 at 36.4 | 2.921 at 35.7 | 2.770 at 82.9 |
11183/// | 14 | 2.330 at 2.1 | 2.283 at 24.3 | 2.213 at 23.5 | 2.113 at 67.6 |
11184/// | 39 | 2.459 at 5.3 | 2.147 at 10.6 | 2.145 at 29.3 | 2.088 at 43.1 |
11185/// | 56 | 4.694 at 6.3 | 4.381 at 51.0 | 4.172 at 50.6 | 3.983 at 86.8 |
11186///
11187/// The best of the three per column is 98 percent of the exhaustive ratio for a tenth of the time.
11188/// `FSST` and `PLAIN` on their own are in the list as a floor rather than to win. `FSST` is the
11189/// right answer for text that does not share prefixes with its neighbours, and `PLAIN` is there so
11190/// that a column nothing compresses is found out in the sample and written at a gigabyte a second
11191/// rather than searched for an answer that does not exist.
11192fn payload_shapes() -> Vec<chooser::Settled> {
11193    let integers = vec![integer::Kind::Packed];
11194    [
11195        vec![string::Kind::Front, string::Kind::Lz],
11196        vec![string::Kind::Lz, string::Kind::Fsst],
11197        vec![string::Kind::Lz, string::Kind::Plain],
11198        vec![string::Kind::Fsst],
11199        vec![string::Kind::Plain],
11200    ]
11201    .into_iter()
11202    .map(|strings| chooser::Settled::new(strings, integers.clone()))
11203    .collect()
11204}
11205
11206/// Syncs the file, and counts the sync and how long it took as a publish wait when a load is being
11207/// profiled.
11208///
11209/// A wait rather than time, because the time is already in the publish span around it. What the
11210/// wait columns add is how much of publish was the device, which on the WSL2 disk of the gaming PC
11211/// is most of it: a sync there costs about two milliseconds (see `rudb_device_card`).
11212fn synced(file: &dyn rudb_io::File, profile: Option<&LoadProfile>) -> Result<()> {
11213    let started = profile.map(|_| std::time::Instant::now());
11214    file.sync()?;
11215    if let (Some(profile), Some(started)) = (profile, started) {
11216        profile.waited(
11217            Stage::Publish,
11218            u64::try_from(started.elapsed().as_nanos()).unwrap_or(u64::MAX),
11219        );
11220    }
11221    Ok(())
11222}
11223
11224/// One sealed dictionary block on its way to being encoded outside the writer's lock.
11225///
11226/// Handed out by the merge that sealed it and encoded with the pages of the same stripe. See
11227/// [`GlobalDictionary::hand_out`].
11228#[derive(Debug)]
11229pub(crate) struct Unencoded {
11230    column: usize,
11231    at: usize,
11232    ends: Vec<u32>,
11233    bytes: Vec<u8>,
11234    shape: chooser::Settled,
11235}
11236
11237impl Unencoded {
11238    /// The encoded block and its signature.
11239    pub(crate) fn encode(&self) -> Result<EncodedBlock> {
11240        let values = block_values(&self.ends, &self.bytes);
11241        Ok((string::encode_with(&values, &self.shape)?, block_grams(&values)))
11242    }
11243
11244    /// The column and the block number the encoded block goes back to.
11245    pub(crate) fn place(&self) -> (usize, usize) {
11246        (self.column, self.at)
11247    }
11248}
11249
11250/// One encoded dictionary block and the signature of the values in it.
11251///
11252/// Boxed because it is carried around in things that are otherwise small.
11253pub(crate) type EncodedBlock = (Vec<u8>, Box<[u8; TEXT_GRAM_BYTES]>);
11254
11255/// The conservative four-byte substring signature of one block's values.
11256fn block_grams(values: &[&[u8]]) -> Box<[u8; TEXT_GRAM_BYTES]> {
11257    let mut grams = Box::new([0_u8; TEXT_GRAM_BYTES]);
11258    for value in values {
11259        for gram in value.windows(4) {
11260            for bit in gram_bits(gram, TEXT_GRAM_BYTES) {
11261                grams[bit / 8] |= 1 << (bit % 8);
11262            }
11263        }
11264    }
11265    grams
11266}
11267
11268/// The values of one block, given where each of them ends relative to the block.
11269fn block_values<'a>(ends: &[u32], bytes: &'a [u8]) -> Vec<&'a [u8]> {
11270    let mut out = Vec::with_capacity(ends.len());
11271    let mut from = 0;
11272    for &to in ends {
11273        out.push(&bytes[from..to as usize]);
11274        from = to as usize;
11275    }
11276    out
11277}
11278
11279/// Encodes every block still raw at the end of a load: the part block each column ends on and,
11280/// for a column too small to have settled a shape, every block it has.
11281///
11282/// Across threads, the way [`encode_ready`] does it. This ran one column at a time on the thread
11283/// closing the table, and a column that never settled a shape encodes each block by trying every
11284/// candidate, so on a million rows of `hits` it was most of the load's CPU on one core.
11285fn finish_dictionaries(dictionaries: &mut [Option<GlobalDictionary>]) -> Result<()> {
11286    for dictionary in dictionaries.iter_mut().flatten() {
11287        if !dictionary.early.is_empty() {
11288            return Err(Error::internal("a dictionary block handed out never came back"));
11289        }
11290        dictionary.seal_rest();
11291        dictionary.settle_rest()?;
11292    }
11293    encode_waiting(dictionaries)?;
11294    // A block handed out and never given back leaves a gap nothing above would notice when it was
11295    // the last one, so the count is checked against the values as well.
11296    if dictionaries
11297        .iter()
11298        .flatten()
11299        .any(|dictionary| dictionary.encoded() != dictionary.values().div_ceil(TEXT_PAYLOAD_VALUES))
11300    {
11301        return Err(Error::internal("a dictionary block handed out never came back"));
11302    }
11303    Ok(())
11304}
11305
11306/// Encodes the waiting blocks of every dictionary across threads, and appends them to their columns
11307/// in order.
11308fn encode_waiting(dictionaries: &mut [Option<GlobalDictionary>]) -> Result<()> {
11309    let jobs = dictionaries
11310        .iter()
11311        .enumerate()
11312        .flat_map(|(column, held)| {
11313            (0..held.as_ref().map_or(0, |held| held.waiting.len())).map(move |at| (column, at))
11314        })
11315        .collect::<Vec<_>>();
11316    if jobs.is_empty() {
11317        return Ok(());
11318    }
11319    let one = |column: usize, at: usize| -> Result<(usize, usize, EncodedBlock)> {
11320        let held = dictionaries[column].as_ref().ok_or_else(|| Error::internal("no dictionary"))?;
11321        Ok((column, at, held.encode_waiting(at)?))
11322    };
11323    let workers = std::thread::available_parallelism()
11324        .map_or(1, usize::from)
11325        .min(MAX_FREQUENCY_WORKERS)
11326        .min(jobs.len());
11327    let made = if workers <= 1 {
11328        jobs.iter().map(|&(column, at)| one(column, at)).collect::<Result<Vec<_>>>()?
11329    } else {
11330        let next = AtomicUsize::new(0);
11331        let jobs = &jobs;
11332        let pieces = std::thread::scope(|scope| {
11333            (0..workers)
11334                .map(|_| {
11335                    scope.spawn(|| {
11336                        let mut mine = Vec::new();
11337                        loop {
11338                            let job = next.fetch_add(1, Atomic::Relaxed);
11339                            let Some(&(column, at)) = jobs.get(job) else { break };
11340                            mine.push(one(column, at)?);
11341                        }
11342                        Ok(mine)
11343                    })
11344                })
11345                .collect::<Vec<_>>()
11346                .into_iter()
11347                .map(|handle| {
11348                    handle
11349                        .join()
11350                        .map_err(|_| Error::internal("a dictionary encode worker panicked"))?
11351                })
11352                .collect::<Result<Vec<_>>>()
11353        })?;
11354        pieces.into_iter().flatten().collect()
11355    };
11356    let mut done: Vec<Vec<(usize, EncodedBlock)>> =
11357        (0..dictionaries.len()).map(|_| Vec::new()).collect();
11358    for (column, at, bytes) in made {
11359        done[column].push((at, bytes));
11360    }
11361    for (column, mut made) in done.into_iter().enumerate() {
11362        if made.is_empty() {
11363            continue;
11364        }
11365        let Some(held) = dictionaries[column].as_mut() else { continue };
11366        made.sort_by_key(|(at, _)| *at);
11367        let waiting = std::mem::take(&mut held.waiting);
11368        for ((at, _), (_, block)) in waiting.into_iter().zip(made) {
11369            if held.encoded() != at {
11370                return Err(Error::internal("a dictionary block was encoded out of order"));
11371            }
11372            held.push_block(block);
11373        }
11374    }
11375    Ok(())
11376}
11377
11378/// Which of [`payload_shapes`] comes out smallest over a sample of the blocks.
11379///
11380/// Every shape is encoded over the same sample and the smallest wins, which is the exhaustive
11381/// search moved up a level: over shapes of a column rather than over candidates of a chunk. The
11382/// sample is spread across the dictionary so that the first and last blocks are both in it, because
11383/// a dictionary written in first seen order has its common values at the front and its long tail at
11384/// the back, and those do not compress alike. Which blocks those are is
11385/// [`GlobalDictionary::seal`]'s to decide, because by the time this is called the rest of them have
11386/// been encoded and the raw bytes are gone.
11387fn settle_shape(sample: &[Vec<&[u8]>]) -> Result<chooser::Settled> {
11388    let mut best: Option<(chooser::Settled, usize)> = None;
11389    for shape in payload_shapes() {
11390        let mut size = 0;
11391        for block in sample {
11392            size += string::encode_with(block, &shape)?.len();
11393        }
11394        if best.as_ref().is_none_or(|(_, smallest)| size < *smallest) {
11395            best = Some((shape, size));
11396        }
11397    }
11398    best.map(|(shape, _)| shape)
11399        .ok_or_else(|| invalid("no shape applies to a global dictionary payload"))
11400}
11401
11402/// The sorted order laid out the way a reader reads it, in blocks of [`TEXT_RANK_BLOCK`] entries.
11403///
11404/// Each block holds its heads first and then its codes, rather than pairing them, because a search
11405/// asks for a head at every probe and for a code about once a search. Keeping the heads together
11406/// means a probe touches eight bytes of a block rather than twelve spread over it, and the last few
11407/// probes of a search, which are the ones that land in the same block, touch the same cache line.
11408fn encode_ranks(order: &[(u64, u32)], code_bits: usize) -> Result<(Vec<u8>, Vec<u64>)> {
11409    let mut out = Vec::with_capacity(order.len() * 4);
11410    let mut ends = Vec::with_capacity(order.len().div_ceil(TEXT_RANK_BLOCK));
11411    let mut heads = Vec::with_capacity(TEXT_RANK_BLOCK);
11412    let mut codes = Vec::with_capacity(TEXT_RANK_BLOCK);
11413    for block in order.chunks(TEXT_RANK_BLOCK) {
11414        // The order is sorted by value and a head is a prefix of a value, so the heads of a block
11415        // rise, the smallest is the first and the largest is the last.
11416        let base = block.first().map_or(0, |&(head, _)| head);
11417        let span = block.last().map_or(0, |&(head, _)| head.wrapping_sub(base));
11418        let width = (u64::BITS - span.leading_zeros()) as usize;
11419        heads.clear();
11420        codes.clear();
11421        for &(head, code) in block {
11422            heads.push(head.wrapping_sub(base));
11423            codes.push(u64::from(code));
11424        }
11425        put_u64(&mut out, base);
11426        out.push(width as u8);
11427        bitpack::pack_tail(&heads, width, &mut out)
11428            .map_err(|_| invalid("global dictionary heads do not pack"))?;
11429        bitpack::pack_tail(&codes, code_bits, &mut out)
11430            .map_err(|_| invalid("global dictionary codes do not pack"))?;
11431        ends.push(out.len() as u64);
11432    }
11433    Ok((out, ends))
11434}
11435
11436/// Opens a column's global dictionary, which reads its index and none of its payload.
11437///
11438/// `keep_budget` is how many decoded payload bytes this dictionary may hold on to, and every
11439/// caller bar the test of the ceiling passes [`TEXT_KEEP_BUDGET`]. It is a parameter rather than
11440/// the constant read where it is used because a test of a ceiling that cannot be moved has to build
11441/// a quarter of a gigabyte of dictionary to reach it.
11442fn open_global_dictionary(
11443    file: Arc<File>,
11444    page: Page,
11445    ty: &LogicalType,
11446    keep_budget: usize,
11447) -> Result<Vector> {
11448    if !coded_type(ty) {
11449        return Err(invalid("global dictionary belongs to a non-string column"));
11450    }
11451    let mut header = [0; DICTIONARY_HEADER];
11452    read_at(&file, page.offset, &mut header)?;
11453    let count = u32::from_le_bytes(header[0..4].try_into().expect("four bytes")) as usize;
11454    let per_block = u32::from_le_bytes(header[4..8].try_into().expect("four bytes")) as usize;
11455    let blocks = u32::from_le_bytes(header[8..12].try_into().expect("four bytes")) as usize;
11456    let width = u32::from_le_bytes(header[12..16].try_into().expect("four bytes"));
11457    let scattered = width & DICTIONARY_SCATTERED != 0;
11458    let has_grams = width & DICTIONARY_GRAMS != 0;
11459    let gram_width =
11460        if width & DICTIONARY_WIDE_GRAMS != 0 { TEXT_GRAM_BYTES } else { NARROW_GRAM_BYTES };
11461    let offset_bits = (width & !DICTIONARY_FLAGS) as usize;
11462    if per_block != TEXT_PAYLOAD_VALUES {
11463        return Err(invalid("global dictionary block width differs"));
11464    }
11465    if blocks != count.div_ceil(TEXT_PAYLOAD_VALUES) {
11466        return Err(invalid("global dictionary block count differs from its value count"));
11467    }
11468    if offset_bits > u32::BITS as usize {
11469        return Err(invalid("global dictionary packs offsets past a payload"));
11470    }
11471    let offset_len = offset_bytes(count, offset_bits);
11472    // The sorted order is kept out of the index on purpose. The index is read and checksummed in
11473    // full the moment the column is first touched, and the order is half again the size of the
11474    // offsets, so putting it there would make every query that reads a string column pay for a
11475    // search that most of them never make.
11476    let ranks = count;
11477    let rank_blocks = ranks.div_ceil(TEXT_RANK_BLOCK);
11478    // Three words a payload block, for where it starts, how long it is and what it hashes to, or
11479    // two of them on a file that has the blocks back to back and needs no start. Two a rank block
11480    // either way, since those are still one run.
11481    let payload_words = if scattered { 3 } else { 2 };
11482    let hash_len = blocks
11483        .checked_mul(payload_words * 8)
11484        .and_then(|len| len.checked_add(rank_blocks.checked_mul(16)?))
11485        .and_then(|len| len.checked_add(usize::from(has_grams) * 8))
11486        .ok_or_else(|| invalid("global dictionary block count overflow"))?;
11487    let gram_len = if has_grams {
11488        blocks
11489            .checked_mul(gram_width)
11490            .ok_or_else(|| invalid("global dictionary signature count overflow"))?
11491    } else {
11492        0
11493    };
11494    let index_len = DICTIONARY_HEADER
11495        .checked_add(offset_len)
11496        .and_then(|len| len.checked_add(hash_len))
11497        .ok_or_else(|| invalid("global dictionary header overflow"))?;
11498    if index_len > page.length as usize {
11499        return Err(invalid("global dictionary offset index exceeds its page"));
11500    }
11501    let mut index = vec![0; index_len];
11502    index[..DICTIONARY_HEADER].copy_from_slice(&header);
11503    read_at(&file, page.offset + DICTIONARY_HEADER as u64, &mut index[DICTIONARY_HEADER..])?;
11504    if checksum(&index) != page.hash {
11505        return Err(invalid("global dictionary index checksum differs"));
11506    }
11507    let word_end = index_len - usize::from(has_grams) * 8;
11508    let gram_hash = has_grams
11509        .then(|| u64::from_le_bytes(index[word_end..index_len].try_into().expect("eight bytes")));
11510    let mut words = index[DICTIONARY_HEADER + offset_len..word_end]
11511        .chunks_exact(8)
11512        .map(|part| u64::from_le_bytes(part.try_into().expect("eight bytes")))
11513        .collect::<Vec<_>>();
11514    let mut rest = words.split_off(blocks * payload_words);
11515    let rank_hashes = rest.split_off(rank_blocks);
11516    let rank_ends = rest;
11517    // A rank block packs its heads at whatever width its own values need, so its length is no longer
11518    // arithmetic on the block number and the reader has to be told where each one ends.
11519    if rank_ends.windows(2).any(|pair| pair[0] >= pair[1]) {
11520        return Err(invalid("global dictionary order blocks do not rise"));
11521    }
11522    let rank_len = usize::try_from(rank_ends.last().copied().unwrap_or_default())
11523        .map_err(|_| invalid("global dictionary rank overflow"))?;
11524    let body_len = index_len
11525        .checked_add(rank_len)
11526        .ok_or_else(|| invalid("global dictionary header overflow"))?;
11527    if body_len > page.length as usize {
11528        return Err(invalid("global dictionary order exceeds its page"));
11529    }
11530    let gram_end = body_len
11531        .checked_add(gram_len)
11532        .ok_or_else(|| invalid("global dictionary signature length overflow"))?;
11533    if gram_end > page.length as usize {
11534        return Err(invalid("global dictionary signatures exceed their page"));
11535    }
11536    let grams = gram_hash.map(|hash| NativeGrams {
11537        start: page.offset + body_len as u64,
11538        length: gram_len,
11539        width: gram_width,
11540        hash,
11541        verdicts: Mutex::new(Vec::new()),
11542    });
11543    // The offsets stay where they were read, behind the header, rather than being copied out. On a
11544    // dictionary of millions of values they are megabytes, and a copy is as many fresh pages to
11545    // fault in again on a query that may want a handful of strings.
11546    let mut offsets = index;
11547    offsets.truncate(DICTIONARY_HEADER + offset_len);
11548    let hashes = words.split_off(blocks * (payload_words - 1));
11549    let (starts, lengths) = if scattered {
11550        let mut starts = Vec::with_capacity(blocks);
11551        let mut lengths = Vec::with_capacity(blocks);
11552        for pair in words.chunks_exact(2) {
11553            starts.push(pair[0]);
11554            lengths.push(pair[1]);
11555        }
11556        (starts, lengths)
11557    } else {
11558        // A file written before the blocks said where they were has them behind one another at the
11559        // end of the page, so the base is where the sorted order stops and each end is the start of
11560        // the one after it. Turning them round here is what lets everything below take one shape.
11561        let base = page.offset + gram_end as u64;
11562        let mut starts = Vec::with_capacity(blocks);
11563        let mut lengths = Vec::with_capacity(blocks);
11564        let mut at = 0_u64;
11565        for &end in &words {
11566            let len = end
11567                .checked_sub(at)
11568                .ok_or_else(|| invalid("global dictionary block ends before it starts"))?;
11569            starts.push(base + at);
11570            lengths.push(len);
11571            at = end;
11572        }
11573        (starts, lengths)
11574    };
11575    // What the offsets bound is the decoded payload, and what the page length counts is the stored
11576    // one, so on a format 26 file the block lengths adding up to the rest of the page is the one
11577    // thing that ties the index to the page. From format 27 the blocks are written during the load
11578    // and the page is only the index and the order, so there the most that can be said is that
11579    // every block is somewhere in the file past its header.
11580    let stored_len = page.length as u64 - gram_end as u64;
11581    if scattered && stored_len == 0 {
11582        let size = file.metadata().map_err(io)?.len();
11583        let inside = starts.iter().zip(&lengths).all(|(&start, &len)| {
11584            start >= HEADER && start.checked_add(len).is_some_and(|end| end <= size)
11585        });
11586        if !inside {
11587            return Err(invalid("global dictionary block lies outside the file"));
11588        }
11589    } else if lengths.iter().try_fold(0_u64, |sum, len| sum.checked_add(*len)) != Some(stored_len) {
11590        return Err(invalid("global dictionary blocks do not bound the payload"));
11591    }
11592    Vector::external_text(
11593        ty.clone(),
11594        Arc::new(NativeText {
11595            file,
11596            values: count,
11597            offsets,
11598            offset_bits,
11599            value_ends: OnceLock::new(),
11600            value_lens: OnceLock::new(),
11601            ends_asked: AtomicUsize::new(0),
11602            ranks,
11603            rank_at: page.offset + index_len as u64,
11604            rank_ends,
11605            rank_hashes,
11606            rank_blocks: (0..rank_blocks).map(|_| OnceLock::new()).collect(),
11607            code_bits: code_width(count),
11608            code_ranks: OnceLock::new(),
11609            starts,
11610            lengths,
11611            hashes,
11612            grams,
11613            blocks: (0..blocks).map(|_| OnceLock::new()).collect(),
11614            char_lens: (0..blocks).map(|_| OnceLock::new()).collect(),
11615            keep_budget,
11616            payload_kept: AtomicUsize::new(0),
11617            swept: (0..blocks).map(|_| AtomicBool::new(false)).collect(),
11618            visit_dropped: AtomicUsize::new(0),
11619            searched: Mutex::new(HashMap::new()),
11620        }),
11621    )
11622}
11623
11624/// What a stored page is, without decoding a value out of it.
11625///
11626/// Two layers, and both of them belong in the answer. The codec byte at the front of every page is
11627/// the format's own choice, and it is what says whether the column came back as codes into a table
11628/// wide dictionary, as a bit packed page, as an encoding cascade or as the bytes themselves. Under
11629/// the cascade codecs there is a second choice the encoder made per chunk, and that is what
11630/// [`integer::describe`] and [`string::describe`] already write out as `DICT(PACKED, PACKED)`.
11631///
11632/// This mirrors the tags [`decode`] reads and has to be kept beside it. A page whose header this
11633/// cannot walk comes back as text rather than as an error, because a caller asking what a file
11634/// looks like is usually asking because something is wrong with it, and a report that stops at the
11635/// first bad page is a report that says nothing about the other nine hundred.
11636fn page_encoding(ty: &LogicalType, rows: usize, bytes: &[u8]) -> String {
11637    /// The page header is the codec, the validity tag and, for a page that stores a mask, the mask.
11638    fn cascade_at(rows: usize, bytes: &[u8]) -> Result<(u8, usize)> {
11639        let mut cur = Cursor::new(bytes);
11640        let codec = cur.u8()?;
11641        if cur.u8()? == 2 {
11642            cur.take(rows.div_ceil(8))?;
11643        }
11644        Ok((codec, cur.at))
11645    }
11646    let Ok((codec, at)) = cascade_at(rows, bytes) else {
11647        return "UNREADABLE".to_string();
11648    };
11649    let tail = &bytes[at..];
11650    let described = |described: Result<String>| described.unwrap_or_else(|_| "UNREADABLE".into());
11651    match codec {
11652        0 => match ty {
11653            LogicalType::Varchar | LogicalType::Blob => "PLAIN".to_string(),
11654            _ => "FIXED".to_string(),
11655        },
11656        1 => "DICT(PLAIN)".to_string(),
11657        2 => "FOR+BITPACK".to_string(),
11658        3 => "TABLE DICT".to_string(),
11659        4 => format!("TABLE DICT({})", described(integer::describe(tail))),
11660        5 => described(integer::describe(tail)),
11661        6 => described(string::describe(tail)),
11662        other => format!("CODEC {other}"),
11663    }
11664}
11665
11666/// Selected stable dictionary codes from one page.
11667///
11668/// Pair-frequency construction needs at most the bounded heavy-hitter rows. Reading those code
11669/// positions directly avoids materializing every code in each part that contains a candidate.
11670fn decode_selected_stable_codes(
11671    rows: usize,
11672    bytes: &[u8],
11673    positions: &[usize],
11674    out: &mut Vec<Option<u32>>,
11675) -> Result<bool> {
11676    if positions.windows(2).any(|pair| pair[0] >= pair[1])
11677        || positions.last().is_some_and(|&position| position >= rows)
11678    {
11679        return Err(invalid("selected code positions are not sorted and in range"));
11680    }
11681    let mut cur = Cursor::new(bytes);
11682    let codec = cur.u8()?;
11683    if codec != 3 && codec != 4 {
11684        return Ok(false);
11685    }
11686    let flag = cur.u8()?;
11687    let mask = match flag {
11688        0 | 1 => None,
11689        2 => {
11690            let at = cur.at;
11691            let len = rows.div_ceil(8);
11692            cur.take(len)?;
11693            Some((at, len))
11694        }
11695        _ => return Err(invalid("page validity tag differs")),
11696    };
11697    let valid = |row: usize| match flag {
11698        0 => true,
11699        1 => false,
11700        2 => mask.is_some_and(|(at, _)| bytes[at + row / 8] >> (row % 8) & 1 == 1),
11701        _ => unreachable!("the validity tag was checked"),
11702    };
11703    if codec == 4 {
11704        let wide = integer::decode_selected(&bytes[cur.at..], positions)?;
11705        for (&row, code) in positions.iter().zip(wide) {
11706            let code = u32::try_from(code).map_err(|_| invalid("code is not a code"))?;
11707            out.push(valid(row).then_some(code));
11708        }
11709        return Ok(true);
11710    }
11711    let codes_at = cur.at;
11712    let codes_len = rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?;
11713    cur.take(codes_len)?;
11714    if cur.at != bytes.len() {
11715        return Err(invalid("global code page has trailing bytes"));
11716    }
11717    let codes = &bytes[codes_at..codes_at + codes_len];
11718    for &row in positions {
11719        let at = row.checked_mul(4).ok_or_else(|| invalid("dictionary code offset overflow"))?;
11720        let code = u32::from_le_bytes(
11721            codes[at..at + 4].try_into().map_err(|_| invalid("dictionary code is truncated"))?,
11722        );
11723        out.push(valid(row).then_some(code));
11724    }
11725    Ok(true)
11726}
11727
11728/// [`decode`] of only the rows at `positions`, which rise.
11729///
11730/// A compressed text page decompresses only those rows, see [`string::decode_flat_at`], and checks
11731/// only those rows are text. Every other page is decoded whole and gathered, since its values are
11732/// fixed width or its strings are shared through a dictionary, and there picking comes after.
11733fn decode_at(
11734    ty: &LogicalType,
11735    rows: usize,
11736    bytes: &[u8],
11737    global: Option<Arc<Vector>>,
11738    positions: &[u32],
11739) -> Result<Vector> {
11740    if positions.last().is_some_and(|&last| last as usize >= rows) {
11741        return Err(invalid("a position is past the end of the part"));
11742    }
11743    if bytes.first() != Some(&6) {
11744        return decode(ty, rows, bytes, global)?.gather(positions);
11745    }
11746    if !coded_type(ty) {
11747        return Err(invalid("compressed text codec belongs to a non-string page"));
11748    }
11749    let mut cur = Cursor::new(bytes);
11750    cur.u8()?;
11751    let validity = match cur.u8()? {
11752        0 => Validity::AllValid,
11753        1 => Validity::AllInvalid,
11754        2 => {
11755            let mask = cur.take(rows.div_ceil(8))?;
11756            Validity::from_iter(positions.len(), |at| {
11757                let row = positions[at] as usize;
11758                mask[row / 8] >> (row % 8) & 1 == 1
11759            })
11760        }
11761        _ => return Err(invalid("page validity tag differs")),
11762    };
11763    let (payload, ends) = string::decode_flat_at(&bytes[cur.at..], positions)?.into_parts();
11764    let mut values = StringColumn::over(Buffer::from_vec(payload).into_page());
11765    let mut start = 0;
11766    for end in ends {
11767        let len = end
11768            .checked_sub(start)
11769            .ok_or_else(|| invalid("compressed text value ends before it starts"))?;
11770        push_value(&mut values, ty, start, len)?;
11771        start = end;
11772    }
11773    Ok(Vector::flat(ty.clone(), Data::Varlen(values))?.with_validity(validity))
11774}
11775
11776/// One value of a string or blob page, found in the page's payload. A varchar is checked for text
11777/// on the way in and a blob is not, since a blob never claimed to hold any.
11778fn push_value(values: &mut StringColumn, ty: &LogicalType, at: usize, len: usize) -> Result<()> {
11779    if ty == &LogicalType::Varchar {
11780        values.push_in_place(at, len)?;
11781    } else {
11782        values.push_bytes_in_place(at, len)?;
11783    }
11784    Ok(())
11785}
11786
11787fn decode(
11788    ty: &LogicalType,
11789    rows: usize,
11790    bytes: &[u8],
11791    global: Option<Arc<Vector>>,
11792) -> Result<Vector> {
11793    let mut cur = Cursor::new(bytes);
11794    let codec = cur.u8()?;
11795    let flag = cur.u8()?;
11796    let validity = match flag {
11797        0 => Validity::AllValid,
11798        1 => Validity::AllInvalid,
11799        2 => {
11800            let mask = cur.take(rows.div_ceil(8))?;
11801            Validity::from_iter(rows, |row| mask[row / 8] >> (row % 8) & 1 == 1)
11802        }
11803        _ => return Err(invalid("page validity tag differs")),
11804    };
11805    if codec == 1 {
11806        if !coded_type(ty) {
11807            return Err(invalid("dictionary codec belongs to a non-string page"));
11808        }
11809        let count = cur.u32()? as usize;
11810        let payload_len = cur.u32()? as usize;
11811        let offset_bytes = cur.take(
11812            (count + 1)
11813                .checked_mul(4)
11814                .ok_or_else(|| invalid("dictionary offset count overflow"))?,
11815        )?;
11816        let offsets = offset_bytes
11817            .chunks_exact(4)
11818            .map(|part| u32::from_le_bytes(part.try_into().expect("four bytes")))
11819            .collect::<Vec<_>>();
11820        let payload = cur.take(payload_len)?.to_vec();
11821        if offsets.first() != Some(&0)
11822            || offsets.last().copied().map(|last| last as usize) != Some(payload.len())
11823            || offsets.windows(2).any(|pair| pair[0] > pair[1])
11824        {
11825            return Err(invalid("dictionary offsets do not bound the payload"));
11826        }
11827        // A page, because every chunk cut out of this dictionary points at the same payload and a
11828        // page is what lets a cut be the views and nothing else.
11829        let mut strings = StringColumn::over(Buffer::from_vec(payload).into_page());
11830        for pair in offsets.windows(2) {
11831            push_value(&mut strings, ty, pair[0] as usize, (pair[1] - pair[0]) as usize)?;
11832        }
11833        let mut codes = Vec::with_capacity(rows);
11834        for _ in 0..rows {
11835            codes.push(cur.u32()?);
11836        }
11837        if codes.iter().any(|code| *code as usize >= count) {
11838            return Err(invalid("dictionary code is out of range"));
11839        }
11840        if cur.at != bytes.len() {
11841            return Err(invalid("dictionary page has trailing bytes"));
11842        }
11843        let dictionary = Vector::flat(ty.clone(), Data::Varlen(strings))?;
11844        return Ok(Vector::dictionary(codes, dictionary)?.with_validity(validity));
11845    }
11846    if codec == 3 || codec == 4 {
11847        let dictionary = global.ok_or_else(|| invalid("global code page has no dictionary"))?;
11848        let codes = if codec == 4 {
11849            // The cascade holds the whole tail of the page and says how long it is itself, so the
11850            // check that nothing is left over is the one the decoder already makes.
11851            let wide = integer::decode(&bytes[cur.at..])?;
11852            if wide.len() != rows {
11853                return Err(invalid("encoded code page holds the wrong number of rows"));
11854            }
11855            // Checked once for the page rather than a fallible conversion per code. Every code a
11856            // file holds is inside a `u32` or the file is corrupt, so or the codes together and the
11857            // answer has a bit set above the low thirty two, or the sign bit, exactly when one of
11858            // them did. The or and the narrowing are two passes because each is then a vector
11859            // loop. As one loop with a `push` a code, the length check and the store kept it scalar,
11860            // and it was sixteen instructions a row on the two flag columns of q1.
11861            let seen = wide.iter().fold(0_i64, |seen, &code| seen | code);
11862            if seen < 0 || seen > i64::from(u32::MAX) {
11863                return Err(invalid("code is not a code"));
11864            }
11865            wide.iter().map(|&code| code as u32).collect()
11866        } else {
11867            let mut codes = Vec::with_capacity(rows);
11868            for _ in 0..rows {
11869                codes.push(cur.u32()?);
11870            }
11871            if cur.at != bytes.len() {
11872                return Err(invalid("global code page has trailing bytes"));
11873            }
11874            codes
11875        };
11876        let highest = codes.iter().copied().max();
11877        return Ok(Vector::stable_dictionary_validated(codes, dictionary, highest)?
11878            .with_validity(validity));
11879    }
11880    if codec == 6 {
11881        if !coded_type(ty) {
11882            return Err(invalid("compressed text codec belongs to a non-string page"));
11883        }
11884        // As codec 5, the layer holds the whole tail of the page and says how long it is itself.
11885        // It comes back as one buffer with the values laid end to end and where each one ends, which
11886        // is the raw form's layout, so what is left to do here is what codec 0 does.
11887        let (payload, ends) = string::decode_flat(&bytes[cur.at..])?.into_parts();
11888        if ends.len() != rows {
11889            return Err(invalid("compressed text page holds the wrong number of rows"));
11890        }
11891        // A page, because this is read once and handed out a chunk at a time, and a cut of a paged
11892        // payload moves views rather than bytes.
11893        let mut values = StringColumn::over(Buffer::from_vec(payload).into_page());
11894        let mut start = 0;
11895        for end in ends {
11896            let len = end
11897                .checked_sub(start)
11898                .ok_or_else(|| invalid("compressed text value ends before it starts"))?;
11899            push_value(&mut values, ty, start, len)?;
11900            start = end;
11901        }
11902        return Ok(Vector::flat(ty.clone(), Data::Varlen(values))?.with_validity(validity));
11903    }
11904    if codec == 5 {
11905        // The cascade holds the whole tail of the page and says how long it is itself.
11906        let values = integer::decode(&bytes[cur.at..])?;
11907        if values.len() != rows {
11908            return Err(invalid("cascade page holds the wrong number of rows"));
11909        }
11910        let data = narrowed(ty, values)?;
11911        return Ok(Vector::flat(ty.clone(), data)?.with_validity(validity));
11912    }
11913    if codec == 2 {
11914        let width = u32::from(cur.u8()?);
11915        let base = i128::from_le_bytes(cur.take(16)?.try_into().expect("sixteen bytes"));
11916        let count = cur.u32()? as usize;
11917        let length = count.checked_mul(8).ok_or_else(|| invalid("packed page is too long"))?;
11918        let words: Vec<u64> = cur
11919            .take(length)?
11920            .chunks_exact(8)
11921            .map(|word| u64::from_le_bytes(word.try_into().expect("eight bytes")))
11922            .collect();
11923        if cur.at != bytes.len() {
11924            return Err(invalid("packed page has trailing bytes"));
11925        }
11926        return Ok(Vector::packed(ty.clone(), words, width, base, rows)?.with_validity(validity));
11927    }
11928    if codec != 0 {
11929        return Err(invalid("page codec is unknown"));
11930    }
11931    let data = match ty {
11932        LogicalType::TinyInt => {
11933            let values = cur.take(rows)?;
11934            Data::Int8(values.iter().map(|item| *item as i8).collect::<Vec<_>>().into())
11935        }
11936        LogicalType::UTinyInt => Data::UInt8(cur.take(rows)?.to_vec().into()),
11937        LogicalType::SmallInt => {
11938            let values =
11939                cur.take(rows.checked_mul(2).ok_or_else(|| invalid("page size overflow"))?)?;
11940            Data::Int16(
11941                values
11942                    .chunks_exact(2)
11943                    .map(|item| i16::from_le_bytes(item.try_into().expect("two bytes")))
11944                    .collect::<Vec<_>>()
11945                    .into(),
11946            )
11947        }
11948        LogicalType::USmallInt => {
11949            let values =
11950                cur.take(rows.checked_mul(2).ok_or_else(|| invalid("page size overflow"))?)?;
11951            Data::UInt16(
11952                values
11953                    .chunks_exact(2)
11954                    .map(|item| u16::from_le_bytes(item.try_into().expect("two bytes")))
11955                    .collect::<Vec<_>>()
11956                    .into(),
11957            )
11958        }
11959        LogicalType::UInteger => {
11960            let values =
11961                cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
11962            Data::UInt32(
11963                values
11964                    .chunks_exact(4)
11965                    .map(|item| u32::from_le_bytes(item.try_into().expect("four bytes")))
11966                    .collect::<Vec<_>>()
11967                    .into(),
11968            )
11969        }
11970        LogicalType::UBigInt => {
11971            let values =
11972                cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
11973            Data::UInt64(
11974                values
11975                    .chunks_exact(8)
11976                    .map(|item| u64::from_le_bytes(item.try_into().expect("eight bytes")))
11977                    .collect::<Vec<_>>()
11978                    .into(),
11979            )
11980        }
11981        LogicalType::Integer | LogicalType::Date => {
11982            let values =
11983                cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
11984            Data::Int32(
11985                values
11986                    .chunks_exact(4)
11987                    .map(|item| i32::from_le_bytes(item.try_into().expect("four bytes")))
11988                    .collect::<Vec<_>>()
11989                    .into(),
11990            )
11991        }
11992        LogicalType::BigInt
11993        | LogicalType::Timestamp
11994        | LogicalType::Time
11995        | LogicalType::TimeTz
11996        | LogicalType::TimestampTz
11997        | LogicalType::TimestampS
11998        | LogicalType::TimestampMs
11999        | LogicalType::TimestampNs => {
12000            let values =
12001                cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
12002            Data::Int64(
12003                values
12004                    .chunks_exact(8)
12005                    .map(|item| i64::from_le_bytes(item.try_into().expect("eight bytes")))
12006                    .collect::<Vec<_>>()
12007                    .into(),
12008            )
12009        }
12010        LogicalType::HugeInt | LogicalType::Uuid => {
12011            let values =
12012                cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
12013            Data::Int128(
12014                values
12015                    .chunks_exact(16)
12016                    .map(|item| i128::from_le_bytes(item.try_into().expect("sixteen bytes")))
12017                    .collect::<Vec<_>>()
12018                    .into(),
12019            )
12020        }
12021        LogicalType::UHugeInt => {
12022            let values =
12023                cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
12024            Data::UInt128(
12025                values
12026                    .chunks_exact(16)
12027                    .map(|item| u128::from_le_bytes(item.try_into().expect("sixteen bytes")))
12028                    .collect::<Vec<_>>()
12029                    .into(),
12030            )
12031        }
12032        LogicalType::Float => {
12033            let values =
12034                cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
12035            Data::Float32(
12036                values
12037                    .chunks_exact(4)
12038                    .map(|item| f32::from_le_bytes(item.try_into().expect("four bytes")))
12039                    .collect::<Vec<_>>()
12040                    .into(),
12041            )
12042        }
12043        LogicalType::Double => {
12044            let values =
12045                cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
12046            Data::Float64(
12047                values
12048                    .chunks_exact(8)
12049                    .map(|item| f64::from_le_bytes(item.try_into().expect("eight bytes")))
12050                    .collect::<Vec<_>>()
12051                    .into(),
12052            )
12053        }
12054        LogicalType::Interval => {
12055            let values =
12056                cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
12057            Data::Interval(
12058                values
12059                    .chunks_exact(16)
12060                    .map(|item| {
12061                        (
12062                            i32::from_le_bytes(item[..4].try_into().expect("four bytes")),
12063                            i32::from_le_bytes(item[4..8].try_into().expect("four bytes")),
12064                            i64::from_le_bytes(item[8..].try_into().expect("eight bytes")),
12065                        )
12066                    })
12067                    .collect::<Vec<_>>()
12068                    .into(),
12069            )
12070        }
12071        LogicalType::Boolean => {
12072            let values = cur.take(rows)?;
12073            if values.iter().any(|value| *value > 1) {
12074                return Err(invalid("boolean page has another value"));
12075            }
12076            Data::Bool(values.iter().map(|value| *value == 1).collect::<Vec<_>>().into())
12077        }
12078        // Whichever integer the declared width says, which is the mapping the rest of the engine
12079        // already uses for a decimal in memory.
12080        LogicalType::Decimal { .. } => match ty.physical() {
12081            PhysicalType::Int16 => {
12082                let values =
12083                    cur.take(rows.checked_mul(2).ok_or_else(|| invalid("page size overflow"))?)?;
12084                Data::Int16(
12085                    values
12086                        .chunks_exact(2)
12087                        .map(|item| i16::from_le_bytes(item.try_into().expect("two bytes")))
12088                        .collect::<Vec<_>>()
12089                        .into(),
12090                )
12091            }
12092            PhysicalType::Int32 => {
12093                let values =
12094                    cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
12095                Data::Int32(
12096                    values
12097                        .chunks_exact(4)
12098                        .map(|item| i32::from_le_bytes(item.try_into().expect("four bytes")))
12099                        .collect::<Vec<_>>()
12100                        .into(),
12101                )
12102            }
12103            PhysicalType::Int64 => {
12104                let values =
12105                    cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
12106                Data::Int64(
12107                    values
12108                        .chunks_exact(8)
12109                        .map(|item| i64::from_le_bytes(item.try_into().expect("eight bytes")))
12110                        .collect::<Vec<_>>()
12111                        .into(),
12112                )
12113            }
12114            _ => {
12115                let values =
12116                    cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
12117                Data::Int128(
12118                    values
12119                        .chunks_exact(16)
12120                        .map(|item| i128::from_le_bytes(item.try_into().expect("sixteen bytes")))
12121                        .collect::<Vec<_>>()
12122                        .into(),
12123                )
12124            }
12125        },
12126        LogicalType::Varchar | LogicalType::Blob | LogicalType::Bit => {
12127            let offset_bytes = cur
12128                .take((rows + 1).checked_mul(4).ok_or_else(|| invalid("offset count overflow"))?)?;
12129            let offsets = offset_bytes
12130                .chunks_exact(4)
12131                .map(|part| u32::from_le_bytes(part.try_into().expect("four bytes")))
12132                .collect::<Vec<_>>();
12133            let payload = cur.take(bytes.len() - cur.at)?.to_vec();
12134            if offsets.first() != Some(&0)
12135                || offsets.last().copied().map(|last| last as usize) != Some(payload.len())
12136                || offsets.windows(2).any(|pair| pair[0] > pair[1])
12137            {
12138                return Err(invalid("string offsets do not bound the payload"));
12139            }
12140            // A page for the reason the dictionary payload above is one: the page is read once and
12141            // handed out a chunk at a time, and a cut of a paged payload moves views rather than
12142            // bytes.
12143            //
12144            // A varchar is checked for text on the way in and a blob and a bit string are not,
12145            // because the second pair never claimed to hold any. Reading them through the checking
12146            // seam would refuse a column for holding exactly what it was told to hold.
12147            let mut values = StringColumn::over(Buffer::from_vec(payload).into_page());
12148            let text = ty == &LogicalType::Varchar;
12149            for pair in offsets.windows(2) {
12150                let (at, len) = (pair[0] as usize, (pair[1] - pair[0]) as usize);
12151                if text {
12152                    values.push_in_place(at, len)?;
12153                } else {
12154                    values.push_bytes_in_place(at, len)?;
12155                }
12156            }
12157            Data::Varlen(values)
12158        }
12159        _ => return Err(Error::not_implemented(format!("native page for {ty}"))),
12160    };
12161    if cur.at != bytes.len() {
12162        return Err(invalid("page has trailing bytes"));
12163    }
12164    Ok(Vector::flat(ty.clone(), data)?.with_validity(validity))
12165}
12166
12167#[cfg(test)]
12168mod tests {
12169    use std::fs::{self, OpenOptions};
12170    use std::io::{Seek, SeekFrom, Write};
12171    use std::path::PathBuf;
12172    use std::time::{SystemTime, UNIX_EPOCH};
12173
12174    use rudb_common::Stat;
12175    use rudb_common::Value;
12176    use rudb_common::bounds::{Frequencies, Op, Remainder, Zones};
12177    use rudb_common::stat::Provenance;
12178
12179    use super::*;
12180
12181    #[test]
12182    fn a_name_taken_in_pieces_is_the_name_of_the_pieces_joined() {
12183        let bytes: Vec<u8> =
12184            (0..300_u32).map(|at| (at.wrapping_mul(2_654_435_761) >> 13) as u8).collect();
12185        for length in [0, 1, 7, 31, 32, 33, 63, 64, 65, 100, 300] {
12186            let whole = content_name(&bytes[..length]);
12187            for step in [1, 3, 8, 31, 32, 33, 64, 301] {
12188                let mut namer = ContentNamer::default();
12189                bytes[..length].chunks(step).for_each(|piece| namer.update(piece));
12190                assert_eq!(namer.finish(), whole, "{length} bytes in pieces of {step}");
12191            }
12192        }
12193    }
12194
12195    /// The chooser as it was before it could rule kinds out up front: the same narrowing, with every
12196    /// kind tested for. What it writes is what the file used to hold.
12197    #[derive(Debug)]
12198    struct TestsEverything<'a>(&'a dyn chooser::Chooser);
12199
12200    impl chooser::Chooser for TestsEverything<'_> {
12201        fn name(&self) -> &'static str {
12202            "tests everything"
12203        }
12204
12205        fn narrow_strings(
12206            &self,
12207            values: &[&[u8]],
12208            offered: &[string::Kind],
12209            depth: u8,
12210        ) -> Vec<string::Kind> {
12211            self.0.narrow_strings(values, offered, depth)
12212        }
12213
12214        fn narrow_integers(
12215            &self,
12216            values: &[i64],
12217            offered: &[integer::Kind],
12218            depth: u8,
12219        ) -> Vec<integer::Kind> {
12220            self.0.narrow_integers(values, offered, depth)
12221        }
12222    }
12223
12224    #[test]
12225    fn ruling_kinds_out_before_testing_for_them_writes_the_same_bytes() {
12226        let columns: Vec<Vec<i64>> = vec![
12227            vec![],
12228            vec![5; 1000],
12229            (0..1000).collect(),
12230            (0..1000).map(|row| 1_600_000_000_000_000 + row * 1_000_000).collect(),
12231            (0..1000).map(|row| row / 50).collect(),
12232            (0..1000).map(|row| if row % 97 == 0 { row } else { 0 }).collect(),
12233            (0..1000).map(|row| (row * 7919) % 13).collect(),
12234            (0..1000).map(|row| (row * 2_654_435_761) % 1_000_003).collect(),
12235            (0..1000).map(|row| [3, 3, 3, 9, 9, 1][row as usize % 6]).collect(),
12236            (0..1000).map(|row| i64::MIN + row % 3).collect(),
12237        ];
12238        let choosers: [&dyn chooser::Chooser; 2] = [&Fixed, &Codes];
12239        for column in &columns {
12240            for chooser in choosers {
12241                let quick = integer::encode_with(column, chooser).unwrap();
12242                let full = integer::encode_with(column, &TestsEverything(chooser)).unwrap();
12243                assert_eq!(
12244                    quick,
12245                    full,
12246                    "{} on {:?}",
12247                    chooser.name(),
12248                    &column[..column.len().min(8)]
12249                );
12250            }
12251        }
12252    }
12253
12254    /// Parts of a column that all look alike come out of a settled shape byte for byte as they
12255    /// come out of a search, because the search would have kept the same tree on every one.
12256    #[test]
12257    fn parts_that_look_alike_replay_to_the_bytes_a_search_writes() {
12258        let mut settling = Settling::default();
12259        for part in 0..STRIPE_PARTS as i64 {
12260            let values: Vec<i64> = (0..2048)
12261                .map(|row| 1_600_000_000_000_000 + (part * 2048 + row) * 1_000_000 + row % 7)
12262                .collect();
12263            let searched = integer::encode_with(&values, &Fixed).unwrap();
12264            assert_eq!(settling.encode(&values).unwrap(), searched, "part {part}");
12265        }
12266    }
12267
12268    /// A column that changes shape partway through a stripe still reads back, and no part comes
12269    /// out much bigger than a search would have made it, because a replay that stops fitting or
12270    /// grows past a quarter a row is searched.
12271    #[test]
12272    fn a_column_that_changes_under_the_shape_is_searched_again() {
12273        let mut state = 0x9e37_79b9_7f4a_7c15_u64;
12274        let mut noise = move || {
12275            state ^= state << 13;
12276            state ^= state >> 7;
12277            state ^= state << 17;
12278            (state % 1_000_000) as i64
12279        };
12280        let mut settling = Settling::default();
12281        for part in 0..STRIPE_PARTS as i64 {
12282            let values: Vec<i64> = match part / 16 {
12283                0 => (0..2048).map(|row| (part * 2048 + row) / 300).collect(),
12284                1 => (0..2048).map(|_| noise()).collect(),
12285                2 => (0..2048).map(|row| if row % 97 == 0 { row } else { 42 }).collect(),
12286                _ => (0..2048).map(|row| 5 + (part * 2048 + row) * 1_000_000).collect(),
12287            };
12288            let settled = settling.encode(&values).unwrap();
12289            assert_eq!(integer::decode(&settled).unwrap(), values, "part {part}");
12290            let searched = integer::encode_with(&values, &Fixed).unwrap();
12291            assert!(
12292                settled.len() * 4 <= searched.len() * 5,
12293                "part {part}: {} settled against {} searched, {} against {}",
12294                settled.len(),
12295                searched.len(),
12296                integer::describe(&settled).unwrap(),
12297                integer::describe(&searched).unwrap(),
12298            );
12299        }
12300    }
12301
12302    #[test]
12303    fn checksum_matches_fixed_vectors() {
12304        assert_eq!(checksum(b""), 0xef46_db37_51d8_e999);
12305        assert_eq!(checksum(b"a"), 0xd24e_c4f1_a98c_6e5b);
12306        assert_eq!(checksum(b"abc"), 0x44bc_2cf5_ad77_0999);
12307    }
12308
12309    #[test]
12310    fn sorting_across_threads_matches_sorting_on_one() {
12311        let mut state = 0x9e37_79b9_7f4a_7c15_u64;
12312        let mut next = move || {
12313            state ^= state << 13;
12314            state ^= state >> 7;
12315            state ^= state << 17;
12316            state
12317        };
12318        let mut values = Vec::new();
12319        for at in 0..150_000_u64 {
12320            let value = match next() % 6 {
12321                0 => Vec::new(),
12322                1 => format!("https://example.com/{}", next() % 5_000).into_bytes(),
12323                2 => format!("https://example.com/path/{at}").into_bytes(),
12324                3 => b"same".to_vec(),
12325                4 => vec![0xff; (next() % 12) as usize],
12326                _ => (0..next() % 20).map(|_| (next() % 3) as u8).collect(),
12327            };
12328            values.push(value);
12329        }
12330        let value = |code: u32| values[code as usize].as_slice();
12331        for workers in [1, 2, 3, 8, 32] {
12332            let mut one = (0..values.len() as u32).rev().collect::<Vec<_>>();
12333            let mut across = one.clone();
12334            sort_by_value(&mut one, value);
12335            sort_by_value_across(&mut across, value, workers);
12336            assert_eq!(one, across, "{workers} workers");
12337        }
12338        let mut sorted = (0..values.len() as u32).collect::<Vec<_>>();
12339        sort_by_value_across(&mut sorted, value, 8);
12340        assert!(sorted.windows(2).all(|pair| value(pair[0]) <= value(pair[1])));
12341    }
12342
12343    fn path(label: &str) -> PathBuf {
12344        let stamp = SystemTime::now().duration_since(UNIX_EPOCH).expect("time advances").as_nanos();
12345        std::env::temp_dir().join(format!("rudb-native-{label}-{}-{stamp}.rdb", std::process::id()))
12346    }
12347
12348    /// Every value of a dictionary in code order, which the tests have no other way to ask for now
12349    /// that a dictionary does not keep the bytes of the values it has seen.
12350    ///
12351    /// Only valid once `finish_blocks` has run, because until then the last part block is still raw.
12352    fn dictionary_values(dictionary: &GlobalDictionary) -> Vec<Vec<u8>> {
12353        let (flat, bases) = dictionary.decoded(None).expect("the blocks decode");
12354        (0..dictionary.values())
12355            .map(|code| {
12356                let (from, to) = GlobalDictionary::value_span(&dictionary.ends, &bases, code);
12357                flat[from..to].to_vec()
12358            })
12359            .collect()
12360    }
12361
12362    /// The sections a test put in the table, which is every one the writer did not.
12363    ///
12364    /// A table now carries a summary and a sketch per column out of the write itself, and a test
12365    /// about the section table is not about those. Filtering by kind rather than by count, so a
12366    /// table that turns out to have no room for its summaries does not quietly change what these
12367    /// tests are asserting over.
12368    fn attached(table: &Table) -> Vec<&Section> {
12369        table.sections().iter().filter(|held| !held.among(section::STATISTICS_KINDS)).collect()
12370    }
12371
12372    /// A read names the offset it wants, so a cursor somebody else moved cannot reach it.
12373    #[test]
12374    fn a_read_at_an_offset_ignores_where_another_thread_left_the_cursor() {
12375        const SPANS: usize = 64;
12376        const SPAN: usize = 512;
12377        let path = path("positional");
12378        let content: Vec<u8> =
12379            (0..SPANS).flat_map(|span| std::iter::repeat_n(span as u8, SPAN)).collect();
12380        fs::write(&path, &content).expect("the file is written");
12381        let file = Arc::new(File::open(&path).expect("the file opens"));
12382        std::thread::scope(|scope| {
12383            for _ in 0..8 {
12384                let file = Arc::clone(&file);
12385                scope.spawn(move || {
12386                    for _ in 0..64 {
12387                        for span in 0..SPANS {
12388                            let mut bytes = [0_u8; SPAN];
12389                            read_at(&file, (span * SPAN) as u64, &mut bytes)
12390                                .expect("the span reads");
12391                            assert!(
12392                                bytes.iter().all(|byte| *byte == span as u8),
12393                                "span {span} came back as {}",
12394                                bytes[0],
12395                            );
12396                        }
12397                    }
12398                });
12399            }
12400        });
12401        let mut past = [0_u8; SPAN];
12402        let end = (SPANS * SPAN) as u64;
12403        let error = read_at(&file, end, &mut past).expect_err("a read past the end is refused");
12404        assert!(error.message().contains("ends before its declared length"), "{error}");
12405        drop(file);
12406        let _ = fs::remove_file(&path);
12407    }
12408
12409    /// The writer records where it put a page and puts it there.
12410    ///
12411    /// This used to move the file's cursor between the steps that record an offset, which is what
12412    /// reading the pages back to build the frequencies did on a platform with no `pread`, and the
12413    /// directory landed on top of a page. The writer's file is an `rudb_io` file now and has no
12414    /// cursor to move, so what is left is the check that every page is where the directory says.
12415    #[test]
12416    fn a_writer_puts_a_page_where_it_said_it_did_wherever_the_cursor_has_got_to() {
12417        let path = path("cursor");
12418        let mut writer = Writer::create(
12419            &path,
12420            "items",
12421            vec![
12422                Field::required("id", LogicalType::Integer),
12423                Field::new("text", LogicalType::Varchar),
12424            ],
12425        )
12426        .expect("new file");
12427        writer.append(&sample()).expect("first part");
12428        writer.append(&sample()).expect("second part");
12429        writer.finish().expect("commit");
12430        let reader = Reader::open(&path).expect("reopen from disk");
12431        assert_eq!(reader.table().rows(), 6);
12432        let ids = reader.read(0, &[0]).expect("the integer page reads back");
12433        assert_eq!(ids.value_at(0, 0), Value::Integer(4));
12434        assert_eq!(ids.value_at(2, 0), Value::Integer(-2));
12435        let text = reader.read(1, &[1]).expect("the text page reads back");
12436        assert_eq!(text.value_at(1, 0), Value::Null);
12437        assert_eq!(text.value_at(2, 0), Value::Varchar("long text after a slash".into()));
12438        // Nothing the directory points at may run past the end of the file, which is the shape the
12439        // failure took: a page recorded at an offset the directory had already been written over.
12440        let end = reader.table().stripes().iter().flat_map(|stripe| {
12441            stripe
12442                .pages
12443                .iter()
12444                .map(|page| page.offset + u64::from(page.length))
12445                .chain(std::iter::once(stripe.index.offset + u64::from(stripe.index.length)))
12446        });
12447        let last = end.fold(HEADER, u64::max);
12448        let directory = fs::metadata(&path).expect("the file is there").len();
12449        assert!(last <= directory, "a page runs to {last} in a file of {directory} bytes");
12450        fs::remove_file(path).expect("remove scratch file");
12451    }
12452
12453    /// How long a global dictionary index is, read out of the page's own header.
12454    ///
12455    /// The tests below damage a byte of the order or of the payload, so they need to know where each
12456    /// one starts, and working it out here rather than writing a number down means adding something
12457    /// to the index does not quietly turn one of them into a test that damages the index instead.
12458    fn dictionary_index_len(header: &[u8; DICTIONARY_HEADER]) -> u64 {
12459        let count = u64::from(u32::from_le_bytes(header[0..4].try_into().expect("four bytes")));
12460        let blocks = u64::from(u32::from_le_bytes(header[8..12].try_into().expect("four bytes")));
12461        let width = u32::from_le_bytes(header[12..16].try_into().expect("four bytes"));
12462        let bits = (width & !DICTIONARY_FLAGS) as usize;
12463        let payload_words = if width & DICTIONARY_SCATTERED == 0 { 2 } else { 3 };
12464        let rank_blocks = count.div_ceil(TEXT_RANK_BLOCK as u64);
12465        DICTIONARY_HEADER as u64
12466            + offset_bytes(count as usize, bits) as u64
12467            + blocks * payload_words * 8
12468            + rank_blocks * 16
12469            + if width & DICTIONARY_GRAMS == 0 { 0 } else { 8 }
12470    }
12471
12472    fn sample() -> Chunk {
12473        Chunk::new(vec![
12474            Vector::from_values(
12475                LogicalType::Integer,
12476                &[Value::Integer(4), Value::Integer(9), Value::Integer(-2)],
12477            )
12478            .expect("integers"),
12479            Vector::from_values(
12480                LogicalType::Varchar,
12481                &[
12482                    Value::Varchar("alpha".into()),
12483                    Value::Null,
12484                    Value::Varchar("long text after a slash".into()),
12485                ],
12486            )
12487            .expect("strings"),
12488        ])
12489        .expect("matching rows")
12490    }
12491
12492    fn sample_ids() -> Chunk {
12493        Chunk::new(vec![
12494            Vector::flat(LogicalType::Integer, Data::Int32(vec![7, 8, 9].into()))
12495                .expect("integers"),
12496        ])
12497        .expect("one column")
12498    }
12499
12500    #[test]
12501    fn the_planner_gets_the_null_count_off_the_same_directory_the_bounds_are_in() {
12502        // Six rows, two of them null. `IS NULL` used to get the same fifth any unreadable
12503        // condition gets, and the number was in the stripe entry next to the bounds all along.
12504        let path = path("nulls_for_the_planner");
12505        let mut writer =
12506            Writer::create(&path, "items", vec![Field::new("a", LogicalType::Integer)])
12507                .expect("new file");
12508        let rows = Chunk::new(vec![
12509            Vector::from_values(
12510                LogicalType::Integer,
12511                &[
12512                    Value::Integer(4),
12513                    Value::Null,
12514                    Value::Integer(9),
12515                    Value::Null,
12516                    Value::Integer(1),
12517                    Value::Integer(2),
12518                ],
12519            )
12520            .expect("integers"),
12521        ])
12522        .expect("one column");
12523        writer.append(&rows).expect("the only part");
12524        writer.finish().expect("commit");
12525        let reader = Reader::open(&path).expect("reopen from disk");
12526        let stripes = Stripes::new(reader);
12527        let column = stripes.column("a").expect("the file has that column");
12528        assert_eq!(stripes.nulls(column), Stat::exact(2, Provenance::NullCount));
12529        // A column the file does not have. Zero here would be a fact about a column that is not
12530        // there, which the planner would then divide by.
12531        assert_eq!(stripes.nulls(column + 1), Stat::Unknown);
12532        fs::remove_file(&path).expect("clean up");
12533    }
12534
12535    #[test]
12536    fn the_planner_gets_a_row_count_per_value_off_a_complete_synopsis() {
12537        // The whole of the frequency half of #1106, end to end over a real file. Six rows, three
12538        // of one value and two of another, and a complete synopsis because six rows is well inside
12539        // what the writer can account for. The estimate for `id = 4` is three rows rather than a
12540        // sixth of the table, and for a value the file does not hold it is none.
12541        let path = path("frequencies_for_the_planner");
12542        let mut writer =
12543            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
12544                .expect("new file");
12545        let rows = Chunk::new(vec![
12546            Vector::from_values(
12547                LogicalType::Integer,
12548                &[
12549                    Value::Integer(4),
12550                    Value::Integer(4),
12551                    Value::Integer(4),
12552                    Value::Integer(9),
12553                    Value::Integer(9),
12554                    Value::Integer(1),
12555                ],
12556            )
12557            .expect("integers"),
12558        ])
12559        .expect("one column");
12560        writer.append(&rows).expect("the only part");
12561        writer.finish().expect("commit");
12562        let reader = Reader::open(&path).expect("reopen from disk");
12563        let common = Common::new(reader);
12564        assert_eq!(common.rows(), 6);
12565        let column = common.column("id").expect("the file has that column");
12566        assert_eq!(common.column("nothing"), None);
12567        assert_eq!(
12568            common.rows_with(column, &Bound::Int(4)),
12569            Stat::exact(3, Provenance::FrequencySynopsis)
12570        );
12571        // Not in the file, and a synopsis that accounts for all six rows proves it.
12572        assert_eq!(
12573            common.rows_with(column, &Bound::Int(7)),
12574            Stat::exact(0, Provenance::FrequencySynopsis)
12575        );
12576        // A constant of another domain against an integer column. Nothing in the list compares
12577        // with it, so the zero above would be an artefact of the mismatch rather than a fact.
12578        assert_eq!(common.rows_with(column, &Bound::Bytes(b"four".to_vec())), Stat::Unknown);
12579        // A complete list has no remainder. Answering one of no rows over no values would hand the
12580        // caller a division to special case, and the counts above already answer this column.
12581        assert_eq!(common.remainder(column), None);
12582        fs::remove_file(&path).expect("clean up");
12583    }
12584
12585    #[test]
12586    fn string_frequency_estimates_do_not_open_the_global_dictionary() {
12587        let path = path("string_frequencies_for_the_planner");
12588        let mut writer =
12589            Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
12590                .expect("new file");
12591        let rows = Chunk::new(vec![
12592            Vector::from_values(
12593                LogicalType::Varchar,
12594                &[
12595                    Value::Varchar(String::new()),
12596                    Value::Varchar("alpha".into()),
12597                    Value::Varchar(String::new()),
12598                    Value::Varchar("beta".into()),
12599                    Value::Varchar(String::new()),
12600                ],
12601            )
12602            .expect("strings"),
12603        ])
12604        .expect("one column");
12605        writer.append(&rows).expect("the only part");
12606        writer.finish().expect("commit");
12607
12608        let reader = Reader::open(&path).expect("reopen from disk");
12609        assert_eq!(reader.reads().dictionaries, 0, "open reads only the directory");
12610        let common = Common::new(reader.clone());
12611        let column = common.column("text").expect("the file has that column");
12612        assert_eq!(
12613            common.rows_with(column, &Bound::Bytes(Vec::new())),
12614            Stat::exact(3, Provenance::FrequencySynopsis)
12615        );
12616        assert_eq!(
12617            common.rows_with(column, &Bound::Bytes(b"missing".to_vec())),
12618            Stat::exact(0, Provenance::FrequencySynopsis)
12619        );
12620        assert_eq!(
12621            reader.reads().dictionaries,
12622            0,
12623            "the bounded spellings answer without opening the dictionary index"
12624        );
12625        fs::remove_file(&path).expect("clean up");
12626    }
12627
12628    #[test]
12629    fn host_groups_certify_omitted_hosts_and_keep_exact_aggregates() {
12630        let path = path("certified_host_groups");
12631        let mut writer =
12632            Writer::create(&path, "hits", vec![Field::required("Referer", LogicalType::Varchar)])
12633                .expect("new file");
12634        let mut values = vec![Value::Varchar("http://www.example.com/a".into()); 150];
12635        values.extend(vec![Value::Varchar("https://example.com/b".into()); 70]);
12636        values.extend((0..550).map(|at| Value::Varchar(format!("https://site{at}.test/x"))));
12637        values.push(Value::Varchar(String::new()));
12638        for part in values.chunks(512) {
12639            writer
12640                .append(
12641                    &Chunk::new(vec![
12642                        Vector::from_values(LogicalType::Varchar, part).expect("strings"),
12643                    ])
12644                    .expect("one column"),
12645                )
12646                .expect("part written");
12647        }
12648        writer.finish().expect("commit");
12649        let reader = Reader::open(&path).expect("reopen");
12650        assert!(reader.table.host_groups.is_none(), "no query-specific host result is stored");
12651        fs::remove_file(&path).expect("clean up");
12652    }
12653
12654    /// A table directory with nothing in it but a name and one column, for the section tests.
12655    ///
12656    /// The section table is orthogonal to everything else in a directory, so the tests that pin it
12657    /// say so by starting from the emptiest table that encodes.
12658    fn bare_table(sections: Vec<Section>) -> Table {
12659        Table {
12660            name: "linked".to_owned(),
12661            fields: vec![Field::required("id", LogicalType::Integer)],
12662            stripes: Vec::new(),
12663            rows: 0,
12664            dictionaries: vec![None],
12665            dictionary_payloads: Vec::new(),
12666            demoted: Vec::new(),
12667            distincts: vec![None],
12668            frequencies: vec![None],
12669            pair_frequencies: Vec::new(),
12670            frequency_texts: Vec::new(),
12671            host_groups: None,
12672            clustering: None,
12673            generation: 1,
12674            sections,
12675        }
12676    }
12677
12678    fn a_key_map_section() -> Section {
12679        Section {
12680            kind: *section::KEY_MAP,
12681            id: 1,
12682            generation: 3,
12683            extents: 1,
12684            extent_page: HEADER,
12685            extent_bytes: section::EXTENT_BYTES as u32,
12686            hash: 0x1234_5678_9abc_def0,
12687            flags: 0,
12688            header_bytes: 24,
12689        }
12690    }
12691
12692    #[test]
12693    fn a_section_table_round_trips_through_a_directory() {
12694        let mut later = a_key_map_section();
12695        later.kind = *b"RUDBZZ9\0";
12696        later.id = 2;
12697        let table = bare_table(vec![a_key_map_section(), later]);
12698        let directory = encode_directory(&table).expect("directory");
12699        let decoded = decode_directory(&directory, 1 << 20).expect("reopen");
12700        assert_eq!(decoded.sections(), &[a_key_map_section(), later]);
12701        // The second is a kind this build has no name for, and it survived the round trip anyway.
12702        // That is what keeps an old build from silently discarding a newer build's work when it
12703        // rewrites a directory.
12704        assert!(decoded.sections()[0].known());
12705        assert!(!decoded.sections()[1].known());
12706    }
12707
12708    #[test]
12709    fn a_directory_written_before_the_section_table_reads_as_a_table_with_none() {
12710        // The G1 exit criterion, at the directory level. A format 22 directory is exactly this
12711        // build's directory with the trailing section block cut off, so cutting it off is the
12712        // honest way to make one: no fixture to go stale, and no separate encoder to drift.
12713        let directory = encode_directory(&bare_table(Vec::new())).expect("directory");
12714        let block = SECTIONS.len() + size_of::<u64>() + size_of::<u16>();
12715        let older = &directory[..directory.len() - block];
12716        let decoded = decode_directory(older, 1 << 20).expect("a directory from before sections");
12717        assert!(decoded.sections().is_empty());
12718        assert_eq!(decoded.generation(), 0, "a format 22 table recorded no generation");
12719        assert_eq!(decoded.name(), "linked");
12720        assert_eq!(decoded.fields().len(), 1, "everything before the block still decodes");
12721    }
12722
12723    #[test]
12724    fn a_file_stamped_with_the_previous_format_still_opens_and_reads() {
12725        // The same criterion end to end, which is the one the milestone actually asks for: a build
12726        // that knows about sections opens a file written by a build that did not, with no rewrite
12727        // and no repair, and answers from it. The version field is patched rather than a file
12728        // committed by an old binary because the bytes either side of it are identical: format 22
12729        // and format 23 differ only in a trailing directory block, and a reader that stops before
12730        // that block gets a table with no sections.
12731        let path = path("format_twenty_two");
12732        let mut writer =
12733            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
12734                .expect("new file");
12735        let rows = Chunk::new(vec![
12736            Vector::from_values(
12737                LogicalType::Integer,
12738                &[Value::Integer(1), Value::Integer(2), Value::Integer(3)],
12739            )
12740            .expect("integers"),
12741        ])
12742        .expect("one column");
12743        writer.append(&rows).expect("the only part");
12744        writer.finish().expect("commit");
12745
12746        let file = OpenOptions::new().write(true).open(&path).expect("reopen to patch");
12747        write_at(&file, 8, &22_u32.to_le_bytes()).expect("stamp the older format");
12748        drop(file);
12749
12750        let reader = Reader::open(&path).expect("a format 22 file opens unchanged");
12751        assert_eq!(reader.table().rows(), 3);
12752        // The rows and not the section table, because the section block is found by the magic at
12753        // the end of the directory rather than by the number in the header, so stamping the header
12754        // back does not take away the summaries this writer put there. What the test is about is
12755        // that the version check accepts 22, and the rows coming back is what says it did.
12756        assert_eq!(reader.read(0, &[0]).expect("the part still reads").len(), 3);
12757
12758        // And a format this build has never written is still refused, so the accept set is a list
12759        // and not an absence of a check.
12760        let file = OpenOptions::new().write(true).open(&path).expect("reopen to patch");
12761        write_at(&file, 8, &21_u32.to_le_bytes()).expect("stamp an unreadable format");
12762        drop(file);
12763        let error = Reader::open(&path).expect_err("format 21 is not readable");
12764        assert!(error.to_string().contains("format 21"), "{error}");
12765
12766        fs::remove_file(&path).expect("clean up");
12767    }
12768
12769    #[test]
12770    fn a_section_whose_extent_table_is_outside_the_file_is_refused() {
12771        // The bound the format has to check and `section` cannot, because only the reader knows how
12772        // big the file is. Reading the payload a section like this names would be reading whatever
12773        // else happens to be at that offset, which is the one way a graph section could turn into a
12774        // wrong answer rather than a slow one.
12775        let mut past = a_key_map_section();
12776        past.extent_page = 1 << 30;
12777        let directory = encode_directory(&bare_table(vec![past])).expect("directory");
12778        let error = decode_directory(&directory, 1 << 20).expect_err("refused");
12779        assert!(error.to_string().contains("outside the file"), "{error}");
12780
12781        let mut inside_the_header = a_key_map_section();
12782        inside_the_header.extent_page = 8;
12783        let directory = encode_directory(&bare_table(vec![inside_the_header])).expect("directory");
12784        assert!(
12785            decode_directory(&directory, 1 << 20).is_err(),
12786            "a section may not overlap a header"
12787        );
12788    }
12789
12790    #[test]
12791    fn a_section_recorded_as_not_built_is_legal_and_names_no_bytes() {
12792        // Section 3.7: a relationship that does not fit the budget is recorded with its size so
12793        // that `rudb_links()` can report what a larger budget would buy. That record is a section
12794        // entry with no extents, so it has to survive a round trip while naming nothing.
12795        let not_built = Section {
12796            kind: *section::FORWARD_LINK,
12797            id: 9,
12798            generation: 3,
12799            extents: 0,
12800            extent_page: 0,
12801            extent_bytes: 0,
12802            hash: 0,
12803            flags: 0,
12804            header_bytes: 0,
12805        };
12806        let directory = encode_directory(&bare_table(vec![not_built])).expect("directory");
12807        let decoded = decode_directory(&directory, 1 << 20).expect("reopen");
12808        assert_eq!(decoded.sections(), &[not_built]);
12809
12810        // But a section with no extents that still names an extent table is incoherent, and an
12811        // incoherent entry is a torn directory rather than a relationship that was skipped.
12812        let mut incoherent = not_built;
12813        incoherent.extent_bytes = 28;
12814        incoherent.extent_page = HEADER;
12815        let directory = encode_directory(&bare_table(vec![incoherent])).expect("directory");
12816        assert!(decode_directory(&directory, 1 << 20).is_err());
12817    }
12818
12819    #[test]
12820    fn a_directory_naming_more_sections_than_the_bound_is_refused() {
12821        let directory = encode_directory(&bare_table(Vec::new())).expect("directory");
12822        let mut torn = directory.clone();
12823        let count_at = torn.len() - size_of::<u16>();
12824        torn[count_at..].copy_from_slice(&u16::MAX.to_le_bytes());
12825        // Not an allocation of sixty five thousand entries off a torn count: either the bound
12826        // refuses it or the bytes run out, and both are errors rather than a read past the end.
12827        assert!(decode_directory(&torn, 1 << 20).is_err());
12828    }
12829
12830    /// A committed one column file of `rows` integers, for the attach tests.
12831    fn linked_file(label: &str, rows: i32) -> PathBuf {
12832        let path = path(label);
12833        let mut writer =
12834            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
12835                .expect("new file");
12836        let values = (0..rows).map(Value::Integer).collect::<Vec<_>>();
12837        let chunk =
12838            Chunk::new(vec![Vector::from_values(LogicalType::Integer, &values).expect("integers")])
12839                .expect("one column");
12840        writer.append(&chunk).expect("the only part");
12841        writer.finish().expect("commit");
12842        path
12843    }
12844
12845    fn a_key_map_payload() -> Vec<u8> {
12846        // Shaped like one without being one: this crate never reads a payload, so what matters here
12847        // is that every byte comes back and that the header the entry measures is at the front.
12848        (0..512_u32).flat_map(u32::to_le_bytes).collect()
12849    }
12850
12851    #[test]
12852    fn a_section_attached_to_a_committed_file_reads_back_byte_for_byte() {
12853        let path = linked_file("attach", 64);
12854        let payload = a_key_map_payload();
12855        let table = attach(
12856            &path,
12857            "items",
12858            &[section::Attachment {
12859                kind: *section::KEY_MAP,
12860                id: 0,
12861                flags: 2,
12862                header_bytes: 40,
12863                bytes: &payload,
12864            }],
12865        )
12866        .expect("attach a key map");
12867        assert_eq!(attached(&table).len(), 1);
12868
12869        let reader = Reader::open(&path).expect("reopen after the attach");
12870        let held = attached(reader.table());
12871        assert_eq!(held.len(), 1);
12872        assert_eq!(held[0].kind, *section::KEY_MAP);
12873        assert_eq!(held[0].flags, 2, "the form a reader must not have to guess");
12874        assert_eq!(held[0].header_bytes, 40);
12875        // The generation is the one the pages were written at, not the one the attach committed at.
12876        // Attaching a section moved no row, so a section written by it is current, and a second
12877        // table added to this file later would not make it stale.
12878        assert_eq!(held[0].generation, 1);
12879        assert!(held[0].usable(reader.table().generation()));
12880        assert_eq!(reader.payload(held[0]).expect("read the payload"), payload);
12881        assert_eq!(reader.extents(held[0]).expect("extent table").len(), 1);
12882
12883        fs::remove_file(&path).expect("clean up");
12884    }
12885
12886    #[test]
12887    fn attaching_a_section_answers_every_row_exactly_as_before() {
12888        // Section 3.1 end to end, and the reason the whole layer is safe to build incrementally. A
12889        // file with a section in it and the same file without one have to agree row for row, so the
12890        // comparison is made against the answers taken before the attach rather than against a
12891        // constant somebody typed.
12892        let path = linked_file("attach_changes_nothing", 300);
12893        let before = Reader::open(&path).expect("open before");
12894        let rows = before.table().rows();
12895        let first = before.read(0, &[0]).expect("read before");
12896        let values = (0..rows).map(|at| first.value_at(at, 0)).collect::<Vec<_>>();
12897        let layout = before.layout().columns_total();
12898        drop(before);
12899
12900        let payload = a_key_map_payload();
12901        attach(
12902            &path,
12903            "items",
12904            &[section::Attachment {
12905                kind: *section::KEY_MAP,
12906                id: 0,
12907                flags: 0,
12908                header_bytes: 0,
12909                bytes: &payload,
12910            }],
12911        )
12912        .expect("attach");
12913
12914        let after = Reader::open(&path).expect("open after");
12915        assert_eq!(after.table().rows(), rows);
12916        let read = after.read(0, &[0]).expect("read after");
12917        for (at, value) in values.iter().enumerate() {
12918            assert_eq!(&read.value_at(at, 0), value, "row {at} moved");
12919        }
12920        assert_eq!(
12921            after.layout().columns_total(),
12922            layout,
12923            "an attach appends and does not rewrite a column page"
12924        );
12925
12926        fs::remove_file(&path).expect("clean up");
12927    }
12928
12929    #[test]
12930    fn a_rebuilt_section_replaces_the_one_it_supersedes() {
12931        // Rebuilding a key map has to be a write and not a question. If an attach added rather than
12932        // replaced, a table rebuilt a few times would name several maps for one column and a reader
12933        // would have to pick, which is a decision with no right answer in it.
12934        let path = linked_file("attach_twice", 32);
12935        let one = a_key_map_payload();
12936        let two = vec![7_u8; 1024];
12937        let entry = |bytes| section::Attachment {
12938            kind: *section::KEY_MAP,
12939            id: 4,
12940            flags: 1,
12941            header_bytes: 0,
12942            bytes,
12943        };
12944        attach(&path, "items", &[entry(&one)]).expect("first build");
12945        attach(&path, "items", &[entry(&two)]).expect("rebuild");
12946
12947        let reader = Reader::open(&path).expect("reopen");
12948        let held = attached(reader.table());
12949        assert_eq!(held.len(), 1, "one map per column and not one per build");
12950        assert_eq!(reader.payload(held[0]).expect("payload"), two);
12951
12952        fs::remove_file(&path).expect("clean up");
12953    }
12954
12955    #[test]
12956    fn an_attach_carries_through_a_kind_it_does_not_know() {
12957        // The first of section 3.2's three rules, at the point where it is easiest to break: a build
12958        // that rewrites a directory has to carry an entry it has no name for, or opening a file with
12959        // an older binary and attaching one section quietly deletes the work of a newer one.
12960        let path = linked_file("attach_unknown", 16);
12961        let payload = vec![3_u8; 96];
12962        attach(
12963            &path,
12964            "items",
12965            &[section::Attachment {
12966                kind: *b"RUDBZZ9\0",
12967                id: 1,
12968                flags: 0,
12969                header_bytes: 0,
12970                bytes: &payload,
12971            }],
12972        )
12973        .expect("a kind this build does not know still writes");
12974        let key_map = a_key_map_payload();
12975        attach(
12976            &path,
12977            "items",
12978            &[section::Attachment {
12979                kind: *section::KEY_MAP,
12980                id: 0,
12981                flags: 0,
12982                header_bytes: 0,
12983                bytes: &key_map,
12984            }],
12985        )
12986        .expect("attach beside it");
12987
12988        let reader = Reader::open(&path).expect("reopen");
12989        let held = attached(reader.table());
12990        assert_eq!(held.len(), 2, "the unfamiliar entry survived a directory rewrite");
12991        let unknown = held.iter().find(|one| !one.known()).expect("the unfamiliar one");
12992        assert_eq!(reader.payload(unknown).expect("its bytes are still there"), payload);
12993
12994        fs::remove_file(&path).expect("clean up");
12995    }
12996
12997    #[test]
12998    fn a_payload_of_nothing_is_a_relationship_recorded_as_not_built() {
12999        let path = linked_file("attach_not_built", 8);
13000        attach(
13001            &path,
13002            "items",
13003            &[section::Attachment {
13004                kind: *section::FORWARD_LINK,
13005                id: 2,
13006                flags: 0,
13007                header_bytes: 0,
13008                bytes: &[],
13009            }],
13010        )
13011        .expect("record a link that did not fit the budget");
13012
13013        let reader = Reader::open(&path).expect("reopen");
13014        let held = attached(reader.table());
13015        assert_eq!(held.len(), 1);
13016        assert_eq!(held[0].extents, 0);
13017        assert_eq!(held[0].extent_page, 0, "an entry that names no bytes points at none");
13018        assert!(reader.extents(held[0]).expect("no extent table").is_empty());
13019        assert!(reader.payload(held[0]).expect("no payload").is_empty());
13020
13021        fs::remove_file(&path).expect("clean up");
13022    }
13023
13024    #[test]
13025    fn a_payload_past_one_extent_is_split_and_joined_back() {
13026        // Issue #745's rule, exercised rather than argued. One byte past the bound is the smallest
13027        // payload that has to be two extents, and it is the case a split written for the common
13028        // size gets wrong.
13029        let path = linked_file("attach_two_extents", 8);
13030        let payload = vec![0x5a_u8; section::MAX_EXTENT as usize + 1];
13031        attach(
13032            &path,
13033            "items",
13034            &[section::Attachment {
13035                kind: *section::KEY_MAP,
13036                id: 0,
13037                flags: 0,
13038                header_bytes: 0,
13039                bytes: &payload,
13040            }],
13041        )
13042        .expect("attach a payload past the bound");
13043
13044        let reader = Reader::open(&path).expect("reopen");
13045        let held = attached(reader.table());
13046        let extents = reader.extents(held[0]).expect("extent table");
13047        assert_eq!(extents.len(), 2, "one byte past the bound is two extents");
13048        assert_eq!(extents[0].length, section::MAX_EXTENT);
13049        assert_eq!(extents[1].length, 1);
13050        assert_eq!(extents[1].first, u64::from(section::MAX_EXTENT));
13051        // And the extent the caller wants is readable on its own, which is the point of the split.
13052        assert_eq!(reader.extent(&extents[1]).expect("the last extent"), vec![0x5a]);
13053        assert_eq!(reader.payload(held[0]).expect("the whole payload").len(), payload.len());
13054
13055        fs::remove_file(&path).expect("clean up");
13056    }
13057
13058    #[test]
13059    fn a_torn_extent_is_refused_rather_than_decoded() {
13060        let path = linked_file("attach_torn", 8);
13061        let payload = a_key_map_payload();
13062        attach(
13063            &path,
13064            "items",
13065            &[section::Attachment {
13066                kind: *section::KEY_MAP,
13067                id: 0,
13068                flags: 0,
13069                header_bytes: 0,
13070                bytes: &payload,
13071            }],
13072        )
13073        .expect("attach");
13074
13075        let reader = Reader::open(&path).expect("reopen");
13076        let extent = reader.extents(&reader.table().sections()[0]).expect("extent table")[0];
13077        let file = OpenOptions::new().write(true).open(&path).expect("reopen to corrupt");
13078        write_at(&file, extent.offset + 7, &[0xff]).expect("flip a byte of the payload");
13079        drop(file);
13080
13081        let reader = Reader::open(&path).expect("the table still opens");
13082        let error = reader
13083            .payload(&reader.table().sections()[0])
13084            .expect_err("a corrupt payload is not handed out");
13085        assert!(error.to_string().contains("checksum"), "{error}");
13086        // And the table is still readable, which is section 3.1: a section that cannot be trusted
13087        // costs the query its shortcut and nothing else.
13088        assert_eq!(reader.read(0, &[0]).expect("the column is untouched").width(), 1);
13089
13090        fs::remove_file(&path).expect("clean up");
13091    }
13092
13093    #[test]
13094    fn attaching_to_a_file_of_the_previous_format_is_refused_rather_than_done() {
13095        // Readable is not writable. A format 22 directory has no section block, and adding one
13096        // without moving the number in the header would leave a file claiming a format it is not.
13097        let path = linked_file("attach_old_format", 8);
13098        let file = OpenOptions::new().write(true).open(&path).expect("reopen to patch");
13099        write_at(&file, 8, &22_u32.to_le_bytes()).expect("stamp the older format");
13100        drop(file);
13101
13102        let payload = a_key_map_payload();
13103        let error = attach(
13104            &path,
13105            "items",
13106            &[section::Attachment {
13107                kind: *section::KEY_MAP,
13108                id: 0,
13109                flags: 0,
13110                header_bytes: 0,
13111                bytes: &payload,
13112            }],
13113        )
13114        .expect_err("format 22 cannot gain a section");
13115        assert!(error.to_string().contains("format 22"), "{error}");
13116        assert!(Reader::open(&path).expect("and the file is untouched").table().rows() == 8);
13117
13118        fs::remove_file(&path).expect("clean up");
13119    }
13120
13121    #[test]
13122    fn a_section_header_longer_than_its_payload_is_refused_at_the_write() {
13123        let path = linked_file("attach_bad_header", 8);
13124        let error = attach(
13125            &path,
13126            "items",
13127            &[section::Attachment {
13128                kind: *section::KEY_MAP,
13129                id: 0,
13130                flags: 0,
13131                header_bytes: 40,
13132                bytes: &[1, 2, 3],
13133            }],
13134        )
13135        .expect_err("a writer's bug stops at the write");
13136        assert!(error.to_string().contains("header is longer"), "{error}");
13137
13138        fs::remove_file(&path).expect("clean up");
13139    }
13140
13141    #[test]
13142    fn attaching_to_a_name_the_file_does_not_hold_says_so() {
13143        let path = linked_file("attach_wrong_name", 8);
13144        let error = attach(&path, "orders", &[]).expect_err("no such table");
13145        assert!(error.to_string().contains("orders"), "{error}");
13146        fs::remove_file(&path).expect("clean up");
13147    }
13148
13149    #[test]
13150    fn the_planner_gets_an_exact_count_for_a_leading_value_of_an_incomplete_synopsis() {
13151        // The case a complete synopsis does not cover, and the one worth the most. 16,000 rows over
13152        // 601 distinct values, 10,000 of them holding a single value and the rest spread ten apiece
13153        // over six hundred more. The writer holds 512 values, so the list is a prefix and most of
13154        // the tail is outside it. The counts inside it are still exact, because the pass recounts
13155        // the candidates that survived it, so `id = 1` is ten thousand rows rather than the
13156        // twenty six a distinct count of 601 would divide its way to.
13157        let path = path("frequency_prefix_for_the_planner");
13158        let mut writer =
13159            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
13160                .expect("new file");
13161        let mut values = vec![Value::Integer(1); 10_000];
13162        for _ in 0..10 {
13163            values.extend((0..600).map(|tail| Value::Integer(1_000 + tail)));
13164        }
13165        // A vector holds 8,192 rows, so this goes in as several parts. The pass that takes the
13166        // synopsis walks the whole column rather than a part, so the counts are the same either way.
13167        for part in values.chunks(8_000) {
13168            let rows = Chunk::new(vec![
13169                Vector::from_values(LogicalType::Integer, part).expect("integers"),
13170            ])
13171            .expect("one column");
13172            writer.append(&rows).expect("a part");
13173        }
13174        writer.finish().expect("commit");
13175        let reader = Reader::open(&path).expect("reopen from disk");
13176        let prefix =
13177            reader.frequency_prefix(0).expect("a readable synopsis").expect("the column has one");
13178        // A prefix and not the whole column, and the writer said how many rows anything left out of
13179        // it can hold.
13180        assert_eq!(prefix.entries.len(), 512);
13181        assert_eq!(prefix.omitted_max, 10);
13182        let common = Common::new(reader);
13183        assert_eq!(common.rows(), 16_000);
13184        let column = common.column("id").expect("the file has that column");
13185        assert_eq!(
13186            common.rows_with(column, &Bound::Int(1)),
13187            Stat::exact(10_000, Provenance::FrequencySynopsis)
13188        );
13189        // In the prefix, because ties go to the smaller value and the prefix reaches 1,510.
13190        assert_eq!(
13191            common.rows_with(column, &Bound::Int(1_100)),
13192            Stat::exact(10, Provenance::FrequencySynopsis)
13193        );
13194        // Outside it, and a prefix says nothing about a value it does not list. Not zero, which is
13195        // what a complete list would say, and the file holds ten rows of this one.
13196        assert_eq!(common.rows_with(column, &Bound::Int(1_550)), Stat::Unknown);
13197        // Not in the file at all, and still nothing rather than a zero. A prefix cannot tell the
13198        // two apart, which is the whole of what it gives up.
13199        assert_eq!(common.rows_with(column, &Bound::Int(9_999)), Stat::Unknown);
13200        // What the prefix left out, which is what turns the unknown above into a number. The 512
13201        // entries account for 15,110 rows, so 890 are left for the 89 values the writer dropped,
13202        // and 890 over 89 is the ten rows each of them really holds.
13203        let remainder = common.remainder(column).expect("the list is a prefix");
13204        assert_eq!(remainder, Remainder { rows: 890, listed: 512, most: 10 });
13205        assert_eq!(remainder.rows / (601 - remainder.listed), 10);
13206        fs::remove_file(&path).expect("clean up");
13207    }
13208
13209    /// A file with no table in it is a file, and opening it says so rather than failing.
13210    #[test]
13211    fn a_file_holding_no_table_commits_and_opens_and_a_table_can_be_added_to_it() {
13212        let path = path("empty");
13213        Writer::empty(&path, &[]).expect("a file with nothing in it");
13214        let catalog = Catalog::open(&path).expect("the empty file opens");
13215        assert_eq!(catalog.len(), 0);
13216        assert!(catalog.is_empty());
13217        assert_eq!(catalog.names().count(), 0);
13218        // The next generation goes over the top of it the way it goes over any other, which is what
13219        // says this is a committed file and not a special case somebody has to know about.
13220        let mut writer =
13221            Writer::open(&path, "items", vec![Field::required("id", LogicalType::Integer)])
13222                .expect("a table goes into the empty file");
13223        writer.append(&sample_ids()).expect("rows");
13224        writer.finish().expect("commit");
13225        let catalog = Catalog::open(&path).expect("the file opens again");
13226        assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
13227        fs::remove_file(&path).expect("clean up");
13228    }
13229
13230    /// A committed table with no rows is a name the next generation takes over, and one with rows
13231    /// is a name it refuses.
13232    ///
13233    /// The refusal is what it always was and it is load bearing: carrying a table that holds rows
13234    /// forward means reading and rewriting its pages, and a writer that quietly wrote a second
13235    /// entry under the same name would leave a file with two tables a reader cannot tell apart. An
13236    /// empty one has no pages and no reader, so there is nothing to carry and nothing to lose, and
13237    /// taking its place is what lets a schema committed by an earlier session be loaded by a stream
13238    /// instead of through memory.
13239    #[test]
13240    fn a_committed_empty_table_gives_up_its_name_and_one_with_rows_does_not() {
13241        let path = path("empty-name");
13242        let field = || vec![Field::required("id", LogicalType::Integer)];
13243        Writer::create(&path, "items", field()).expect("new file").finish().expect("commit");
13244        let catalog = Catalog::open(&path).expect("the file opens");
13245        assert_eq!(catalog.rows().collect::<Vec<_>>(), vec![("items", 0)]);
13246
13247        let mut writer = Writer::open(&path, "items", field()).expect("the empty name is free");
13248        writer.append(&sample_ids()).expect("rows");
13249        writer.finish().expect("commit");
13250        let catalog = Catalog::open(&path).expect("the file opens again");
13251        // One entry and not two. The generation replaced the empty table rather than joining it.
13252        assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
13253        let held = catalog.rows().collect::<Vec<_>>();
13254        assert_eq!(held.len(), 1);
13255        assert!(held[0].1 > 0, "the rows that were appended are the ones the catalog counts");
13256
13257        // The same call against the same name now that it holds rows, which is still refused.
13258        let error = Writer::open(&path, "items", field()).expect_err("a name with rows is taken");
13259        assert!(error.to_string().contains("same name"), "{error}");
13260        fs::remove_file(&path).expect("clean up");
13261    }
13262
13263    /// A view, with everything about it that a reopened catalog has to be able to answer from.
13264    fn sample_view(name: &str) -> ViewEntry {
13265        ViewEntry {
13266            name: name.to_string(),
13267            sql: "SELECT id FROM items WHERE id > 0".to_string(),
13268            statement: format!("CREATE VIEW {name} AS SELECT id FROM items WHERE (id > 0);"),
13269            aliases: vec!["n".to_string()],
13270            columns: vec![Field::new("n", LogicalType::Integer)],
13271        }
13272    }
13273
13274    #[test]
13275    fn a_view_written_into_the_catalog_comes_back_whole() {
13276        let path = path("views");
13277        let mut writer =
13278            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
13279                .expect("new file");
13280        writer.append(&sample_ids()).expect("rows");
13281        writer.with_views(vec![sample_view("v")]).finish().expect("commit");
13282        let catalog = Catalog::open(&path).expect("reopen");
13283        assert_eq!(catalog.views().cloned().collect::<Vec<_>>(), vec![sample_view("v")]);
13284        // The tables are still there and are still read the same way, so the section on the end did
13285        // not move anything in front of it.
13286        assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
13287        fs::remove_file(&path).expect("clean up");
13288    }
13289
13290    /// A writer opened to append a table says nothing about views and must not lose them.
13291    #[test]
13292    fn appending_a_table_carries_the_views_forward() {
13293        let path = path("viewscarry");
13294        let mut writer =
13295            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
13296                .expect("new file");
13297        writer.append(&sample_ids()).expect("rows");
13298        writer.with_views(vec![sample_view("v")]).finish().expect("commit");
13299        let mut writer =
13300            Writer::open(&path, "other", vec![Field::required("id", LogicalType::Integer)])
13301                .expect("a second table");
13302        writer.append(&sample_ids()).expect("rows");
13303        writer.finish().expect("commit");
13304        let catalog = Catalog::open(&path).expect("reopen");
13305        assert_eq!(catalog.views().count(), 1);
13306        assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items", "other"]);
13307        fs::remove_file(&path).expect("clean up");
13308    }
13309
13310    /// The whole point of [`Writer::restate`]: the views change and the pages do not move.
13311    #[test]
13312    fn restating_the_views_leaves_every_table_where_it_was() {
13313        let path = path("restate");
13314        let mut writer =
13315            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
13316                .expect("new file");
13317        writer.append(&sample_ids()).expect("rows");
13318        writer.finish().expect("commit");
13319        let before = fs::metadata(&path).expect("the file is there").len();
13320        Writer::restate(&path, &[sample_view("v"), sample_view("w")]).expect("two views");
13321        let catalog = Catalog::open(&path).expect("reopen");
13322        assert_eq!(catalog.views().count(), 2);
13323        assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
13324        // A catalog on the end and nothing else, so what it grew by is the size of a catalog rather
13325        // than the size of the table.
13326        let after = fs::metadata(&path).expect("the file is there").len();
13327        assert!(after > before, "a generation was written");
13328        assert!(after - before < before, "the table was not written again");
13329        // The rows are still readable through the new generation, which is the part that would go
13330        // wrong if the catalog carried the wrong directory pointers forward.
13331        let reader = Catalog::open(&path).expect("reopen").table("items").expect("the table");
13332        assert_eq!(reader.table().rows, 3);
13333        // And a restate over a restate keeps working, because each one reads the slot that
13334        // checksummed rather than the highest number in the header.
13335        Writer::restate(&path, &[]).expect("no views at all");
13336        assert_eq!(Catalog::open(&path).expect("reopen").views().count(), 0);
13337        fs::remove_file(&path).expect("clean up");
13338    }
13339
13340    /// Two entries under one name is a catalog no lookup can answer, whichever two they are.
13341    #[test]
13342    fn a_view_named_after_a_table_is_refused_when_the_catalog_is_read() {
13343        let bytes = encode_catalog(
13344            &[Entry {
13345                name: "items".to_string(),
13346                fields: vec![Field::required("id", LogicalType::Integer)],
13347                rows: 1,
13348                directory: Page { offset: HEADER, length: 8, hash: 0 },
13349                nonzero: vec![None],
13350                aggregates: vec![None],
13351                distincts: vec![None],
13352                extremes: vec![None],
13353                frequencies: vec![None],
13354            }],
13355            &[sample_view("items")],
13356        )
13357        .expect("it encodes, because encoding does not look");
13358        let error = decode_catalog(&bytes, HEADER + 8).expect_err("and decoding does");
13359        assert!(error.to_string().contains("same name"), "{error}");
13360    }
13361
13362    /// A compressed text page read at some rows is those rows of the page read whole, nulls and
13363    /// all, and a row past the end or rows out of order are refused rather than guessed at.
13364    #[test]
13365    fn a_compressed_text_page_read_at_some_rows_is_those_rows_of_the_whole() {
13366        let rows: usize = 300;
13367        let text: Vec<String> =
13368            (0..rows).map(|row| format!("a street named after number {}", row * 7)).collect();
13369        let values: Vec<&[u8]> = text.iter().map(String::as_bytes).collect();
13370        let mut page = vec![6, 2];
13371        page.extend((0..rows.div_ceil(8)).map(|byte| {
13372            (0..8).filter(|bit| (byte * 8 + bit) % 5 != 3).fold(0_u8, |mask, bit| mask | 1 << bit)
13373        }));
13374        let compressed = string::encode_only(string::Kind::Fsst, &values)
13375            .expect("encoded")
13376            .expect("text this repetitive compresses");
13377        page.extend_from_slice(&compressed);
13378        let whole = decode(&LogicalType::Varchar, rows, &page, None).expect("the whole page");
13379        let positions = [0_u32, 3, 8, 13, 200, 299];
13380        let some =
13381            decode_at(&LogicalType::Varchar, rows, &page, None, &positions).expect("some rows");
13382        assert_eq!(some.len(), positions.len());
13383        for (at, &row) in positions.iter().enumerate() {
13384            assert_eq!(some.value_at(at), whole.value_at(row as usize), "row {row}");
13385        }
13386        assert_eq!(some.value_at(1), Value::Null, "row 3 is null");
13387        assert!(decode_at(&LogicalType::Varchar, rows, &page, None, &[300]).is_err());
13388        assert!(decode_at(&LogicalType::Varchar, rows, &page, None, &[8, 3]).is_err());
13389    }
13390
13391    /// Every column of a part read at some rows is the part read whole and gathered, whatever the
13392    /// page holds.
13393    #[test]
13394    fn a_part_read_at_some_rows_is_the_part_read_whole_and_gathered() {
13395        let path = path("rows");
13396        let mut writer = Writer::create(
13397            &path,
13398            "items",
13399            vec![
13400                Field::required("id", LogicalType::Integer),
13401                Field::new("text", LogicalType::Varchar),
13402            ],
13403        )
13404        .expect("new file");
13405        let rows = 2_000;
13406        let chunk = Chunk::new(vec![
13407            Vector::from_values(
13408                LogicalType::Integer,
13409                &(0..rows).map(Value::Integer).collect::<Vec<_>>(),
13410            )
13411            .expect("integers"),
13412            Vector::from_values(
13413                LogicalType::Varchar,
13414                &(0..rows)
13415                    .map(|row| {
13416                        if row % 7 == 2 {
13417                            Value::Null
13418                        } else {
13419                            Value::Varchar(format!("a comment about order {}", row * 13))
13420                        }
13421                    })
13422                    .collect::<Vec<_>>(),
13423            )
13424            .expect("strings"),
13425        ])
13426        .expect("matching rows");
13427        writer.append(&chunk).expect("one part");
13428        writer.finish().expect("commit");
13429        let reader = Reader::open(&path).expect("reopen from disk");
13430        let positions = [1_u32, 2, 9, 1_000, 1_999];
13431        for whole in [true, false] {
13432            let some = reader.read_rows(0, &[0, 1], &positions, whole).expect("some rows");
13433            let all = reader.read(0, &[0, 1]).expect("the whole part");
13434            assert_eq!(some.len(), positions.len());
13435            for column in 0..2 {
13436                for (at, &row) in positions.iter().enumerate() {
13437                    assert_eq!(some.value_at(at, column), all.value_at(row as usize, column));
13438                }
13439            }
13440        }
13441        assert!(reader.read_rows(0, &[1], &[2_000], true).is_err());
13442    }
13443
13444    #[test]
13445    fn committed_file_reopens_and_reads_only_requested_columns() {
13446        let path = path("reopen");
13447        let mut writer = Writer::create(
13448            &path,
13449            "items",
13450            vec![
13451                Field::required("id", LogicalType::Integer),
13452                Field::new("text", LogicalType::Varchar),
13453            ],
13454        )
13455        .expect("new file");
13456        writer.append(&sample()).expect("first part");
13457        writer.append(&sample()).expect("second part");
13458        writer.finish().expect("commit");
13459        let reader = Reader::open(&path).expect("reopen from disk");
13460        assert_eq!(reader.table().rows(), 6);
13461        // Two appends below the stripe bound are two parts of one stripe, which is the whole point
13462        // of the split: the directory describes the stripe and the scan still reads a part.
13463        assert_eq!(reader.table().stripes().len(), 1);
13464        assert_eq!(reader.parts(), 2);
13465        assert_eq!(reader.part_rows(0), 3);
13466        assert_eq!(reader.part_rows(1), 3);
13467        let text = reader.read(1, &[1]).expect("only text page");
13468        assert_eq!(text.width(), 1);
13469        assert_eq!(text.value_at(1, 0), Value::Null);
13470        assert_eq!(text.value_at(2, 0), Value::Varchar("long text after a slash".into()));
13471        let sparse = reader.read_sparse(1, &[1]).expect("one part without its whole page");
13472        assert_eq!(sparse.width(), 1);
13473        assert_eq!(sparse.value_at(1, 0), Value::Null);
13474        assert_eq!(sparse.value_at(2, 0), Value::Varchar("long text after a slash".into()));
13475        assert!(!reader.skips_codes(0, 1, &[0]).expect("alpha is in the stripe"));
13476        assert!(!reader.skips_codes(0, 1, &[2]).expect("long text is in the stripe"));
13477        assert!(reader.skips_codes(0, 1, &[3]).expect("unknown code is absent"));
13478        let count = reader.read(0, &[]).expect("no page is needed for count");
13479        assert_eq!(count.len(), 3);
13480        assert!(reader.skips(0, &[Probe { column: 0, op: Op::Greater, value: Bound::Int(100) }]));
13481        assert!(!reader.skips(0, &[Probe { column: 0, op: Op::Greater, value: Bound::Int(0) }]));
13482        let integers = reader.top_frequencies(0, 1).expect("valid integer synopsis").expect("kept");
13483        assert_eq!(
13484            integers,
13485            vec![(Value::Integer(-2), 2), (Value::Integer(4), 2), (Value::Integer(9), 2),]
13486        );
13487        let strings = reader.top_frequencies(1, 1).expect("valid string synopsis").expect("kept");
13488        assert_eq!(strings.len(), 3);
13489        assert!(strings.contains(&(Value::Null, 2)));
13490        assert!(strings.contains(&(Value::Varchar("alpha".into()), 2)));
13491        assert!(strings.contains(&(Value::Varchar("long text after a slash".into()), 2)));
13492        fs::remove_file(path).expect("remove scratch file");
13493    }
13494
13495    /// Two pipeline instances handing over whole runs, which is what makes the native sink safe to
13496    /// instance.
13497    ///
13498    /// The runs arrive in the order the instances finished reading them rather than in source
13499    /// order, and the second one to finish is the one that read the earlier rows. Each run is still
13500    /// a stripe of its own and the table still reads back in source order, which is the whole of
13501    /// what the writer promises about ordering.
13502    #[test]
13503    fn runs_handed_over_out_of_order_still_read_back_in_source_order() {
13504        let path = path("interleaved-runs");
13505        let mut writer =
13506            Writer::create(&path, "interleaved", vec![Field::new("v", LogicalType::BigInt)])
13507                .expect("new file");
13508        for morsel in [2_u64, 0, 3, 1] {
13509            let parts = (0..4_u64)
13510                .map(|chunk| {
13511                    let first = i64::try_from(morsel * 32 + chunk * 8).expect("small");
13512                    let values =
13513                        (0..8_i64).map(|row| Value::BigInt(first + row)).collect::<Vec<_>>();
13514                    let column =
13515                        Vector::from_values(LogicalType::BigInt, &values).expect("a column");
13516                    ((morsel, chunk), Chunk::new(vec![column]).expect("one column"))
13517                })
13518                .collect::<Vec<_>>();
13519            writer.append_stripe(parts).expect("a stripe");
13520        }
13521        writer.finish().expect("commit");
13522
13523        let reader = Reader::open(&path).expect("valid directory");
13524        assert_eq!(reader.table().stripes().len(), 4, "a run is a stripe of its own");
13525        assert_eq!(reader.table().rows(), 128);
13526        for part in 0..16_usize {
13527            let read = reader.read(part, &[0]).expect("a part back");
13528            for row in 0..8_usize {
13529                let want = i64::try_from(part * 8 + row).expect("small");
13530                assert_eq!(read.value_at(row, 0), Value::BigInt(want), "part {part} row {row}");
13531            }
13532        }
13533        fs::remove_file(path).expect("remove scratch file");
13534    }
13535
13536    /// Runs from different callers may interleave and may not overlap, and the commit is what
13537    /// catches an overlap.
13538    #[test]
13539    fn runs_that_overlap_each_other_are_refused_at_commit() {
13540        let path = path("overlapping-runs");
13541        let mut writer =
13542            Writer::create(&path, "overlapping", vec![Field::new("v", LogicalType::BigInt)])
13543                .expect("new file");
13544        let one = |order: (u64, u64)| {
13545            let column =
13546                Vector::from_values(LogicalType::BigInt, &[Value::BigInt(1)]).expect("a column");
13547            (order, Chunk::new(vec![column]).expect("one column"))
13548        };
13549        // The second run sits inside the first rather than after it, which is a thing no instance
13550        // holding its own contiguous run can produce and a thing the file cannot represent.
13551        writer.append_stripe(vec![one((0, 0)), one((0, 2))]).expect("a stripe");
13552        writer.append_stripe(vec![one((0, 1))]).expect("a stripe");
13553        let error = writer.finish().expect_err("the runs overlap");
13554        assert!(error.message().contains("source order"), "{error}");
13555        fs::remove_file(path).expect("remove scratch file");
13556    }
13557
13558    /// A stripe holds [`STRIPE_PARTS`] parts, so a run longer than that is a caller bug rather than
13559    /// something to split, and the writer says so at the door instead of quietly cutting it in two.
13560    #[test]
13561    fn a_run_longer_than_a_stripe_is_refused() {
13562        let path = path("overlong-run");
13563        let mut writer =
13564            Writer::create(&path, "overlong", vec![Field::new("v", LogicalType::BigInt)])
13565                .expect("new file");
13566        let parts = (0..=STRIPE_PARTS)
13567            .map(|at| {
13568                let column = Vector::from_values(LogicalType::BigInt, &[Value::BigInt(1)])
13569                    .expect("a column");
13570                let chunk = Chunk::new(vec![column]).expect("one column");
13571                ((0, u64::try_from(at).expect("small")), chunk)
13572            })
13573            .collect::<Vec<_>>();
13574        let error = writer.append_stripe(parts).expect_err("one part too many");
13575        assert!(error.message().contains("more parts than it holds"), "{error}");
13576        fs::remove_file(path).expect("remove scratch file");
13577    }
13578
13579    /// Parts past the stripe bound start a new stripe, and every part stays addressable on its own.
13580    ///
13581    /// This is the shape the format exists for, so both ends of the split are checked here. The
13582    /// directory holds three stripes rather than a hundred and thirty one, and a read of any one
13583    /// part still answers with that part's rows rather than with its whole stripe's.
13584    #[test]
13585    fn parts_past_the_stripe_bound_start_a_new_stripe() {
13586        let path = path("stripe-bound");
13587        let mut writer = Writer::create(
13588            &path,
13589            "items",
13590            vec![
13591                Field::required("id", LogicalType::Integer),
13592                Field::new("text", LogicalType::Varchar),
13593            ],
13594        )
13595        .expect("new file");
13596        let parts = STRIPE_PARTS * 2 + 3;
13597        for part in 0..parts {
13598            let id = part as i32;
13599            let chunk = Chunk::new(vec![
13600                Vector::from_values(
13601                    LogicalType::Integer,
13602                    &[Value::Integer(id), Value::Integer(-id)],
13603                )
13604                .expect("integers"),
13605                Vector::from_values(
13606                    LogicalType::Varchar,
13607                    &[Value::Varchar(format!("value {part}")), Value::Null],
13608                )
13609                .expect("strings"),
13610            ])
13611            .expect("matching rows");
13612            writer.append(&chunk).expect("one part");
13613        }
13614        writer.finish().expect("commit");
13615
13616        let reader = Reader::open(&path).expect("reopen from disk");
13617        assert_eq!(reader.parts(), parts);
13618        assert_eq!(reader.table().rows(), parts * 2);
13619        assert_eq!(reader.table().stripes().len(), parts.div_ceil(STRIPE_PARTS));
13620        assert_eq!(reader.table().stripes()[0].parts(), STRIPE_PARTS);
13621        assert_eq!(reader.table().stripes()[0].rows(), STRIPE_PARTS * 2);
13622        assert_eq!(reader.table().stripes()[2].parts(), 3);
13623        // Backwards on purpose. The reader keeps four stripes a column, so a scan that walks the
13624        // table the other way is what catches a cache that only ever holds what it just read.
13625        for part in (0..parts).rev() {
13626            let dense = reader.read(part, &[0, 1]).expect("a whole page read");
13627            let sparse = reader.read_sparse(part, &[0, 1]).expect("one part read");
13628            for chunk in [&dense, &sparse] {
13629                assert_eq!(chunk.len(), 2, "part {part} has its own row count");
13630                assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
13631                assert_eq!(chunk.value_at(1, 0), Value::Integer(-(part as i32)));
13632                assert_eq!(chunk.value_at(0, 1), Value::Varchar(format!("value {part}")));
13633                assert_eq!(chunk.value_at(1, 1), Value::Null);
13634            }
13635        }
13636        // The bounds are merged over the stripe, so they answer for the range the whole stripe
13637        // covers and not for the part that was asked about.
13638        let above = [Probe { column: 0, op: Op::Greater, value: Bound::Int(100) }];
13639        assert!(reader.skips(0, &above), "the first stripe stops at 63");
13640        assert!(!reader.skips(STRIPE_PARTS * 2, &above), "the third stripe reaches 130");
13641        fs::remove_file(path).expect("remove scratch file");
13642    }
13643
13644    /// A scattered value in the column that decides `WHERE UserID = ?`.
13645    fn scattered(n: i64) -> i64 {
13646        n.wrapping_mul(-7_046_029_254_386_353_131)
13647    }
13648
13649    /// A part whose sieve does not hold the constant is skipped, and a range would skip none of them.
13650    ///
13651    /// This is ClickBench query 19 in miniature. The values are spread over the whole of `BIGINT`, so
13652    /// every stripe's bounds cover nearly all of it and rule out nothing, and the part that really
13653    /// holds the value is the only one a scan has to read.
13654    #[test]
13655    fn a_part_is_skipped_when_its_sieve_does_not_hold_the_constant() {
13656        let path = path("sieve-skip");
13657        let mut writer =
13658            Writer::create(&path, "hits", vec![Field::required("id", LogicalType::BigInt)])
13659                .expect("new file");
13660        let parts = STRIPE_PARTS + 3;
13661        // Big enough that the filter is worth its bytes. A part of eight numbers packs to under a
13662        // hundred bytes and the smallest filter there is is sixty nine, so a filter over a part
13663        // that small costs about as much to read as the rows do and is no longer written.
13664        let per_part = 128;
13665        for part in 0..parts {
13666            let held: Vec<Value> = (0..per_part)
13667                .map(|row| Value::BigInt(scattered((part * per_part + row) as i64)))
13668                .collect();
13669            let chunk =
13670                Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
13671                    .expect("one column");
13672            writer.append(&chunk).expect("one part");
13673        }
13674        writer.finish().expect("commit");
13675
13676        let reader = Reader::open(&path).expect("reopen from disk");
13677        let probe = |value: i64| Probe {
13678            column: 0,
13679            op: Op::Equal,
13680            value: Bound::Int(i128::from(scattered(value))),
13681        };
13682        for wanted in [0_i64, (per_part + 1) as i64, (parts * per_part - 1) as i64] {
13683            let tests = [probe(wanted)];
13684            let kept: Vec<usize> = (0..parts).filter(|&part| !reader.skips(part, &tests)).collect();
13685            let home = wanted as usize / per_part;
13686            assert!(kept.contains(&home), "the part holding {wanted} is read");
13687            // A filter answers maybe, so a part it keeps need not hold the value. Sixty seven parts
13688            // of a hundred and twenty eight numbers each, at a dozen bits a value, is about one
13689            // stray part across the whole file and that is what this leaves room for.
13690            assert!(kept.len() <= 2, "{wanted} keeps {kept:?}, which is more than one stray part");
13691        }
13692        let absent = [probe((parts * per_part) as i64 + 1)];
13693        let kept = (0..parts).filter(|&part| !reader.skips(part, &absent)).count();
13694        assert!(kept <= 1, "{kept} parts of {parts} kept a value no part holds");
13695        // The same probes against the bounds alone, which is what this replaces. A column of
13696        // scattered numbers has a range per stripe that covers nearly the whole type.
13697        let tests = [probe(0)];
13698        assert!(
13699            reader.table().stripes().iter().all(|stripe| !stripe.zone.skips(&tests)),
13700            "the bounds rule out no stripe at all"
13701        );
13702        fs::remove_file(path).expect("remove scratch file");
13703    }
13704
13705    /// A part whose own bounds rule out an ordered comparison is skipped where the stripe's keep it.
13706    ///
13707    /// This is the shape of ClickBench 24. Each part covers a narrow stretch of the column and the
13708    /// stripe covers all sixty four of them at once, so a comparison that lands inside the stripe
13709    /// rules out none of it and rules out all but a few parts.
13710    #[test]
13711    fn a_part_is_skipped_when_its_own_bounds_rule_out_a_comparison_the_stripe_keeps() {
13712        let path = path("part-range-skip");
13713        let mut writer =
13714            Writer::create(&path, "hits", vec![Field::required("at", LogicalType::BigInt)])
13715                .expect("new file");
13716        let parts = STRIPE_PARTS + 3;
13717        let per_part = 128;
13718        for part in 0..parts {
13719            // Scattered inside the part's own band rather than a run, because a run of
13720            // consecutive numbers encodes to a stride of a few bytes and then the page of ranges
13721            // costs more than reading the column it indexes, which is the case the writer declines.
13722            let held: Vec<Value> = (0..per_part)
13723                .map(|row| {
13724                    Value::BigInt((part * 1_000) as i64 + (scattered(row as i64).rem_euclid(900)))
13725                })
13726                .collect();
13727            let chunk =
13728                Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
13729                    .expect("one column");
13730            writer.append(&chunk).expect("one part");
13731        }
13732        writer.finish().expect("commit");
13733
13734        let reader = Reader::open(&path).expect("reopen from disk");
13735        let under = [Probe { column: 0, op: Op::Less, value: Bound::Int(3_000) }];
13736        let kept: Vec<usize> = (0..parts).filter(|&part| !reader.skips(part, &under)).collect();
13737        assert_eq!(kept, vec![0, 1, 2], "only the three parts that start under three thousand");
13738        // The same question asked of the stripe alone, which is what this replaces.
13739        assert!(!reader.stripe_skips(0, &under), "the stripe reaches from zero and keeps itself");
13740        fs::remove_file(path).expect("remove scratch file");
13741    }
13742
13743    /// The other half of the same page. A part whose own bounds put every row of it inside the
13744    /// filter is waved through, so the comparison never runs on it, where the stripe's bounds reach
13745    /// across every part and can prove nothing.
13746    #[test]
13747    fn a_part_is_waved_through_when_its_own_bounds_pass_a_comparison_the_stripe_cannot() {
13748        let path = path("part-range-certain");
13749        let mut writer =
13750            Writer::create(&path, "hits", vec![Field::required("at", LogicalType::BigInt)])
13751                .expect("new file");
13752        let parts = STRIPE_PARTS + 3;
13753        let per_part = 128;
13754        for part in 0..parts {
13755            let held: Vec<Value> = (0..per_part)
13756                .map(|row| {
13757                    Value::BigInt((part * 1_000) as i64 + (scattered(row as i64).rem_euclid(900)))
13758                })
13759                .collect();
13760            let chunk =
13761                Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
13762                    .expect("one column");
13763            writer.append(&chunk).expect("one part");
13764        }
13765        writer.finish().expect("commit");
13766
13767        let reader = Reader::open(&path).expect("reopen from disk");
13768        let under = [Probe { column: 0, op: Op::Less, value: Bound::Int(3_000) }];
13769        let waved: Vec<usize> = (0..parts).filter(|&part| reader.certain(part, &under)).collect();
13770        assert_eq!(waved, vec![0, 1, 2], "the three parts that end under three thousand");
13771        // The first stripe reaches from zero to past sixty thousand, so it straddles three thousand
13772        // and settles nothing either way. The three yeses above are the parts' own ends talking.
13773        assert!(!reader.stripe_skips(0, &under), "the stripe straddles the comparison");
13774        fs::remove_file(path).expect("remove scratch file");
13775    }
13776
13777    /// The page is worth its bytes on a column with parts to tell apart and is not written on one
13778    /// that has a single part, where the stripe bounds already are the part's.
13779    #[test]
13780    fn a_stripe_of_one_part_writes_no_range_page_and_a_stripe_of_many_does() {
13781        for (parts, wanted) in [(1_usize, false), (STRIPE_PARTS, true)] {
13782            let path = path("part-range-page");
13783            let mut writer =
13784                Writer::create(&path, "hits", vec![Field::required("at", LogicalType::BigInt)])
13785                    .expect("new file");
13786            for part in 0..parts {
13787                let held: Vec<Value> = (0..128)
13788                    .map(|row| {
13789                        Value::BigInt((part * 1_000) as i64 + scattered(row as i64).rem_euclid(900))
13790                    })
13791                    .collect();
13792                let chunk = Chunk::new(vec![
13793                    Vector::from_values(LogicalType::BigInt, &held).expect("numbers"),
13794                ])
13795                .expect("one column");
13796                writer.append(&chunk).expect("one part");
13797            }
13798            writer.finish().expect("commit");
13799            let reader = Reader::open(&path).expect("reopen from disk");
13800            let bytes = reader.layout().columns[0].part_ranges;
13801            assert_eq!(bytes > 0, wanted, "{parts} parts wrote {bytes} bytes of ranges");
13802            fs::remove_file(path).expect("remove scratch file");
13803        }
13804    }
13805
13806    /// A cut down string end is still an end on the side it was, which is the only thing that keeps
13807    /// a shortened bound from turning a skip into a wrong answer.
13808    #[test]
13809    fn a_string_end_that_is_cut_down_still_covers_the_value_it_came_from() {
13810        let long = vec![b'a'; PART_BOUND_BYTES * 2];
13811        let low = shortened(Some(Bound::Bytes(long.clone())), false).expect("a low end");
13812        let high = shortened(Some(Bound::Bytes(long.clone())), true).expect("a high end");
13813        let Bound::Bytes(low) = low else { panic!("a string stays a string") };
13814        let Bound::Bytes(high) = high else { panic!("a string stays a string") };
13815        assert!(low.len() <= PART_BOUND_BYTES && high.len() <= PART_BOUND_BYTES);
13816        assert!(low.as_slice() <= long.as_slice(), "the low end is at or under the value");
13817        assert!(high.as_slice() >= long.as_slice(), "the high end is at or over the value");
13818    }
13819
13820    /// A string of nothing but the largest byte has no prefix that can be stepped up, so the high
13821    /// end is given up rather than claimed too small. No end keeps the part, which is always safe.
13822    #[test]
13823    fn a_string_end_with_no_room_to_step_up_gives_up_the_bound() {
13824        let long = vec![u8::MAX; PART_BOUND_BYTES * 2];
13825        assert_eq!(shortened(Some(Bound::Bytes(long.clone())), true), None);
13826        let low = shortened(Some(Bound::Bytes(long)), false).expect("a low end is still a prefix");
13827        assert_eq!(low, Bound::Bytes(vec![u8::MAX; PART_BOUND_BYTES]));
13828    }
13829
13830    /// What a column is stored as, asked of two files holding the same rows in a different order.
13831    ///
13832    /// This is the question the report exists to answer and it is the one the directory cannot. The
13833    /// two files have the same rows, the same schema and the same number of parts, and the column
13834    /// comes out four times smaller in one of them, because ascending keys delta encode to a few
13835    /// bits a row and shuffled ones do not. Nothing about the file's shape says so. The page header
13836    /// says so, and reading it is what this does.
13837    ///
13838    /// It is q18 on TPC-H in miniature: clustering lineitem by ship date leaves `l_orderkey`
13839    /// ascending inside a partition but sparse, its deltas go from six bits to twelve, and the scan
13840    /// pays for the wider ones.
13841    #[test]
13842    fn what_a_column_is_stored_as_follows_the_order_the_rows_were_written_in() {
13843        let parts = 4;
13844        let per_part = 1024;
13845        let rows = parts * per_part;
13846        let written = |name: &str, keys: &[i64]| {
13847            let path = path(name);
13848            let fields = vec![Field::required("key", LogicalType::BigInt)];
13849            let mut writer = Writer::create(&path, "keys", fields).expect("new file");
13850            for part in 0..parts {
13851                let values: Vec<Value> = keys[part * per_part..(part + 1) * per_part]
13852                    .iter()
13853                    .map(|key| Value::BigInt(*key))
13854                    .collect();
13855                let chunk = Chunk::new(vec![
13856                    Vector::from_values(LogicalType::BigInt, &values).expect("numbers"),
13857                ])
13858                .expect("one column");
13859                writer.append(&chunk).expect("one part");
13860            }
13861            writer.finish().expect("commit");
13862            path
13863        };
13864        // Ascending with a small irregular step, which is what a key column in arrival order looks
13865        // like: an order has one to seven line items, so the key repeats and then moves on by one.
13866        let climbing = |step: &dyn Fn(usize) -> i64| {
13867            let mut key = 0;
13868            (0..rows)
13869                .map(|row| {
13870                    key += step(row);
13871                    key
13872                })
13873                .collect::<Vec<i64>>()
13874        };
13875        let ascending = climbing(&|row| (row % 3) as i64);
13876        // The same rows in the same direction over a range a thousand times wider, which is what a
13877        // partition of a clustered table holds: still ascending, and far enough apart that the
13878        // deltas no longer fit in a handful of bits.
13879        let sparse = climbing(&|row| ((row * 2_654_435_761) % 4096) as i64);
13880        let near_path = written("stored-near", &ascending);
13881        let far_path = written("stored-far", &sparse);
13882
13883        let one = Reader::open(&near_path).expect("reopen from disk");
13884        let other = Reader::open(&far_path).expect("reopen from disk");
13885        let near = one.stored(0).expect("the column is stored");
13886        let far = other.stored(0).expect("the column is stored");
13887        assert_eq!(near.len(), parts, "one row per part");
13888        assert_eq!(far.len(), parts);
13889        // The bytes are the same bytes the directory totals, which is the check that this is
13890        // reading the pages the file really holds rather than some other pages.
13891        let total = |stored: &[StoredPart]| stored.iter().map(|part| part.bytes).sum::<u64>();
13892        assert_eq!(total(&near), one.layout().columns[0].pages);
13893        assert_eq!(total(&far), other.layout().columns[0].pages);
13894        assert!(
13895            total(&near) * 2 < total(&far),
13896            "the sparse keys cost more, {} against {}",
13897            total(&far),
13898            total(&near)
13899        );
13900        // Every part accounted for, in order, with the row it starts at following the one before.
13901        for (at, part) in near.iter().enumerate() {
13902            assert_eq!(part.part, at);
13903            assert_eq!(part.row, at * per_part);
13904            assert_eq!(part.rows, per_part);
13905            let held = &ascending[at * per_part..(at + 1) * per_part];
13906            assert_eq!(part.low, Some(Value::BigInt(held[0])));
13907            assert_eq!(part.high, Some(Value::BigInt(held[per_part - 1])));
13908            assert_eq!(part.nulls, Some(0));
13909        }
13910        // And the encoding is a line of text that names what the encoder chose, which is the whole
13911        // point. Both are a cascade over deltas and the widths inside them are what differ.
13912        assert!(near[0].encoding.contains("DELTA"), "{}", near[0].encoding);
13913        assert!(far[0].encoding.contains("DELTA"), "{}", far[0].encoding);
13914        assert_ne!(near[0].encoding, far[0].encoding);
13915        fs::remove_file(near_path).expect("remove scratch file");
13916        fs::remove_file(far_path).expect("remove scratch file");
13917    }
13918
13919    /// A sieve bigger than the part it indexes is not written, and one smaller than it still is.
13920    ///
13921    /// Both columns hold values spread over the whole of `BIGINT`, so neither gets a bitmap and both
13922    /// reach the filter. They differ in what the part costs to read. `spread` is a thousand distinct
13923    /// numbers and packs to eight kilobytes, so a filter of about thirteen hundred bytes is a good
13924    /// trade. `repeated` is the same thousand rows over four numbers in runs and encodes to
13925    /// almost nothing, but the filter is sized for the rows rather than the values it turns out to
13926    /// hold, so it comes out larger than the data. Reading it to decide whether to read the part spends more than
13927    /// the part, every time, and that is the case this drops.
13928    #[test]
13929    fn a_sieve_larger_than_the_part_it_indexes_is_not_written() {
13930        let path = path("sieve-pays");
13931        let fields = vec![
13932            Field::required("spread", LogicalType::BigInt),
13933            Field::required("repeated", LogicalType::BigInt),
13934        ];
13935        let mut writer = Writer::create(&path, "hits", fields).expect("new file");
13936        let parts = 3;
13937        let per_part = 1024;
13938        for part in 0..parts {
13939            let base = (part * per_part) as i64;
13940            let spread: Vec<Value> =
13941                (0..per_part).map(|row| Value::BigInt(scattered(base + row as i64))).collect();
13942            let repeated: Vec<Value> =
13943                (0..per_part).map(|row| Value::BigInt(scattered((row / 256) as i64))).collect();
13944            let chunk = Chunk::new(vec![
13945                Vector::from_values(LogicalType::BigInt, &spread).expect("numbers"),
13946                Vector::from_values(LogicalType::BigInt, &repeated).expect("numbers"),
13947            ])
13948            .expect("two columns");
13949            writer.append(&chunk).expect("one part");
13950        }
13951        writer.finish().expect("commit");
13952
13953        let reader = Reader::open(&path).expect("reopen from disk");
13954        let layout = reader.layout();
13955        let spread = &layout.columns[0];
13956        let repeated = &layout.columns[1];
13957        assert!(spread.sieves > 0, "a column whose parts are worth a filter keeps one");
13958        assert_eq!(
13959            repeated.sieves, 0,
13960            "a column whose filter costs more than its parts keeps none"
13961        );
13962        // Per part this is the rule itself, so it holds over the column as well: a part without a
13963        // sieve adds to one side of this and to nothing on the other.
13964        for column in &layout.columns {
13965            assert!(
13966                column.sieves < column.pages,
13967                "{} spends {} on sieves over {} of data",
13968                column.name,
13969                column.sieves,
13970                column.pages
13971            );
13972        }
13973        // The filter that was kept still does what it is for.
13974        let absent = [Probe {
13975            column: 0,
13976            op: Op::Equal,
13977            value: Bound::Int(i128::from(scattered((parts * per_part) as i64 + 1))),
13978        }];
13979        assert!((0..parts).all(|part| reader.skips(part, &absent)), "no part holds it");
13980        fs::remove_file(path).expect("remove scratch file");
13981    }
13982
13983    /// A damaged sieve page is a part that gets read, not a query that fails.
13984    ///
13985    /// A sieve is an index over rows that are still there and still correct, so losing one costs
13986    /// time and costs no answers. That is the opposite of the membership index beside it, which is
13987    /// the only thing standing between a string page and a wrong answer.
13988    #[test]
13989    fn a_damaged_sieve_page_is_read_through_rather_than_refused() {
13990        let path = path("sieve-damaged");
13991        let mut writer =
13992            Writer::create(&path, "hits", vec![Field::required("id", LogicalType::BigInt)])
13993                .expect("new file");
13994        let rows = 128;
13995        let held: Vec<Value> = (0..rows).map(|row| Value::BigInt(scattered(row))).collect();
13996        let chunk =
13997            Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
13998                .expect("one column");
13999        writer.append(&chunk).expect("one part");
14000        writer.finish().expect("commit");
14001
14002        let page = Reader::open(&path).expect("reopen").table.stripes[0]
14003            .sieves
14004            .get(0)
14005            .expect("a sieve page");
14006        let mut file = OpenOptions::new().write(true).open(&path).expect("open the sieve page");
14007        file.seek(SeekFrom::Start(page.offset + u64::from(page.length) - 1)).expect("seek");
14008        file.write_all(&[0xff]).expect("damage one byte");
14009        drop(file);
14010
14011        let reader = Reader::open(&path).expect("reopen the damaged file");
14012        let absent =
14013            [Probe { column: 0, op: Op::Equal, value: Bound::Int(i128::from(scattered(99))) }];
14014        assert!(!reader.skips(0, &absent), "a sieve that cannot be read skips nothing");
14015        assert_eq!(
14016            reader.read(0, &[0]).expect("the rows are untouched").len(),
14017            usize::try_from(rows).expect("a small count")
14018        );
14019        fs::remove_file(path).expect("remove scratch file");
14020    }
14021
14022    /// Eight workers over one stripe read it once between them.
14023    ///
14024    /// This is the shape a scan actually has. Parts are handed out in order, so every worker on a
14025    /// column crosses into a stripe within a few parts of the others, and before [`Reader::held`]
14026    /// started sharing the read every one of them read the whole page. On the full ClickBench file
14027    /// that was a `MIN(EventDate), MAX(EventDate)` moving 3.2 GB off the disk to look at 400 MB of
14028    /// column, which is most of what a first touch costs.
14029    ///
14030    /// The workers that lose the race still answer, out of the part reads they do instead, which is
14031    /// what the values below are checking.
14032    #[test]
14033    fn workers_that_want_the_same_stripe_read_it_once() {
14034        let path = path("single-flight");
14035        let mut writer =
14036            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
14037                .expect("new file");
14038        for part in 0..STRIPE_PARTS {
14039            let id = part as i32;
14040            let chunk = Chunk::new(vec![
14041                Vector::from_values(
14042                    LogicalType::Integer,
14043                    &[Value::Integer(id), Value::Integer(-id)],
14044                )
14045                .expect("integers"),
14046            ])
14047            .expect("matching rows");
14048            writer.append(&chunk).expect("one part");
14049        }
14050        writer.finish().expect("commit");
14051
14052        let reader = Reader::open(&path).expect("reopen from disk");
14053        assert_eq!(reader.table().stripes().len(), 1, "one stripe is the point of the test");
14054        let barrier = std::sync::Barrier::new(8);
14055        std::thread::scope(|scope| {
14056            for worker in 0..8 {
14057                let reader = &reader;
14058                let barrier = &barrier;
14059                scope.spawn(move || {
14060                    barrier.wait();
14061                    for part in (worker..STRIPE_PARTS).step_by(8) {
14062                        let chunk = reader.read(part, &[0]).expect("a whole page read");
14063                        assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
14064                        assert_eq!(chunk.value_at(1, 0), Value::Integer(-(part as i32)));
14065                    }
14066                });
14067            }
14068        });
14069        assert_eq!(reader.pages.load(Atomic::Relaxed), 1, "one stripe, one page read, whoever won");
14070        fs::remove_file(path).expect("remove scratch file");
14071    }
14072
14073    /// Opening a file reads the header and the directory, and nothing that depends on the rows.
14074    ///
14075    /// `spec/stats/04-in-memory.md` section 4.2. There are no statistics in the file yet, so this
14076    /// holds today by not having anything to load, and that is exactly why it is worth pinning now.
14077    /// The change that breaks it is the reasonable looking one: summaries are a few hundred bytes,
14078    /// the next query will want them, so read them on the way past. A process that opened the
14079    /// database to run one trivial query pays for all of it and gets nothing.
14080    ///
14081    /// Two files of the same shape and a thousand times the rows in one of them, opened, and the
14082    /// two openings cost the same. The stripe count is held equal so that the directory is the same
14083    /// size in both, which leaves the rows as the only thing that changed. Anything read out of the
14084    /// data would show up here.
14085    #[test]
14086    fn opening_costs_the_same_over_a_thousand_times_the_rows() {
14087        let opened = |label: &str, rows_per_part: i32| {
14088            let path = path(label);
14089            let mut writer =
14090                Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
14091                    .expect("new file");
14092            for part in 0..STRIPE_PARTS * 3 {
14093                // Scrambled rather than sequential, so that the fat file is actually fatter. A run
14094                // of consecutive integers encodes to almost nothing and would leave the two files
14095                // the same size, which would make this test pass for the wrong reason.
14096                let values = (0..rows_per_part)
14097                    .map(|row| {
14098                        Value::Integer((part as i32 * rows_per_part + row).wrapping_mul(2_654_435))
14099                    })
14100                    .collect::<Vec<_>>();
14101                let chunk = Chunk::new(vec![
14102                    Vector::from_values(LogicalType::Integer, &values).expect("integers"),
14103                ])
14104                .expect("matching rows");
14105                writer.append(&chunk).expect("one part");
14106            }
14107            writer.finish().expect("commit");
14108            let reader = Reader::open(&path).expect("reopen from disk");
14109            let size = fs::metadata(&path).expect("the file is there").len();
14110            let out = (reader.reads(), reader.table().stripes().len(), size);
14111            fs::remove_file(path).expect("remove scratch file");
14112            out
14113        };
14114
14115        let (thin, thin_stripes, thin_size) = opened("open-thin", 1);
14116        let (fat, fat_stripes, fat_size) = opened("open-fat", 1000);
14117        assert_eq!(
14118            thin_stripes, fat_stripes,
14119            "the same stripe count is what makes this a fair ask"
14120        );
14121        assert!(
14122            fat_size > thin_size * 50,
14123            "the fat file has to actually be larger, and it is {fat_size} against {thin_size}"
14124        );
14125
14126        assert_eq!(thin.opening.reads, fat.opening.reads, "the same reads either way");
14127        assert_eq!(thin.pages, 0, "opening read a page");
14128        assert_eq!(fat.pages, 0, "opening read a page");
14129        assert_eq!(thin.indexes, 0, "opening read an index");
14130        assert_eq!(fat.indexes, 0, "opening read an index");
14131        // Not exactly equal, because a directory holds offsets and a larger file has larger ones,
14132        // and a handful of bytes of varint is not somebody loading statistics. A factor is.
14133        assert!(
14134            fat.opening.bytes < thin.opening.bytes * 2,
14135            "opening the thin file read {} bytes and the fat one read {}",
14136            thin.opening.bytes,
14137            fat.opening.bytes
14138        );
14139    }
14140
14141    /// The reads a file costs to open are fixed by its shape and not by what ran before.
14142    ///
14143    /// `spec/stats/04-in-memory.md` section 4.3, which is the rule that keeps a plan reproducible:
14144    /// the plan is a function of the data, the generation and the settings, and never of what
14145    /// happened to be in cache. Opening the same file twice in the same process has to cost the
14146    /// same, because a second open that read less would be an open that was about to plan
14147    /// differently.
14148    #[test]
14149    fn two_opens_of_one_file_cost_the_same_and_the_second_is_not_cheaper() {
14150        let path = path("open-twice");
14151        let mut writer =
14152            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
14153                .expect("new file");
14154        for part in 0..STRIPE_PARTS * 3 {
14155            let chunk = Chunk::new(vec![
14156                Vector::from_values(LogicalType::Integer, &[Value::Integer(part as i32)])
14157                    .expect("integers"),
14158            ])
14159            .expect("matching rows");
14160            writer.append(&chunk).expect("one part");
14161        }
14162        writer.finish().expect("commit");
14163
14164        let first = Reader::open(&path).expect("open");
14165        // A whole scan in between, so the operating system's page cache is as warm as it gets and
14166        // anything that consulted it would show up in the second open.
14167        for part in 0..first.parts() {
14168            first.read(part, &[0]).expect("a part");
14169        }
14170        assert!(first.reads().pages > 0, "the scan has to have read something");
14171        let second = Reader::open(&path).expect("open again");
14172
14173        assert_eq!(first.reads().opening, second.reads().opening);
14174        assert_eq!(
14175            second.reads().pages,
14176            0,
14177            "the second open read a page off the back of the first"
14178        );
14179        assert_eq!(second.reads().indexes, 0, "the second open read an index it inherited");
14180        fs::remove_file(path).expect("remove scratch file");
14181    }
14182
14183    /// A scan reads a stripe's index once for the whole scan, not once per part that misses.
14184    ///
14185    /// The page cache holds four stripes and an index used to ride inside it, so a table with more
14186    /// stripes than that read the index again every time a stripe came back around. The index is a
14187    /// few hundred bytes and the page is a quarter of a megabyte, which is why they are now under
14188    /// different budgets. This is the test that keeps them there, since the saving is small enough
14189    /// that nothing in a benchmark would notice it going away again.
14190    #[test]
14191    fn an_index_is_read_once_per_stripe_however_often_the_page_is_evicted() {
14192        let path = path("index-cache");
14193        let mut writer =
14194            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
14195                .expect("new file");
14196        let parts = STRIPE_PARTS * (CACHED_STRIPES_PER_COLUMN + 2);
14197        for part in 0..parts {
14198            let id = part as i32;
14199            let chunk = Chunk::new(vec![
14200                Vector::from_values(LogicalType::Integer, &[Value::Integer(id)]).expect("integers"),
14201            ])
14202            .expect("matching rows");
14203            writer.append(&chunk).expect("one part");
14204        }
14205        writer.finish().expect("commit");
14206
14207        let reader = Reader::open(&path).expect("reopen from disk");
14208        let stripes = reader.table().stripes().len();
14209        assert!(stripes > CACHED_STRIPES_PER_COLUMN, "the page cache has to be too small for this");
14210        // Twice over, so that the second pass finds every page evicted and every index kept.
14211        for _ in 0..2 {
14212            for part in 0..parts {
14213                let chunk = reader.read(part, &[0]).expect("a part");
14214                assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
14215            }
14216        }
14217        assert_eq!(reader.indexes.load(Atomic::Relaxed), stripes, "one index read per stripe");
14218        assert!(
14219            reader.pages.load(Atomic::Relaxed) > stripes,
14220            "the pages are the ones that get read again, which is what makes the index count mean \
14221             something"
14222        );
14223        fs::remove_file(path).expect("remove scratch file");
14224    }
14225
14226    /// A page stays in memory from one scan to the next while the pool has room for it, and a
14227    /// table that is being read takes room from one that is not, down to the floor and no further.
14228    ///
14229    /// This is what the pool is for. Each reader lives as long as its database, so a second query
14230    /// over the same table should find every page it read the first time, and before the pool it
14231    /// found four stripes a column and read the rest off the file again.
14232    #[test]
14233    fn a_pool_keeps_pages_between_scans_and_gives_them_to_the_table_being_read() {
14234        let path = path("page-pool");
14235        let parts = STRIPE_PARTS * (CACHED_STRIPES_PER_COLUMN + 2);
14236        let fields = || vec![Field::required("id", LogicalType::Integer)];
14237        let mut writer = Writer::create(&path, "a", fields()).expect("new file");
14238        for table in ["a", "b"] {
14239            if table == "b" {
14240                writer = writer.next("b".to_string(), fields()).expect("a second table");
14241            }
14242            for part in 0..parts {
14243                let chunk = Chunk::new(vec![
14244                    Vector::from_values(LogicalType::Integer, &[Value::Integer(part as i32)])
14245                        .expect("integers"),
14246                ])
14247                .expect("matching rows");
14248                writer.append(&chunk).expect("one part");
14249            }
14250        }
14251        writer.finish().expect("commit");
14252
14253        let pool = PagePool::new(usize::MAX);
14254        let catalog = Catalog::open_in(&path, &pool).expect("the file opens");
14255        let (a, b) = (catalog.table("a").expect("a"), catalog.table("b").expect("b"));
14256        let stripes = a.table().stripes().len();
14257        assert!(stripes > CACHED_STRIPES_PER_COLUMN, "the floor has to be smaller than a table");
14258        let scan = |reader: &Reader| {
14259            for part in 0..parts {
14260                let chunk = reader.read(part, &[0]).expect("a part");
14261                assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
14262            }
14263        };
14264        scan(&a);
14265        scan(&a);
14266        assert_eq!(a.pages.load(Atomic::Relaxed), stripes, "the second scan reads nothing");
14267        let one = pool.bytes();
14268        assert!(one > 0, "the pool counts what the reader holds");
14269
14270        // Room for one table. Reading the other takes the first one's pages down to its floor.
14271        pool.budget.store(one, Atomic::Relaxed);
14272        scan(&b);
14273        assert_eq!(b.pages.load(Atomic::Relaxed), stripes, "a page is never let go while in use");
14274        assert_eq!(a.cache.held[0].load(Atomic::Relaxed), CACHED_STRIPES_PER_COLUMN);
14275        let held = a.cache.columns[0].lock().expect("the column").pages.iter().flatten().count();
14276        assert_eq!(held, CACHED_STRIPES_PER_COLUMN, "the count and the slots agree");
14277
14278        // A reader that goes takes its pages out of the count with it.
14279        drop((a, b, catalog));
14280        let c = Catalog::open_in(&path, &pool).expect("again").table("a").expect("a");
14281        scan(&c);
14282        assert!(pool.bytes() <= one, "only what the live reader holds is counted");
14283        fs::remove_file(path).expect("remove scratch file");
14284    }
14285
14286    /// A worker per stripe reads its stripe once, once the cache has been told how many there are.
14287    ///
14288    /// This is the shape a scan has when it hands out a whole stripe per morsel rather than a part.
14289    /// Nobody races for a page any more, but every worker holds a different one for the length of a
14290    /// stripe, so a cache that keeps four pages while eight workers are in eight stripes evicts
14291    /// every one of them before its owner has finished with it, and the owner reads a quarter of a
14292    /// megabyte again for the next part. The barrier is what makes that certain rather than likely:
14293    /// without it a worker can run a whole stripe before the next one starts and never collide.
14294    #[test]
14295    fn a_worker_per_stripe_reads_its_page_once_when_the_cache_was_told_to_expect_it() {
14296        let workers = CACHED_STRIPES_PER_COLUMN + 4;
14297        let path = path("stripe-per-worker");
14298        let mut writer =
14299            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
14300                .expect("new file");
14301        for part in 0..STRIPE_PARTS * workers {
14302            let chunk = Chunk::new(vec![
14303                Vector::from_values(LogicalType::Integer, &[Value::Integer(part as i32)])
14304                    .expect("integers"),
14305            ])
14306            .expect("matching rows");
14307            writer.append(&chunk).expect("one part");
14308        }
14309        writer.finish().expect("commit");
14310
14311        let read = |told: bool| {
14312            let reader = Reader::open(&path).expect("reopen from disk");
14313            assert_eq!(reader.table().stripes().len(), workers, "a stripe per worker");
14314            if told {
14315                reader.keep_stripes(workers);
14316            }
14317            let barrier = std::sync::Barrier::new(workers);
14318            std::thread::scope(|scope| {
14319                for (worker, run) in reader.stripe_parts().into_iter().enumerate() {
14320                    let reader = &reader;
14321                    let barrier = &barrier;
14322                    scope.spawn(move || {
14323                        for part in run {
14324                            barrier.wait();
14325                            let chunk = reader.read(part, &[0]).expect("a part of my own stripe");
14326                            assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
14327                        }
14328                        assert!(worker < workers);
14329                    });
14330                }
14331            });
14332            reader.pages.load(Atomic::Relaxed)
14333        };
14334
14335        assert_eq!(read(true), workers, "one page read per stripe and no more");
14336        assert!(read(false) > workers, "a cache that small is read again on every part");
14337        fs::remove_file(path).expect("remove scratch file");
14338    }
14339
14340    /// A damaged index page is caught before anything decodes a part out of it.
14341    ///
14342    /// The index is the one structure a reader trusts to find bytes with, so it carries a checksum
14343    /// per column section rather than one for the page, and this is what says that check runs.
14344    #[test]
14345    fn a_damaged_index_page_is_an_error() {
14346        let path = path("damaged-index");
14347        let mut writer =
14348            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
14349                .expect("new file");
14350        writer.append(&sample_ids()).expect("first part");
14351        writer.append(&sample_ids()).expect("second part");
14352        writer.finish().expect("commit");
14353
14354        let reader = Reader::open(&path).expect("valid directory");
14355        let index = reader.table.stripes[0].index;
14356        let mut byte = [0; 1];
14357        read_at(&reader.file, index.offset, &mut byte).expect("the first part length");
14358        let mut file = OpenOptions::new().write(true).open(&path).expect("open index page");
14359        file.seek(SeekFrom::Start(index.offset)).expect("index start");
14360        file.write_all(&[!byte[0]]).expect("damage the first part length");
14361        let error = reader.read(1, &[0]).expect_err("a damaged index must not be used");
14362        assert!(error.message().contains("index page section checksum differs"), "{error}");
14363        fs::remove_file(path).expect("remove scratch file");
14364    }
14365
14366    /// Every integer width the format knows about, written and read back.
14367    ///
14368    /// The unsigned ones are the reason ClickBench can be stored at all: `hits` types `EventDate`
14369    /// as `USMALLINT`, and one unsupported column meant the whole table was refused. The extremes
14370    /// are in here on purpose, because a width that round trips through the wrong signedness only
14371    /// goes wrong at the end of its range.
14372    #[test]
14373    fn every_integer_width_round_trips_through_a_page() {
14374        let path = path("integer-widths");
14375        let columns = [
14376            (LogicalType::TinyInt, vec![Value::TinyInt(i8::MIN), Value::TinyInt(i8::MAX)]),
14377            (LogicalType::UTinyInt, vec![Value::UTinyInt(0), Value::UTinyInt(u8::MAX)]),
14378            (LogicalType::SmallInt, vec![Value::SmallInt(i16::MIN), Value::SmallInt(i16::MAX)]),
14379            (LogicalType::USmallInt, vec![Value::USmallInt(0), Value::USmallInt(u16::MAX)]),
14380            (LogicalType::Integer, vec![Value::Integer(i32::MIN), Value::Integer(i32::MAX)]),
14381            (LogicalType::UInteger, vec![Value::UInteger(0), Value::UInteger(u32::MAX)]),
14382            (LogicalType::BigInt, vec![Value::BigInt(i64::MIN), Value::BigInt(i64::MAX)]),
14383            (LogicalType::UBigInt, vec![Value::UBigInt(0), Value::UBigInt(u64::MAX)]),
14384        ];
14385        let fields = columns
14386            .iter()
14387            .enumerate()
14388            .map(|(at, (ty, _))| Field::required(format!("c{at}"), ty.clone()))
14389            .collect::<Vec<_>>();
14390        let vectors = columns
14391            .iter()
14392            .map(|(ty, values)| Vector::from_values(ty.clone(), values).expect("a vector"))
14393            .collect::<Vec<_>>();
14394        let mut writer = Writer::create(&path, "widths", fields).expect("new file");
14395        writer.append(&Chunk::new(vectors).expect("matching rows")).expect("one stripe");
14396        writer.finish().expect("commit");
14397
14398        let reader = Reader::open(&path).expect("reopen from disk");
14399        let wanted = (0..columns.len()).collect::<Vec<_>>();
14400        let read = reader.read(0, &wanted).expect("every column");
14401        assert_eq!(read.len(), 2);
14402        // row at a time: each column has its own type and its own pair of extremes.
14403        for (at, (ty, values)) in columns.iter().enumerate() {
14404            assert_eq!(read.value_at(0, at), values[0], "the low end of {ty}");
14405            assert_eq!(read.value_at(1, at), values[1], "the high end of {ty}");
14406        }
14407        fs::remove_file(path).expect("remove scratch file");
14408    }
14409
14410    /// The rest of the fixed width types, and the byte strings, written and read back.
14411    ///
14412    /// The extremes again, and for a float that means more than the ends of the range. Negative
14413    /// zero and a NaN are the two values that go through an encoder unnoticed and come back
14414    /// different, so they are here on purpose, and the NaN is compared by its bits rather than by
14415    /// `==`, which a NaN fails against itself.
14416    ///
14417    /// A blob is here beside them because it is the same round trip asked of bytes that are not
14418    /// text. The value in it is not UTF-8, so a path that reads a payload as a string on the way
14419    /// past turns this test red rather than turning a user's column into nulls.
14420    #[test]
14421    fn every_other_type_the_format_knows_round_trips_through_a_page() {
14422        let path = path("other-types");
14423        let columns = [
14424            (LogicalType::Float, vec![Value::Float(f32::MIN), Value::Float(-0.0)]),
14425            (LogicalType::Double, vec![Value::Double(f64::MIN), Value::Double(f64::MAX)]),
14426            (LogicalType::HugeInt, vec![Value::HugeInt(i128::MIN), Value::HugeInt(i128::MAX)]),
14427            (LogicalType::UHugeInt, vec![Value::UHugeInt(0), Value::UHugeInt(u128::MAX)]),
14428            (LogicalType::Time, vec![Value::Time(0), Value::Time(86_399_999_999)]),
14429            (LogicalType::TimeTz, vec![Value::TimeTz(-50_400_000_000), Value::TimeTz(0)]),
14430            (
14431                LogicalType::TimestampTz,
14432                vec![Value::TimestampTz(i64::MIN + 1), Value::TimestampTz(i64::MAX)],
14433            ),
14434            (
14435                LogicalType::Interval,
14436                vec![
14437                    Value::Interval { months: i32::MIN, days: i32::MAX, micros: i64::MIN },
14438                    Value::Interval { months: 13, days: -1, micros: 1 },
14439                ],
14440            ),
14441            (
14442                LogicalType::Blob,
14443                vec![Value::Blob(vec![0, 0xff, 0x80, 0xfe]), Value::Blob(Vec::new())],
14444            ),
14445        ];
14446        let fields = columns
14447            .iter()
14448            .enumerate()
14449            .map(|(at, (ty, _))| Field::required(format!("c{at}"), ty.clone()))
14450            .collect::<Vec<_>>();
14451        let vectors = columns
14452            .iter()
14453            .map(|(ty, values)| Vector::from_values(ty.clone(), values).expect("a vector"))
14454            .collect::<Vec<_>>();
14455        let mut writer = Writer::create(&path, "others", fields).expect("new file");
14456        writer.append(&Chunk::new(vectors).expect("matching rows")).expect("one stripe");
14457        writer.finish().expect("commit");
14458
14459        let reader = Reader::open(&path).expect("reopen from disk");
14460        let wanted = (0..columns.len()).collect::<Vec<_>>();
14461        let read = reader.read(0, &wanted).expect("every column");
14462        assert_eq!(read.len(), 2);
14463        for (at, (ty, values)) in columns.iter().enumerate() {
14464            assert_eq!(read.value_at(0, at), values[0], "the low end of {ty}");
14465            assert_eq!(read.value_at(1, at), values[1], "the high end of {ty}");
14466        }
14467        // A float keeps its sign through a zero, which `==` says nothing about because negative
14468        // zero and zero compare equal.
14469        let Value::Float(zero) = read.value_at(1, 0) else { panic!("a float stays a float") };
14470        assert!(zero.is_sign_negative(), "a negative zero came back as {zero}");
14471
14472        fs::remove_file(path).expect("remove scratch file");
14473    }
14474
14475    /// A NaN is still a NaN after a trip through a page.
14476    ///
14477    /// Apart from the other floats because it cannot be asserted the same way. A NaN is not equal
14478    /// to itself, so a comparison against the value that was written passes for every NaN and for
14479    /// nothing else, which is the one assertion that would not catch a page that lost it.
14480    #[test]
14481    fn a_nan_survives_being_written_down() {
14482        let path = path("nan");
14483        let nan = Vector::from_values(LogicalType::Double, &[Value::Double(f64::NAN)])
14484            .expect("a NaN vector");
14485        let mut writer =
14486            Writer::create(&path, "nan", vec![Field::required("d", LogicalType::Double)])
14487                .expect("new file");
14488        writer.append(&Chunk::new(vec![nan]).expect("one column")).expect("one stripe");
14489        writer.finish().expect("commit");
14490        let read = Reader::open(&path).expect("reopen").read(0, &[0]).expect("the column");
14491        let Value::Double(back) = read.value_at(0, 0) else { panic!("a double stays a double") };
14492        assert!(back.is_nan(), "a NaN came back as {back}");
14493        fs::remove_file(path).expect("remove scratch file");
14494    }
14495
14496    /// A uuid and a bit string, which have no `Value` arm of their own and are checked as bits.
14497    ///
14498    /// A uuid is the 128 bit lane and a bit string is bytes, and neither of them reads back as
14499    /// anything in `Value` today, so asking for a value here would compare two nulls and pass
14500    /// whatever the file held. The data underneath is what the storage promise is about, so that is
14501    /// what this reads.
14502    #[test]
14503    fn a_uuid_and_a_bit_string_come_back_as_the_bits_that_went_in() {
14504        let path = path("uuid-and-bit");
14505        let uuids = vec![0_i128, i128::MIN, -1];
14506        let mut bits = StringColumn::new();
14507        for value in [&b"\x02\xff"[..], &b""[..], &b"\x00\x01\x02\x03\x04\x05"[..]] {
14508            bits.push_bytes(value);
14509        }
14510        let expected = bits.clone();
14511        let fields =
14512            vec![Field::required("u", LogicalType::Uuid), Field::required("b", LogicalType::Bit)];
14513        let vectors = vec![
14514            Vector::flat(LogicalType::Uuid, Data::Int128(uuids.clone().into())).expect("uuids"),
14515            Vector::flat(LogicalType::Bit, Data::Varlen(bits)).expect("bit strings"),
14516        ];
14517        let mut writer = Writer::create(&path, "ids", fields).expect("new file");
14518        writer.append(&Chunk::new(vectors).expect("matching rows")).expect("one stripe");
14519        writer.finish().expect("commit");
14520
14521        let reader = Reader::open(&path).expect("reopen from disk");
14522        let read = reader.read(0, &[0, 1]).expect("both columns").flatten().expect("flat");
14523        let Some(Data::Int128(back)) = read.column(0).expect("the uuids").data() else {
14524            panic!("a uuid column is the 128 bit lane")
14525        };
14526        assert_eq!(back.as_slice(), uuids.as_slice());
14527        let Some(Data::Varlen(back)) = read.column(1).expect("the bits").data() else {
14528            panic!("a bit column is bytes")
14529        };
14530        for row in 0..expected.len() {
14531            assert_eq!(back.bytes(row), expected.bytes(row), "row {row} of the bit column");
14532        }
14533        fs::remove_file(path).expect("remove scratch file");
14534    }
14535
14536    /// Counting a run at once has to leave the candidate table exactly where counting its rows one
14537    /// at a time would, including once the table is full and a run is turned away row by row.
14538    #[test]
14539    fn a_run_counted_at_once_leaves_the_candidates_a_row_at_a_time_would() {
14540        let mut rows: Vec<Option<u64>> = Vec::new();
14541        let mut state = 0x2545_f491_4f6c_dd1d_u64;
14542        for index in 0..400_000_u64 {
14543            state ^= state << 13;
14544            state ^= state >> 7;
14545            state ^= state << 17;
14546            let times = 1 + (state % 7) as usize;
14547            let bits = match state % 11 {
14548                0 => None,
14549                1..=3 => Some(state % 16),
14550                _ => Some(index.wrapping_mul(0x9e37_79b9_7f4a_7c15)),
14551            };
14552            rows.extend(std::iter::repeat_n(bits, times));
14553        }
14554        let mut by_row = Candidates::default();
14555        for &bits in &rows {
14556            by_row.add(bits, 1);
14557        }
14558        let mut by_run = Candidates::default();
14559        let mut run = Run::default();
14560        let mut runs = 0_usize;
14561        for &bits in &rows {
14562            if let Some((bits, times)) = run.push(bits) {
14563                by_run.add(bits, times);
14564                runs += 1;
14565            }
14566        }
14567        if let Some((bits, times)) = run.take() {
14568            by_run.add(bits, times);
14569        }
14570        assert!(runs < rows.len() / 2, "the rows came in runs");
14571        assert!(by_row.decrements > 0, "the table filled and turned values away");
14572        assert_eq!(sorted_candidates(&by_run), sorted_candidates(&by_row));
14573        assert_eq!(by_run.nulls, by_row.nulls);
14574        assert_eq!(by_run.decrements, by_row.decrements);
14575    }
14576
14577    fn sorted_candidates(candidates: &Candidates) -> Vec<(u64, u32)> {
14578        let mut pairs = candidates.pairs().collect::<Vec<_>>();
14579        pairs.sort_unstable();
14580        assert_eq!(pairs.len(), candidates.held, "the count of held slots drifted");
14581        pairs
14582    }
14583
14584    /// The Misra-Gries table as it was written over a `HashMap`, kept as the oracle the open
14585    /// addressed one has to agree with.
14586    #[derive(Default)]
14587    struct MapCandidates {
14588        counts: HashMap<u64, u32>,
14589        nulls: u32,
14590        decrements: u64,
14591    }
14592
14593    impl MapCandidates {
14594        fn add(&mut self, bits: Option<u64>, mut times: u32) {
14595            while times > 0 {
14596                let held = match bits {
14597                    Some(bits) => self.counts.get_mut(&bits),
14598                    None if self.nulls != 0 => Some(&mut self.nulls),
14599                    None => None,
14600                };
14601                if let Some(count) = held {
14602                    *count = count.saturating_add(times);
14603                    return;
14604                }
14605                if self.counts.len() + usize::from(self.nulls != 0) < FREQUENCY_CANDIDATES {
14606                    match bits {
14607                        Some(bits) => {
14608                            self.counts.insert(bits, times);
14609                        }
14610                        None => self.nulls = times,
14611                    }
14612                    return;
14613                }
14614                self.counts.retain(|_, count| {
14615                    *count -= 1;
14616                    *count != 0
14617                });
14618                self.nulls = self.nulls.saturating_sub(1);
14619                self.decrements = self.decrements.saturating_add(1);
14620                times -= 1;
14621            }
14622        }
14623    }
14624
14625    /// Near unique values, a few heavy ones, nulls, and runs, through enough rows that the table
14626    /// fills, grows through every size and is decremented many times over. Both tables have to hold
14627    /// the same candidates with the same counts at the end, and at points along the way.
14628    #[test]
14629    fn the_open_addressed_candidates_agree_with_the_map_they_replaced() {
14630        for seed in [0x2545_f491_4f6c_dd1d_u64, 0x9e37_79b9_7f4a_7c15, 7] {
14631            let mut table = Candidates::default();
14632            let mut oracle = MapCandidates::default();
14633            let mut state = seed;
14634            for index in 0..300_000_u64 {
14635                state ^= state << 13;
14636                state ^= state >> 7;
14637                state ^= state << 17;
14638                let bits = match state % 13 {
14639                    0 => None,
14640                    1..=4 => Some(state % 40),
14641                    5 => Some((index % 1000) * 1_000_000),
14642                    _ => Some(state),
14643                };
14644                let times = 1 + (state >> 60) as u32 % 3;
14645                table.add(bits, times);
14646                oracle.add(bits, times);
14647                if index % 50_000 == 0 {
14648                    let mut expected =
14649                        oracle.counts.iter().map(|(&b, &c)| (b, c)).collect::<Vec<_>>();
14650                    expected.sort_unstable();
14651                    assert_eq!(sorted_candidates(&table), expected, "seed {seed} row {index}");
14652                }
14653            }
14654            let mut expected = oracle.counts.iter().map(|(&b, &c)| (b, c)).collect::<Vec<_>>();
14655            expected.sort_unstable();
14656            assert_eq!(sorted_candidates(&table), expected, "seed {seed}");
14657            assert_eq!(table.nulls, oracle.nulls, "seed {seed}");
14658            assert_eq!(table.decrements, oracle.decrements, "seed {seed}");
14659            assert!(table.decrements > 0, "seed {seed} never filled the table");
14660            for &(bits, _) in &expected {
14661                assert!(table.position(bits).is_some(), "seed {seed} lost {bits}");
14662            }
14663        }
14664    }
14665
14666    #[test]
14667    fn numeric_frequency_candidates_keep_bounded_row_ordinals() {
14668        let path = path("frequency-ordinals");
14669        let mut writer =
14670            Writer::create(&path, "items", vec![Field::required("id", LogicalType::BigInt)])
14671                .expect("new file");
14672        let mut values = Vec::new();
14673        for leader in 0..10_i64 {
14674            values.extend(std::iter::repeat_n(leader, 100));
14675        }
14676        values.extend(1_000_i64..41_000);
14677        for part in values.chunks(1_024) {
14678            let vector = Vector::flat(LogicalType::BigInt, Data::Int64(part.to_vec().into()))
14679                .expect("big integers");
14680            writer.append(&Chunk::new(vec![vector]).expect("one column")).expect("one stripe");
14681        }
14682        writer.finish().expect("commit");
14683
14684        let reader = Reader::open(&path).expect("reopen from disk");
14685        let occurrences =
14686            reader.frequency_occurrences(0).expect("valid metadata").expect("bounded ordinals");
14687        assert!(occurrences.omitted_max < 100);
14688        assert!(occurrences.ordinals.len() <= FREQUENCY_ORDINALS);
14689        assert_eq!(occurrences.anchor_indices.len(), occurrences.ordinals.len());
14690        assert!(occurrences.ordinals.windows(2).all(|pair| pair[0] < pair[1]));
14691        assert_eq!(&occurrences.ordinals[..1_000], &(0_u64..1_000).collect::<Vec<_>>());
14692        assert_eq!(
14693            &occurrences.anchor_indices[..1_000]
14694                .iter()
14695                .map(|&entry| occurrences.anchors[entry as usize].clone())
14696                .collect::<Vec<_>>(),
14697            &(0_i64..10)
14698                .flat_map(|leader| std::iter::repeat_n(Value::BigInt(leader), 100))
14699                .collect::<Vec<_>>()
14700        );
14701        fs::remove_file(path).expect("remove scratch file");
14702    }
14703
14704    #[test]
14705    fn numeric_frequencies_count_nulls_and_values_past_the_top_of_bigint() {
14706        // Ten leaders, then more unique values than the candidate table holds, so the first pass
14707        // has to decrement and the counts come from the recount. The unsigned leaders sit above
14708        // `i64::MAX`, where reading the bits as signed would give a different value, and the signed
14709        // ones are negative, where reading them as unsigned would.
14710        let path = path("frequency-bits");
14711        let mut writer = Writer::create(
14712            &path,
14713            "items",
14714            vec![Field::new("u", LogicalType::UBigInt), Field::new("s", LogicalType::BigInt)],
14715        )
14716        .expect("new file");
14717        let mut rows = Vec::new();
14718        let mut leaders = Vec::new();
14719        for leader in 0..10_u64 {
14720            let count = 300 - leader * 10;
14721            let (unsigned, signed) = if leader == 0 {
14722                (Value::Null, Value::Null)
14723            } else {
14724                (Value::UBigInt(u64::MAX - leader), Value::BigInt(-(leader as i64)))
14725            };
14726            rows.extend(std::iter::repeat_n((unsigned.clone(), signed.clone()), count as usize));
14727            leaders.push(((unsigned, count), (signed, count)));
14728        }
14729        rows.extend((1_000..41_000_u64).map(|id| (Value::UBigInt(id), Value::BigInt(id as i64))));
14730        for part in rows.chunks(1_024) {
14731            let unsigned = part.iter().map(|(value, _)| value.clone()).collect::<Vec<_>>();
14732            let signed = part.iter().map(|(_, value)| value.clone()).collect::<Vec<_>>();
14733            let chunk = Chunk::new(vec![
14734                Vector::from_values(LogicalType::UBigInt, &unsigned).expect("unsigned"),
14735                Vector::from_values(LogicalType::BigInt, &signed).expect("signed"),
14736            ])
14737            .expect("matching columns");
14738            writer.append(&chunk).expect("rows");
14739        }
14740        writer.finish().expect("commit");
14741
14742        let reader = Reader::open(&path).expect("reopen from disk");
14743        for column in 0..2 {
14744            let prefix =
14745                reader.frequency_prefix(column).expect("valid metadata").expect("a synopsis");
14746            let wanted = leaders
14747                .iter()
14748                .map(|(unsigned, signed)| if column == 0 { unsigned } else { signed })
14749                .cloned()
14750                .collect::<Vec<_>>();
14751            assert_eq!(&prefix.entries[..10], &wanted[..], "column {column}");
14752            assert!(prefix.omitted_max < 210, "column {column}");
14753            assert_eq!(
14754                reader.distinct_values(column).expect("valid metadata"),
14755                Some(9 + 40_000),
14756                "column {column}"
14757            );
14758        }
14759        fs::remove_file(path).expect("remove scratch file");
14760    }
14761
14762    #[test]
14763    fn a_narrow_column_takes_its_frequencies_from_the_tally_and_they_match_the_rows() {
14764        // Every column here has fewer distinct values than the tally holds, so the close takes its
14765        // counts from the gather rather than reading the pages back. The types are the ones whose
14766        // bits could come out wrong on that road: a negative tiny integer that has to be sign
14767        // extended, an unsigned one past the top of `INTEGER`, a date and a timestamp. A null every
14768        // thirteenth row checks that the nulls come from the pass and not from the list.
14769        let path = path("frequency-tally");
14770        let types = [
14771            LogicalType::TinyInt,
14772            LogicalType::UInteger,
14773            LogicalType::Date,
14774            LogicalType::Timestamp,
14775        ];
14776        let value = |ty: &LogicalType, at: i64| match ty {
14777            LogicalType::TinyInt => Value::TinyInt((at % 250 - 125) as i8),
14778            LogicalType::UInteger => Value::UInteger(u32::MAX - at as u32),
14779            LogicalType::Date => Value::Date(19_000 - at as i32),
14780            _ => Value::Timestamp(1_700_000_000_000_000 - at * 1_000_003),
14781        };
14782        let fields = types
14783            .iter()
14784            .enumerate()
14785            .map(|(at, ty)| Field::new(format!("c{at}"), ty.clone()))
14786            .collect::<Vec<_>>();
14787        let mut writer = Writer::create(&path, "items", fields).expect("new file");
14788        let mut rows = Vec::new();
14789        for at in 0..250_i64 {
14790            for _ in 0..=(at % 37) {
14791                rows.push(if rows.len() % 13 == 0 { None } else { Some(at) });
14792            }
14793        }
14794        for part in rows.chunks(1_000) {
14795            let columns = types
14796                .iter()
14797                .map(|ty| {
14798                    let values = part
14799                        .iter()
14800                        .map(|row| row.map_or(Value::Null, |at| value(ty, at)))
14801                        .collect::<Vec<_>>();
14802                    Vector::from_values(ty.clone(), &values).expect("a column")
14803                })
14804                .collect();
14805            writer.append(&Chunk::new(columns).expect("matching columns")).expect("rows");
14806        }
14807        writer.finish().expect("commit");
14808
14809        let reader = Reader::open(&path).expect("reopen from disk");
14810        for (column, ty) in types.iter().enumerate() {
14811            let mut counts = HashMap::<Option<i64>, u64>::new();
14812            for row in &rows {
14813                *counts.entry(*row).or_default() += 1;
14814            }
14815            let wanted = counts
14816                .into_iter()
14817                .map(|(row, count)| (row.map_or(Value::Null, |at| value(ty, at)), count))
14818                .collect::<Vec<_>>();
14819            let prefix =
14820                reader.frequency_prefix(column).expect("valid metadata").expect("a synopsis");
14821            assert_eq!(prefix.entries.len(), wanted.len(), "column {column}");
14822            assert_eq!(prefix.omitted_max, 0, "column {column}");
14823            for (value, count) in &prefix.entries {
14824                let held =
14825                    wanted.iter().find(|(wanted, _)| wanted == value).map(|(_, count)| count);
14826                assert_eq!(held, Some(count), "column {column} value {value:?}");
14827            }
14828            assert!(prefix.entries.windows(2).all(|pair| pair[0].1 >= pair[1].1));
14829            assert_eq!(
14830                reader.distinct_values(column).expect("valid metadata"),
14831                Some(wanted.len() as u64 - 1),
14832                "column {column}"
14833            );
14834        }
14835        fs::remove_file(path).expect("remove scratch file");
14836    }
14837
14838    #[test]
14839    fn distinct_counts_are_exact_either_side_of_a_full_candidate_table() {
14840        // The count comes from the candidate table while it has room and from the set once it
14841        // fills, so the sizes around the fill, with and without a null taking a place, are where a
14842        // value could be counted twice or missed. Zero is in every column because the set keeps it
14843        // apart from the other values, and every value comes back later to be counted again.
14844        let edge = FREQUENCY_CANDIDATES as i64;
14845        for distinct in [0, 1, 7, edge - 2, edge - 1, edge, edge + 1, edge + 2, 3 * edge] {
14846            for with_null in [false, true] {
14847                let path = path("distinct-edge");
14848                let mut writer =
14849                    Writer::create(&path, "items", vec![Field::new("id", LogicalType::BigInt)])
14850                        .expect("new file");
14851                let mut values = Vec::new();
14852                for round in 0..2 {
14853                    for value in 0..distinct {
14854                        let repeat = if round == 0 { 1 + (value % 3) as usize } else { 1 };
14855                        values.extend(std::iter::repeat_n(
14856                            Value::BigInt(value * 7_919 % distinct),
14857                            repeat,
14858                        ));
14859                        if with_null && value % 1_000 == 0 {
14860                            values.push(Value::Null);
14861                        }
14862                    }
14863                }
14864                if with_null {
14865                    values.push(Value::Null);
14866                }
14867                for part in values.chunks(1_024) {
14868                    let chunk = Chunk::new(vec![
14869                        Vector::from_values(LogicalType::BigInt, part).expect("ids"),
14870                    ])
14871                    .expect("one column");
14872                    writer.append(&chunk).expect("rows");
14873                }
14874                writer.finish().expect("commit");
14875                let reader = Reader::open(&path).expect("reopen from disk");
14876                assert_eq!(
14877                    reader.distinct_values(0).expect("valid metadata"),
14878                    Some(distinct as u64),
14879                    "{distinct} values, null {with_null}"
14880                );
14881                fs::remove_file(path).expect("remove scratch file");
14882            }
14883        }
14884    }
14885
14886    #[test]
14887    fn narrow_nonzero_count_matches_the_full_reader_across_stripes() {
14888        let path = path("quick-nonzero");
14889        let mut writer = Writer::create(
14890            &path,
14891            "items",
14892            vec![Field::new("label", LogicalType::Varchar), Field::new("id", LogicalType::Integer)],
14893        )
14894        .expect("create");
14895        for ids in [
14896            &[Value::Integer(0), Value::Null, Value::Integer(3)][..],
14897            &[Value::Integer(0), Value::Integer(7), Value::Null][..],
14898        ] {
14899            let labels = vec![Value::Varchar("same".into()); ids.len()];
14900            writer
14901                .append(
14902                    &Chunk::new(vec![
14903                        Vector::from_values(LogicalType::Varchar, &labels).expect("labels"),
14904                        Vector::from_values(LogicalType::Integer, ids).expect("ids"),
14905                    ])
14906                    .expect("chunk"),
14907                )
14908                .expect("append");
14909        }
14910        writer.finish().expect("finish");
14911        let catalog = Catalog::open(&path).expect("catalog");
14912        assert_eq!(catalog.entries[0].nonzero, vec![None, None]);
14913        assert_eq!(catalog.entries[0].aggregates, vec![None, Some((10, 4))]);
14914        assert_eq!(catalog.entries[0].distincts, vec![Some(1), Some(3)]);
14915        let frequencies =
14916            catalog.exact_numeric_frequencies("items", 1).expect("frequencies").expect("complete");
14917        assert_eq!(frequencies.len(), 4);
14918        for pair in [(Some(0), 2), (Some(3), 1), (Some(7), 1), (None, 2)] {
14919            assert!(frequencies.contains(&pair), "missing {pair:?}");
14920        }
14921        assert_eq!(catalog.distinct_count("items", 1).expect("distinct count"), Some(3));
14922        assert_eq!(
14923            catalog.integer_extremes("items", 1).expect("extremes"),
14924            Some(IntegerExtremes::Values { low: 0, high: 7 })
14925        );
14926        assert_eq!(
14927            catalog.aggregate_sums("items", &[1]).expect("catalog sums"),
14928            Some(CertifiedSums { columns: vec![(10, 4)], rows: 6 })
14929        );
14930        assert_eq!(catalog.nonzero_count("items", 1).expect("quick count"), Some(2));
14931        let mut legacy = catalog.clone();
14932        Arc::make_mut(&mut legacy.entries)[0].nonzero[1] = Some(999);
14933        assert_eq!(legacy.nonzero_count("items", 1).expect("ignore legacy count"), Some(2));
14934        Arc::make_mut(&mut legacy.entries)[0].frequencies[1] = None;
14935        assert_eq!(legacy.nonzero_count("items", 1).expect("directory fallback"), Some(2));
14936        Writer::certify_counts(&path).expect("recertify");
14937        assert_eq!(
14938            Catalog::open(&path).expect("reopen").nonzero_count("items", 1).expect("count"),
14939            Some(2)
14940        );
14941        assert_eq!(
14942            Catalog::open(&path).expect("reopen").aggregate_sums("items", &[1]).expect("sums"),
14943            Some(CertifiedSums { columns: vec![(10, 4)], rows: 6 })
14944        );
14945        assert_eq!(
14946            Catalog::open(&path).expect("reopen").distinct_count("items", 1).expect("distinct"),
14947            Some(3)
14948        );
14949        assert_eq!(
14950            Catalog::open(&path).expect("reopen").integer_extremes("items", 1).expect("ends"),
14951            Some(IntegerExtremes::Values { low: 0, high: 7 })
14952        );
14953        assert_eq!(
14954            Catalog::open(&path)
14955                .expect("reopen")
14956                .exact_numeric_frequencies("items", 1)
14957                .expect("frequencies"),
14958            Some(frequencies)
14959        );
14960        assert_eq!(catalog.table("items").expect("reader").null_count(1).expect("nulls"), 2);
14961        fs::remove_file(path).expect("remove scratch file");
14962    }
14963
14964    #[test]
14965    fn numeric_string_pair_leaders_are_certified_in_the_directory() {
14966        let path = path("pair-frequencies");
14967        let mut pairs = Vec::new();
14968        pairs.extend(std::iter::repeat_n((1_i64, "alpha".to_string()), 100));
14969        pairs.extend(std::iter::repeat_n((1_i64, "beta".to_string()), 50));
14970        pairs.extend(std::iter::repeat_n((2_i64, "gamma".to_string()), 40));
14971        pairs.extend((1_000_i64..1_600).map(|id| (id, format!("tail {id}"))));
14972        let mut writer = Writer::create(
14973            &path,
14974            "items",
14975            vec![
14976                Field::required("id", LogicalType::BigInt),
14977                Field::required("phrase", LogicalType::Varchar),
14978            ],
14979        )
14980        .expect("new file");
14981        for part in pairs.chunks(1_024) {
14982            let ids = part.iter().map(|(id, _)| Value::BigInt(*id)).collect::<Vec<_>>();
14983            let phrases =
14984                part.iter().map(|(_, phrase)| Value::Varchar(phrase.clone())).collect::<Vec<_>>();
14985            writer
14986                .append(
14987                    &Chunk::new(vec![
14988                        Vector::from_values(LogicalType::BigInt, &ids).expect("ids"),
14989                        Vector::from_values(LogicalType::Varchar, &phrases).expect("phrases"),
14990                    ])
14991                    .expect("matching columns"),
14992                )
14993                .expect("rows");
14994        }
14995        writer.finish().expect("commit");
14996
14997        let reader = Reader::open(&path).expect("reopen from disk");
14998        assert!(
14999            reader.table.pair_frequencies.is_empty(),
15000            "no query-specific pair result is stored"
15001        );
15002        fs::remove_file(path).expect("remove scratch file");
15003    }
15004
15005    /// The bug this is here for cost a 43 GB ClickBench table and an hour of reloading it. The
15006    /// format went from 11 to 12, every binary built after that said "magic or major version is
15007    /// unsupported" about the file, and there was no way to tell from the message whether the path
15008    /// was wrong, the file was truncated, or it was ours and simply older. The number this build
15009    /// wants is the whole answer and it was the one thing the message did not carry.
15010    #[test]
15011    fn a_file_from_another_format_says_which_format_it_is() {
15012        let older = path("older-format");
15013        let mut writer =
15014            Writer::create(&older, "items", vec![Field::new("id", LogicalType::Integer)])
15015                .expect("new file");
15016        let chunk = Chunk::new(vec![
15017            Vector::flat(LogicalType::Integer, Data::Int32(vec![1, 2, 3].into()))
15018                .expect("integers"),
15019        ])
15020        .expect("chunk");
15021        writer.append(&chunk).expect("page written");
15022        writer.finish().expect("commit");
15023
15024        // A format below the whole readable set, rather than `FORMAT - 1`, because the set has
15025        // more than one member now: format 22 is deliberately still readable, so the version that
15026        // has to be refused is the one under the oldest one accepted.
15027        let unreadable =
15028            READABLE.iter().copied().min().expect("at least one format is readable") - 1;
15029        let mut file = OpenOptions::new().write(true).open(&older).expect("open for the header");
15030        file.seek(SeekFrom::Start(8)).expect("the version follows the magic");
15031        file.write_all(&unreadable.to_le_bytes()).expect("write an older version");
15032        drop(file);
15033        let complaint = Reader::open(&older).expect_err("an older format is refused").to_string();
15034        assert!(complaint.contains(&format!("format {unreadable}")), "{complaint}");
15035        assert!(complaint.contains(&format!("format {FORMAT}")), "{complaint}");
15036
15037        let mut file = OpenOptions::new().write(true).open(&older).expect("open for the header");
15038        file.seek(SeekFrom::Start(0)).expect("the magic is first");
15039        file.write_all(b"NOTRUDB!").expect("write another engine's magic");
15040        drop(file);
15041        let complaint = Reader::open(&older).expect_err("a foreign file is refused").to_string();
15042        assert!(complaint.contains("magic"), "{complaint}");
15043        assert!(!complaint.contains("format"), "a version has nothing to do with it: {complaint}");
15044        fs::remove_file(older).expect("remove scratch file");
15045    }
15046
15047    #[test]
15048    fn an_unfinished_or_damaged_file_does_not_answer_with_partial_rows() {
15049        let unfinished = path("unfinished");
15050        let mut writer =
15051            Writer::create(&unfinished, "items", vec![Field::new("id", LogicalType::Integer)])
15052                .expect("new file");
15053        let chunk = Chunk::new(vec![
15054            Vector::flat(LogicalType::Integer, Data::Int32(vec![1, 2, 3].into()))
15055                .expect("integers"),
15056        ])
15057        .expect("chunk");
15058        writer.append(&chunk).expect("page written");
15059        drop(writer);
15060        assert!(Reader::open(&unfinished).is_err(), "no directory was committed");
15061        fs::remove_file(unfinished).expect("remove scratch file");
15062
15063        let damaged = path("damaged");
15064        let mut writer =
15065            Writer::create(&damaged, "items", vec![Field::new("id", LogicalType::Integer)])
15066                .expect("new file");
15067        writer.append(&chunk).expect("page written");
15068        writer.finish().expect("commit");
15069        let reader = Reader::open(&damaged).expect("valid directory");
15070        let mut file =
15071            OpenOptions::new().write(true).open(&damaged).expect("open for a damaged page");
15072        file.seek(SeekFrom::Start(HEADER + 1)).expect("inside first page");
15073        file.write_all(&[255]).expect("damage one byte");
15074        assert!(reader.read(0, &[0]).is_err(), "page checksum rejects corruption");
15075        fs::remove_file(damaged).expect("remove scratch file");
15076    }
15077
15078    #[test]
15079    fn damaged_lazy_dictionary_payload_is_an_error() {
15080        let path = path("damaged-dictionary");
15081        let mut writer = Writer::create(
15082            &path,
15083            "items",
15084            vec![
15085                Field::required("id", LogicalType::Integer),
15086                Field::new("text", LogicalType::Varchar),
15087            ],
15088        )
15089        .expect("new file");
15090        writer.append(&sample()).expect("stripe written");
15091        writer.finish().expect("commit");
15092
15093        let reader = Reader::open(&path).expect("valid directory");
15094        let dictionary = reader.table.dictionaries[1].expect("string dictionary page");
15095        // Read the count out of the page rather than writing it here, so that adding something
15096        // else to the index does not silently turn this into a test that damages the index.
15097        let mut header = [0; DICTIONARY_HEADER];
15098        read_at(&reader.file, dictionary.offset, &mut header).expect("dictionary header");
15099        // The first block's start is the first word after the offsets, since the blocks are written
15100        // during the load and are wherever the writer was when each was encoded.
15101        let count = u32::from_le_bytes(header[0..4].try_into().expect("four bytes")) as usize;
15102        let width = u32::from_le_bytes(header[12..16].try_into().expect("four bytes"));
15103        assert_ne!(width & DICTIONARY_SCATTERED, 0, "the blocks say where they are");
15104        let bits = (width & !DICTIONARY_FLAGS) as usize;
15105        let mut start = [0; 8];
15106        let at = dictionary.offset + (DICTIONARY_HEADER + offset_bytes(count, bits)) as u64;
15107        read_at(&reader.file, at, &mut start).expect("the first block's start");
15108        let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
15109        file.seek(SeekFrom::Start(u64::from_le_bytes(start))).expect("inside dictionary payload");
15110        file.write_all(&[255]).expect("damage dictionary payload");
15111
15112        let chunk = reader.read(0, &[1]).expect("code page and dictionary index remain valid");
15113        let error =
15114            chunk.validate_external().expect_err("payload corruption must reach the caller");
15115        assert!(error.message().contains("payload checksum differs"), "{error}");
15116        fs::remove_file(path).expect("remove scratch file");
15117    }
15118
15119    /// A column whose values are all different is written without a dictionary, and one whose
15120    /// values repeat keeps it.
15121    ///
15122    /// The two columns go in the same table and hold the same number of rows, so the only thing
15123    /// separating them is how much of the first stripe was a value it had not seen before. Both have
15124    /// to read back the values that were written, because the decision is about cost and nothing
15125    /// else. The file size is the other half of it: a column written without a dictionary goes
15126    /// through the string cascade instead, so dropping the dictionary must not turn into storing the
15127    /// column raw.
15128    #[test]
15129    fn a_column_of_all_different_values_is_written_without_a_dictionary() {
15130        let path = path("dictionary-decide");
15131        let rows = 20_000;
15132        // Long enough that storing it raw would show, and different in every row.
15133        let unique =
15134            |row: usize| format!("{row:09} a value that appears exactly once in the table");
15135        // The same values in the same shape, each one used forty times over.
15136        let repeated = |row: usize| unique(row / 40);
15137        let mut writer = Writer::create(
15138            &path,
15139            "items",
15140            vec![
15141                Field::required("unique", LogicalType::Varchar),
15142                Field::required("repeated", LogicalType::Varchar),
15143            ],
15144        )
15145        .expect("new file");
15146        for part in (0..rows).step_by(1_000) {
15147            let span = part..(part + 1_000).min(rows);
15148            let left = span.clone().map(|row| Value::Varchar(unique(row))).collect::<Vec<_>>();
15149            let right = span.map(|row| Value::Varchar(repeated(row))).collect::<Vec<_>>();
15150            writer
15151                .append(
15152                    &Chunk::new(vec![
15153                        Vector::from_values(LogicalType::Varchar, &left).expect("strings"),
15154                        Vector::from_values(LogicalType::Varchar, &right).expect("strings"),
15155                    ])
15156                    .expect("two columns"),
15157                )
15158                .expect("a part");
15159        }
15160        writer.finish().expect("commit");
15161
15162        let reader = Reader::open(&path).expect("reopen from disk");
15163        assert!(
15164            reader.table.dictionaries[0].is_none(),
15165            "a column with no repeats has nothing to say twice"
15166        );
15167        assert!(
15168            reader.table.dictionaries[1].is_some(),
15169            "a column whose values come round again keeps its dictionary"
15170        );
15171        let mut first = 0;
15172        for part in 0..reader.parts() {
15173            let chunk = reader.read(part, &[0, 1]).expect("a part");
15174            for row in 0..chunk.len() {
15175                assert_eq!(chunk.value_at(row, 0), Value::Varchar(unique(first + row)));
15176                assert_eq!(chunk.value_at(row, 1), Value::Varchar(repeated(first + row)));
15177            }
15178            first += chunk.len();
15179        }
15180        assert_eq!(first, rows, "every row was read back");
15181        let raw = (0..rows).map(|row| unique(row).len()).sum::<usize>();
15182        let size = fs::metadata(&path).expect("the file is there").len() as usize;
15183        assert!(size < raw, "a column without a dictionary is still encoded: {size} against {raw}");
15184        fs::remove_file(path).expect("remove scratch file");
15185    }
15186
15187    /// A payload of many blocks reads and checks every block of it.
15188    ///
15189    /// The test above has a dictionary of three values, which is one block, so it says nothing
15190    /// about a reader finding the right block among many. This one has thirty two thousand values,
15191    /// which is thirty two blocks, and it reads a value out of the first block and a value out of
15192    /// the last and then damages the last and asks for it again.
15193    ///
15194    /// Forty thousand rows over those thirty two thousand values, because a column the writer finds
15195    /// to be all distinct does not get a dictionary at all and there would be nothing here to test.
15196    /// Four rows in five holding a value the stripe has not seen before is a column that keeps one.
15197    /// The repeats are put at the front so that the values still arrive in order after them, which
15198    /// is what keeps the last part of the table on the last block of the payload.
15199    #[test]
15200    fn a_dictionary_over_many_blocks_checks_every_block_of_it() {
15201        let path = path("dictionary-blocks");
15202        let value = |row: usize| {
15203            let row = row.saturating_sub(8_000);
15204            format!("{row:07} a value long enough to be worth a payload block")
15205        };
15206        let parts = 40;
15207        let per_part = 1000;
15208        let mut writer =
15209            Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
15210                .expect("new file");
15211        for part in 0..parts {
15212            let values = (0..per_part)
15213                .map(|row| Value::Varchar(value(part * per_part + row)))
15214                .collect::<Vec<_>>();
15215            let chunk = Chunk::new(vec![
15216                Vector::from_values(LogicalType::Varchar, &values).expect("strings"),
15217            ])
15218            .expect("matching rows");
15219            writer.append(&chunk).expect("a part");
15220        }
15221        writer.finish().expect("commit");
15222
15223        let reader = Reader::open(&path).expect("reopen from disk");
15224        let dictionary = reader.table.dictionaries[0].expect("string dictionary page");
15225        assert!(
15226            parts * per_part > TEXT_PAYLOAD_VALUES * 4,
15227            "the dictionary has to be several blocks for this to be testing anything"
15228        );
15229        for part in [0, parts - 1] {
15230            let chunk = reader.read(part, &[0]).expect("a part");
15231            chunk.validate_external().expect("every payload block checks out");
15232            assert_eq!(chunk.value_at(0, 0), Value::Varchar(value(part * per_part)));
15233        }
15234
15235        // The last block is wherever the writer was when it was encoded, which the index says.
15236        let mut header = [0; DICTIONARY_HEADER];
15237        read_at(&reader.file, dictionary.offset, &mut header).expect("dictionary header");
15238        let count = u32::from_le_bytes(header[0..4].try_into().expect("four bytes")) as usize;
15239        let blocks = u32::from_le_bytes(header[8..12].try_into().expect("four bytes")) as usize;
15240        let width = u32::from_le_bytes(header[12..16].try_into().expect("four bytes"));
15241        let bits = (width & !DICTIONARY_FLAGS) as usize;
15242        let mut place = [0; 16];
15243        let at = DICTIONARY_HEADER + offset_bytes(count, bits) + (blocks - 1) * 16;
15244        read_at(&reader.file, dictionary.offset + at as u64, &mut place).expect("its place");
15245        let start = u64::from_le_bytes(place[..8].try_into().expect("eight bytes"));
15246        let length = u64::from_le_bytes(place[8..].try_into().expect("eight bytes"));
15247        let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
15248        file.seek(SeekFrom::Start(start + length - 4)).expect("the last bytes of the last block");
15249        file.write_all(&[255]).expect("damage the last payload block");
15250        let reader = Reader::open(&path).expect("the directory and the index are untouched");
15251        let chunk = reader.read(parts - 1, &[0]).expect("the code page remains valid");
15252        let error = chunk.validate_external().expect_err("the damage must reach the caller");
15253        assert!(error.message().contains("payload checksum differs"), "{error}");
15254        fs::remove_file(path).expect("remove scratch file");
15255    }
15256
15257    /// Values of different lengths read back where the offsets say they do.
15258    ///
15259    /// The offsets are packed at one width for the column, they are relative to the payload block a
15260    /// value lands in, and they go in runs of half a block, so there are two boundaries where the
15261    /// arithmetic could be off by one and neither shows up on values that are all the same length.
15262    /// This writes 5,000 values whose lengths cycle through a wide range and reads every one back,
15263    /// so the first value of a block, the last value of a run and the last value of a block are all
15264    /// covered several times over. An empty value is in the cycle because a zero length span is the
15265    /// case the reader short circuits.
15266    ///
15267    /// Six thousand rows over those 5,000 values, because a column the writer finds to be all
15268    /// distinct is written without a dictionary and then there are no packed offsets to be off by
15269    /// one in.
15270    #[test]
15271    fn values_of_different_lengths_read_back_out_of_packed_offsets() {
15272        let path = path("dictionary-offsets");
15273        let value = |row: usize| {
15274            let row = row % 5_000;
15275            if row % 511 == 3 { String::new() } else { "x".repeat(row % 97) + &format!("{row:05}") }
15276        };
15277        let rows = 6_000;
15278        let mut writer =
15279            Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
15280                .expect("new file");
15281        let values = (0..rows).map(|row| Value::Varchar(value(row))).collect::<Vec<_>>();
15282        for part in values.chunks(1_000) {
15283            let chunk =
15284                Chunk::new(vec![Vector::from_values(LogicalType::Varchar, part).expect("strings")])
15285                    .expect("matching rows");
15286            writer.append(&chunk).expect("a part");
15287        }
15288        writer.finish().expect("commit");
15289
15290        let reader = Reader::open(&path).expect("reopen from disk");
15291        assert!(
15292            rows > TEXT_PAYLOAD_VALUES * 4,
15293            "the dictionary has to be several blocks for this to be testing anything"
15294        );
15295        for part in 0..rows / 1_000 {
15296            let chunk = reader.read(part, &[0]).expect("a part");
15297            for row in 0..1_000 {
15298                let row = part * 1_000 + row;
15299                assert_eq!(
15300                    chunk.value_at(row % 1_000, 0),
15301                    Value::Varchar(value(row)),
15302                    "value {row}"
15303                );
15304            }
15305        }
15306        // The lengths a vector at a time, twice over, because the first pass is what makes the
15307        // table of ends worth building and the second is read out of the lengths worked out of it.
15308        for _ in 0..2 {
15309            for part in 0..rows / 1_000 {
15310                let chunk = reader.read(part, &[0]).expect("a part");
15311                let mut lens = vec![0_i64; 1_000];
15312                let column = chunk.column(0).expect("one column");
15313                assert!(column.try_bytes_lens(&mut lens).expect("lengths"), "a stored column");
15314                for (row, &len) in lens.iter().enumerate() {
15315                    let row = part * 1_000 + row;
15316                    assert_eq!(len as usize, value(row).len(), "the length of value {row}");
15317                }
15318            }
15319        }
15320        fs::remove_file(path).expect("remove scratch file");
15321    }
15322
15323    /// Lengths start again at every block, and ends that go backwards inside one give no table.
15324    #[test]
15325    fn lengths_restart_at_each_block_and_refuse_ends_that_go_backwards() {
15326        let mut ends: Vec<u32> = (1..=TEXT_PAYLOAD_VALUES as u32).map(|at| at * 2).collect();
15327        ends.extend([3, 3, 10]);
15328        let Some(Lengths::Narrow(lens)) = lengths_of(&ends) else { panic!("short ordered ends") };
15329        assert!(lens[..TEXT_PAYLOAD_VALUES].iter().all(|&len| len == 2));
15330        assert_eq!(&lens[TEXT_PAYLOAD_VALUES..], &[3, 0, 7]);
15331        // One value longer than sixteen bits keeps every length at four bytes.
15332        let long = [5, 70_005, 70_006];
15333        let Some(Lengths::Wide(lens)) = lengths_of(&long) else { panic!("long ordered ends") };
15334        assert_eq!(lens, [5, 70_000, 1]);
15335        let mut read = Vec::new();
15336        Lengths::Wide(lens).extend_at(&[1, 9, 0], &mut read);
15337        assert_eq!(read, [70_000, 0, 5], "a position past the end is no length");
15338        ends.push(9);
15339        assert!(lengths_of(&ends).is_none());
15340    }
15341
15342    /// Every worker of a scan wants the dictionary at the same moment and one of them fetches it.
15343    ///
15344    /// Asking a `OnceLock` whether it holds something answers the question a worker that already has
15345    /// the dictionary is asking and not the one a worker without it is asking, which is whether
15346    /// somebody is already on their way with it. Sixteen workers that all miss will all read the
15347    /// page, all verify it and all decode it, and fifteen will drop the result. Nothing about that
15348    /// is incorrect, which is why it went unnoticed, and it showed up as ClickBench 38 getting
15349    /// slower when the scan in front of it got faster and stopped staggering the arrivals.
15350    ///
15351    /// The barrier is what makes the test about that rather than about luck. Without it the first
15352    /// thread is usually finished before the last one starts and the count is one either way.
15353    #[test]
15354    fn a_global_dictionary_is_opened_once_however_many_workers_ask_at_once() {
15355        let path = path("dictionary-once");
15356        let parts = 8;
15357        let per_part = 500;
15358        let value =
15359            |row: usize| format!("{row:07} a value long enough to be worth a payload block");
15360        let mut writer =
15361            Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
15362                .expect("new file");
15363        for part in 0..parts {
15364            let values = (0..per_part)
15365                .map(|row| Value::Varchar(value(part * per_part + row)))
15366                .collect::<Vec<_>>();
15367            let chunk = Chunk::new(vec![
15368                Vector::from_values(LogicalType::Varchar, &values).expect("strings"),
15369            ])
15370            .expect("matching rows");
15371            writer.append(&chunk).expect("a part");
15372        }
15373        writer.finish().expect("commit");
15374
15375        let reader = Reader::open(&path).expect("reopen from disk");
15376        assert!(reader.table.dictionaries[0].is_some(), "the column has to have one to share");
15377        assert_eq!(reader.reads().dictionaries, 0, "opening the file does not open a dictionary");
15378
15379        let workers = 16;
15380        let gate = std::sync::Barrier::new(workers);
15381        std::thread::scope(|scope| {
15382            for worker in 0..workers {
15383                let reader = reader.clone();
15384                let gate = &gate;
15385                scope.spawn(move || {
15386                    gate.wait();
15387                    let chunk = reader.read(worker % parts, &[0]).expect("a part");
15388                    assert_eq!(
15389                        chunk.value_at(0, 0),
15390                        Value::Varchar(value((worker % parts) * per_part))
15391                    );
15392                });
15393            }
15394        });
15395
15396        assert_eq!(reader.reads().dictionaries, 1, "sixteen workers, one dictionary, one open");
15397        fs::remove_file(path).expect("remove scratch file");
15398    }
15399
15400    /// The sorted order sits outside the index the page checksum covers, because a query that
15401    /// never searches a dictionary should not read it, so it carries its own checksums and this is
15402    /// what says they are checked. A search that trusted a damaged order would give a wrong answer
15403    /// rather than a slow one.
15404    #[test]
15405    fn a_damaged_sorted_order_is_an_error() {
15406        let path = path("damaged-order");
15407        let mut writer = Writer::create(
15408            &path,
15409            "items",
15410            vec![
15411                Field::required("id", LogicalType::Integer),
15412                Field::new("text", LogicalType::Varchar),
15413            ],
15414        )
15415        .expect("new file");
15416        writer.append(&sample()).expect("stripe written");
15417        writer.finish().expect("commit");
15418
15419        let reader = Reader::open(&path).expect("valid directory");
15420        let page = reader.table.dictionaries[1].expect("string dictionary page");
15421        let mut header = [0; DICTIONARY_HEADER];
15422        read_at(&reader.file, page.offset, &mut header).expect("dictionary header");
15423        let index_len = dictionary_index_len(&header);
15424        let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
15425        file.seek(SeekFrom::Start(page.offset + index_len)).expect("the first head");
15426        file.write_all(&[255]).expect("damage the order");
15427
15428        let dictionary = reader.dictionary(1).expect("read").expect("a string column has one");
15429        let error = dictionary.compare_rank(0, b"anything").expect_err("a damaged order is caught");
15430        assert!(error.message().contains("rank checksum differs"), "{error}");
15431        fs::remove_file(path).expect("remove scratch file");
15432    }
15433
15434    /// Codes stay in first appearance order and the sorted order is written beside them, so a
15435    /// reader can put the values back in order without the writer having had to know them all
15436    /// before it handed out the first code.
15437    #[test]
15438    fn a_global_dictionary_carries_the_sorted_order_of_its_values() {
15439        // Chosen so the sort cannot be decided on the first eight bytes alone. Three values share
15440        // a nine byte prefix, one is a prefix of another, and one is empty.
15441        let spellings = ["overlong1z", "b", "", "overlong1a", "overlong", "ab", "a", "overlong1"];
15442        let path = path("dictionary-order");
15443        let mut writer =
15444            Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
15445                .expect("new file");
15446        writer
15447            .append(
15448                &Chunk::new(vec![
15449                    Vector::from_values(
15450                        LogicalType::Varchar,
15451                        &spellings.map(|text| Value::Varchar(text.into())),
15452                    )
15453                    .expect("strings"),
15454                ])
15455                .expect("one column"),
15456            )
15457            .expect("stripe written");
15458        writer.finish().expect("commit");
15459
15460        let reader = Reader::open(&path).expect("valid directory");
15461        let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
15462        let count = dictionary.ranks().expect("a v10 file stores one");
15463        assert_eq!(count, spellings.len(), "every distinct value has a rank");
15464        let order = (0..count)
15465            .map(|rank| dictionary.code_at_rank(rank).expect("a code"))
15466            .collect::<Vec<_>>();
15467        let mut seen = order.clone();
15468        seen.sort_unstable();
15469        assert_eq!(seen, (0..spellings.len() as u32).collect::<Vec<_>>(), "a permutation of codes");
15470
15471        let ranked = order
15472            .iter()
15473            .map(|&code| {
15474                dictionary.try_bytes_at(code as usize).expect("read").expect("a value").to_vec()
15475            })
15476            .collect::<Vec<_>>();
15477        let mut expected = spellings.map(|text| text.as_bytes().to_vec()).to_vec();
15478        expected.sort();
15479        assert_eq!(ranked, expected, "rank order is value order");
15480
15481        // What a search asks, on the values themselves rather than through a kernel, so that a
15482        // file whose heads disagree with its bytes is caught here rather than as a wrong answer.
15483        for (rank, value) in expected.iter().enumerate() {
15484            assert_eq!(
15485                dictionary.compare_rank(rank, value).expect("compare"),
15486                Ordering::Equal,
15487                "rank {rank} is its own value"
15488            );
15489            if rank > 0 {
15490                assert_eq!(
15491                    dictionary.compare_rank(rank - 1, value).expect("compare"),
15492                    Ordering::Less,
15493                    "rank {rank} follows the one before it"
15494                );
15495            }
15496        }
15497        fs::remove_file(path).expect("remove scratch file");
15498    }
15499
15500    /// Five text columns of different sizes close at the same time, and each comes back with its
15501    /// own values in its own order.
15502    ///
15503    /// The sizes differ so that the columns are taken in an order that is not the column order, and
15504    /// the values of each column are spelled with its number so that one column's page written in
15505    /// another's place would read back as the wrong strings rather than the right ones by chance.
15506    #[test]
15507    fn text_columns_closed_at_once_each_keep_their_own_dictionary() {
15508        let sizes = [300_usize, 5_000, 40, 2_000, 1_200];
15509        let path = path("dictionaries-at-once");
15510        let fields = (0..sizes.len())
15511            .map(|column| Field::new(format!("text{column}"), LogicalType::Varchar))
15512            .collect::<Vec<_>>();
15513        let mut writer = Writer::create(&path, "items", fields).expect("new file");
15514        let rows = 10_000_usize;
15515        for start in (0..rows).step_by(1_024) {
15516            let columns = sizes
15517                .iter()
15518                .enumerate()
15519                .map(|(column, &size)| {
15520                    let values = (start..(start + 1_024).min(rows))
15521                        .map(|row| Value::Varchar(format!("c{column}-{:05}", (row * 7919) % size)))
15522                        .collect::<Vec<_>>();
15523                    Vector::from_values(LogicalType::Varchar, &values).expect("strings")
15524                })
15525                .collect::<Vec<_>>();
15526            writer.append(&Chunk::new(columns).expect("five columns")).expect("stripe written");
15527        }
15528        writer.finish().expect("commit");
15529
15530        let reader = Reader::open(&path).expect("valid directory");
15531        for (column, &size) in sizes.iter().enumerate() {
15532            let dictionary =
15533                reader.dictionary(column).expect("read").expect("a string column has one");
15534            let count = dictionary.ranks().expect("a v10 file stores one");
15535            assert_eq!(count, size, "column {column} has its own distinct count");
15536            let ranked = (0..count)
15537                .map(|rank| {
15538                    let code = dictionary.code_at_rank(rank).expect("a code");
15539                    dictionary.try_bytes_at(code as usize).expect("read").expect("a value").to_vec()
15540                })
15541                .collect::<Vec<_>>();
15542            let expected = (0..size)
15543                .map(|value| format!("c{column}-{value:05}").into_bytes())
15544                .collect::<Vec<_>>();
15545            assert_eq!(ranked, expected, "column {column} ranks its own values in order");
15546        }
15547        fs::remove_file(path).expect("remove scratch file");
15548    }
15549
15550    /// A dictionary large enough to be decoded and sorted on several threads ranks the way one small
15551    /// enough for one thread does.
15552    ///
15553    /// Seventy thousand values over sixty nine blocks, in no order and each four times over so the
15554    /// column is worth a dictionary, written and ranked in the close.
15555    /// Some share a long prefix and some differ only in the last byte, so the buckets of the sort cut
15556    /// through runs of values that agree for a long way.
15557    #[test]
15558    fn a_large_dictionary_ranks_in_value_order() {
15559        let path = path("dictionary-large-rank");
15560        let value = |row: u64| {
15561            let mixed = row.wrapping_mul(0x9e37_79b9_7f4a_7c15) >> 40;
15562            match row % 3 {
15563                0 => format!("https://example.com/a/long/shared/path/{mixed:08}"),
15564                1 => format!("{mixed}"),
15565                _ => format!("x{}", row % 1000).repeat(1 + (row % 4) as usize) + &row.to_string(),
15566            }
15567        };
15568        let distinct = 70_000;
15569        let parts = 4 * distinct / 1000;
15570        let per_part = 1000;
15571        let mut writer =
15572            Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
15573                .expect("new file");
15574        for part in 0..parts {
15575            let values = (0..per_part)
15576                .map(|row| Value::Varchar(value((part * per_part + row) / 4)))
15577                .collect::<Vec<_>>();
15578            let chunk = Chunk::new(vec![
15579                Vector::from_values(LogicalType::Varchar, &values).expect("strings"),
15580            ])
15581            .expect("matching rows");
15582            writer.append(&chunk).expect("a part");
15583        }
15584        writer.finish().expect("commit");
15585
15586        let reader = Reader::open(&path).expect("reopen from disk");
15587        let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
15588        let count = dictionary.ranks().expect("a ranked dictionary");
15589        assert_eq!(count, distinct as usize, "every distinct value has a rank");
15590        assert!(count >= PARALLEL_SORT_MIN, "too few values to be sorted on more than one thread");
15591        let ranked = (0..count)
15592            .map(|rank| {
15593                let code = dictionary.code_at_rank(rank).expect("a code");
15594                dictionary.try_bytes_at(code as usize).expect("read").expect("a value").to_vec()
15595            })
15596            .collect::<Vec<_>>();
15597        let mut expected = (0..distinct).map(|row| value(row).into_bytes()).collect::<Vec<_>>();
15598        expected.sort();
15599        assert_eq!(ranked, expected, "rank order is value order");
15600        fs::remove_file(path).expect("remove scratch file");
15601    }
15602
15603    /// A string column's synopsis is turned into values without keeping the blocks it went through.
15604    ///
15605    /// Three thousand values, every fifth of them four times over, so the synopsis is a prefix of
15606    /// five hundred and twelve codes spread over all three payload blocks. Reading it used to leave
15607    /// all three decoded for as long as the reader lived. It leaves none of them now, and the second
15608    /// read answers out of what the first remembered.
15609    /// A directory read out of the file a window at a time is the directory read whole.
15610    ///
15611    /// The windows here are far smaller than any field is long, so every kind of field is split
15612    /// across a refill somewhere, and a bound is offered to its codec short more than once. The
15613    /// synopses are left in the file, and each one read back from where it was left is the one the
15614    /// whole read decoded.
15615    #[test]
15616    fn a_directory_read_a_window_at_a_time_is_the_directory_read_whole() {
15617        let path = path("windowed-directory");
15618        let fields = vec![
15619            Field::required("id", LogicalType::BigInt),
15620            Field::required("word", LogicalType::Varchar),
15621            Field::new("score", LogicalType::Double),
15622        ];
15623        let mut writer = Writer::create(&path, "items", fields).expect("new file");
15624        for part in 0..70_i64 {
15625            let ids = (0..100).map(|row| Value::BigInt(part * 100 + row % 7)).collect::<Vec<_>>();
15626            let words = (0..100)
15627                .map(|row| Value::Varchar(format!("word {}", row % 13)))
15628                .collect::<Vec<_>>();
15629            let scores = (0..100)
15630                .map(|row| if row % 4 == 0 { Value::Null } else { Value::Double(row as f64) })
15631                .collect::<Vec<_>>();
15632            let chunk = Chunk::new(vec![
15633                Vector::from_values(LogicalType::BigInt, &ids).expect("integers"),
15634                Vector::from_values(LogicalType::Varchar, &words).expect("strings"),
15635                Vector::from_values(LogicalType::Double, &scores).expect("doubles"),
15636            ])
15637            .expect("three columns");
15638            writer.append(&chunk).expect("a part");
15639        }
15640        writer.finish().expect("commit");
15641
15642        let catalog = Catalog::open(&path).expect("reopen");
15643        let entry = catalog.entries.first().expect("one table").directory;
15644        let (offset, length) = (entry.offset, entry.length as usize);
15645        let mut bytes = vec![0; length];
15646        read_at(&catalog.file, offset, &mut bytes).expect("the directory");
15647        assert_eq!(file_checksum(&catalog.file, offset, length).expect("checksum"), entry.hash);
15648        let whole = decode_directory(&bytes, catalog.size).expect("whole");
15649        assert!(whole.stripes.len() > 1, "the table should span stripes");
15650        for size in [1, 7, 33, 4_096] {
15651            let mut cursor = Cursor::over(&catalog.file, offset, length);
15652            cursor.window.as_mut().expect("a window").size = size;
15653            let windowed = read_directory(cursor, catalog.size, Some(offset)).expect("windowed");
15654            assert_eq!(format!("{:?}", windowed.stripes), format!("{:?}", whole.stripes));
15655            assert_eq!(format!("{:?}", windowed.fields), format!("{:?}", whole.fields));
15656            let mut stored = 0;
15657            for (column, (left, held)) in
15658                windowed.frequencies.iter().zip(&whole.frequencies).enumerate()
15659            {
15660                match (left, held) {
15661                    (None, None) => {}
15662                    (
15663                        Some(super::Frequencies::Stored { span, values }),
15664                        Some(super::Frequencies::Held(summary)),
15665                    ) => {
15666                        let mut one = vec![0; span.length as usize];
15667                        read_at(&catalog.file, span.offset, &mut one).expect("a synopsis");
15668                        let read = decode_summary(
15669                            &mut Cursor::new(&one),
15670                            &whole.fields[column],
15671                            whole.rows,
15672                            *values,
15673                        )
15674                        .expect("a valid synopsis")
15675                        .expect("one is there");
15676                        assert_eq!(format!("{read:?}"), format!("{summary:?}"));
15677                        stored += 1;
15678                    }
15679                    other => panic!("column {column} came back as {other:?}"),
15680                }
15681            }
15682            assert!(stored >= 2, "only {stored} synopses were left in the file");
15683        }
15684        let reader = catalog.table("items").expect("the table");
15685        assert!(reader.frequency_summaries[1].get().is_none());
15686        assert!(reader.top_frequencies(1, 1).expect("a readable synopsis").is_some());
15687        let first = reader.frequency_summaries[1].get().expect("decoded synopsis");
15688        let clone = reader.clone();
15689        assert!(clone.top_frequencies(1, 1).expect("cached synopsis").is_some());
15690        assert!(Arc::ptr_eq(first, clone.frequency_summaries[1].get().expect("same synopsis")));
15691        fs::remove_file(path).expect("remove scratch file");
15692    }
15693
15694    #[test]
15695    fn a_checksum_carried_across_reads_is_the_checksum_of_the_whole() {
15696        let path = path("file-checksum");
15697        let bytes = (0..200_000_u32)
15698            .map(|at| (at.wrapping_mul(2_654_435_761) >> 13) as u8)
15699            .collect::<Vec<_>>();
15700        fs::write(&path, &bytes).expect("scratch file");
15701        let file = File::open(&path).expect("open");
15702        for (offset, length) in [
15703            (0, 0),
15704            (3, 1),
15705            (5, 31),
15706            (0, 32),
15707            (9, 33),
15708            (1, 65_536),
15709            (7, 65_567),
15710            (0, 200_000),
15711            (11, 131_101),
15712        ] {
15713            let whole = checksum(&bytes[offset..offset + length]);
15714            assert_eq!(
15715                file_checksum(&file, offset as u64, length).expect("read"),
15716                whole,
15717                "{offset} {length}"
15718            );
15719        }
15720        fs::remove_file(path).expect("remove scratch file");
15721    }
15722
15723    #[test]
15724    fn a_string_synopsis_is_read_without_keeping_the_dictionary_blocks() {
15725        let path = path("synopsis-keeps-no-block");
15726        let spelled = |index: usize| Value::Varchar(format!("phrase {index:05}"));
15727        let mut values = (0..3_000).map(spelled).collect::<Vec<_>>();
15728        for _ in 0..3 {
15729            values.extend((0..3_000).step_by(5).map(spelled));
15730        }
15731        let mut writer =
15732            Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
15733                .expect("new file");
15734        for part in values.chunks(1_024) {
15735            writer
15736                .append(
15737                    &Chunk::new(vec![
15738                        Vector::from_values(LogicalType::Varchar, part).expect("strings"),
15739                    ])
15740                    .expect("one column"),
15741                )
15742                .expect("a part");
15743        }
15744        writer.finish().expect("commit");
15745
15746        let reader = Reader::open(&path).expect("reopen from disk");
15747        let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
15748        let resting = dictionary.footprint();
15749        let prefix = reader.frequency_prefix(0).expect("a readable synopsis").expect("one");
15750        assert_eq!(prefix.entries.len(), 512);
15751        for (value, count) in &prefix.entries {
15752            let Value::Varchar(text) = value else { panic!("a string column gave {value:?}") };
15753            let index = text["phrase ".len()..].parse::<usize>().expect("a spelled number");
15754            assert_eq!((index % 5, *count), (0, 4), "{text} came back with {count}");
15755        }
15756        assert_eq!(dictionary.footprint(), resting, "reading the synopsis kept a decoded block");
15757        let again = reader.frequency_prefix(0).expect("a readable synopsis").expect("one");
15758        assert_eq!(again.entries, prefix.entries);
15759        fs::remove_file(path).expect("remove scratch file");
15760    }
15761
15762    /// `length` over a stored column keeps a count a value rather than the blocks it counted.
15763    ///
15764    /// Reading the bytes a row at a time keeps every block it touches, so a scan of `length` over a
15765    /// whole column used to end up holding the column decoded. The counts are what is kept now, and
15766    /// they have to be the counts of characters rather than bytes, which is why the values here are
15767    /// not ASCII.
15768    #[test]
15769    fn character_lengths_are_counted_without_keeping_the_dictionary_blocks() {
15770        let path = path("character-lengths");
15771        let spellings = (0..2_500)
15772            .map(|index| Value::Varchar(format!("héllo {index:05} {}", "ü".repeat(index % 30))))
15773            .collect::<Vec<_>>();
15774        let mut writer =
15775            Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
15776                .expect("new file");
15777        for part in spellings.chunks(1_024) {
15778            writer
15779                .append(
15780                    &Chunk::new(vec![
15781                        Vector::from_values(LogicalType::Varchar, part).expect("strings"),
15782                    ])
15783                    .expect("one column"),
15784                )
15785                .expect("a part");
15786        }
15787        writer.finish().expect("commit");
15788
15789        let reader = Reader::open(&path).expect("reopen from disk");
15790        let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
15791        let resting = dictionary.footprint();
15792        let mut lens = Vec::new();
15793        assert!(dictionary.try_chars_lens(&mut lens).expect("counted"), "a stored source counts");
15794        let counted = dictionary.footprint() - resting;
15795        let blocks = dictionary.len().div_ceil(TEXT_PAYLOAD_VALUES);
15796        assert!(
15797            counted <= blocks * TEXT_PAYLOAD_VALUES * size_of::<u32>(),
15798            "counting kept {counted} bytes, more than a count a value"
15799        );
15800        let expected = (0..dictionary.len())
15801            .map(|code| {
15802                let bytes = dictionary.try_bytes_at(code).expect("read").expect("a value");
15803                i64::try_from(std::str::from_utf8(bytes).expect("utf-8").chars().count())
15804                    .expect("small")
15805            })
15806            .collect::<Vec<_>>();
15807        assert_eq!(lens, expected, "a count is the number of characters, not of bytes");
15808        let mut again = Vec::new();
15809        assert!(dictionary.try_chars_lens(&mut again).expect("counted"));
15810        assert_eq!(again, lens, "the kept counts answer the second time");
15811        fs::remove_file(path).expect("remove scratch file");
15812    }
15813
15814    /// Writes one column of strings whose code is where they sit in `spellings`, and reopens it.
15815    fn stored_spellings(label: &str, spellings: &[String]) -> (PathBuf, Reader) {
15816        let path = path(label);
15817        let values = spellings.iter().map(|text| Value::Varchar(text.clone())).collect::<Vec<_>>();
15818        let mut writer =
15819            Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
15820                .expect("new file");
15821        for part in values.chunks(1_024) {
15822            writer
15823                .append(
15824                    &Chunk::new(vec![
15825                        Vector::from_values(LogicalType::Varchar, part).expect("strings"),
15826                    ])
15827                    .expect("one column"),
15828                )
15829                .expect("a part");
15830        }
15831        writer.finish().expect("commit");
15832        let reader = Reader::open(&path).expect("reopen from disk");
15833        (path, reader)
15834    }
15835
15836    /// Codes that go all over a dictionary of `len` values, and every seventh row null.
15837    ///
15838    /// The shape of a vector a scan hands out: its codes are in row order, which lands them in
15839    /// every block of the dictionary in no order at all, so a read of the whole vector has to put
15840    /// them in block order itself to read each block once.
15841    fn scattered_rows(len: usize) -> (Vec<u32>, Vec<bool>) {
15842        let codes = (0..len)
15843            .map(|row| u32::try_from(row * 7_919 % len).expect("a small dictionary"))
15844            .collect::<Vec<_>>();
15845        let valid = (0..len).map(|row| row % 7 != 3).collect::<Vec<_>>();
15846        (codes, valid)
15847    }
15848
15849    /// `length` over a vector with nulls keeps the counts and not the blocks, the same as over one
15850    /// without.
15851    ///
15852    /// The whole vector count used to be taken only when no row was null, and every other vector
15853    /// went a row at a time through the bytes, which keeps every block it reads. A column with a
15854    /// null in each vector was held decoded after one `length` over it.
15855    #[test]
15856    fn character_lengths_with_nulls_are_counted_without_keeping_the_dictionary_blocks() {
15857        let spellings = (0..2_500)
15858            .map(|index| format!("héllo {index:05} {}", "ü".repeat(index % 30)))
15859            .collect::<Vec<_>>();
15860        let (path, reader) = stored_spellings("character-lengths-nulls", &spellings);
15861        let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
15862        let (codes, valid) = scattered_rows(spellings.len());
15863        let rows = Vector::dictionary_over(codes.clone(), Arc::clone(&dictionary))
15864            .expect("every code is inside")
15865            .with_validity(Validity::from_run(&valid));
15866
15867        let resting = dictionary.footprint();
15868        let lens = rudb_kernels::call("length", &[&rows], &LogicalType::BigInt, None)
15869            .expect("length reads");
15870        let counted = dictionary.footprint() - resting;
15871        let blocks = dictionary.len().div_ceil(TEXT_PAYLOAD_VALUES);
15872        assert!(
15873            counted <= blocks * TEXT_PAYLOAD_VALUES * size_of::<u32>(),
15874            "length over a vector with nulls kept {counted} bytes, more than a count a value"
15875        );
15876        let expected = (0..rows.len())
15877            .map(|row| match valid[row] {
15878                true => Value::BigInt(
15879                    i64::try_from(spellings[codes[row] as usize].chars().count()).expect("small"),
15880                ),
15881                false => Value::Null,
15882            })
15883            .collect::<Vec<_>>();
15884        let answers = (0..lens.len()).map(|row| lens.value_at(row)).collect::<Vec<_>>();
15885        assert_eq!(answers, expected, "a count of characters where a row has one, null elsewhere");
15886        fs::remove_file(path).expect("remove scratch file");
15887    }
15888
15889    /// `lower`, `upper` and `substring` read a stored dictionary a block at a time and keep none of
15890    /// it while the column is at its budget, until reading without keeping stops being cheap.
15891    ///
15892    /// The three used to read a row at a time through the bytes, which keeps every block a row lands
15893    /// in for as long as the table is open. They read the whole vector in one visit now, and the
15894    /// dictionary here is opened with a budget of zero so that what a visit would keep under the
15895    /// budget of a running database is what the test sees dropped. After a column's worth of blocks
15896    /// has been decoded and dropped the visit keeps what it reads, which is what bounds its cost on
15897    /// a scan whose codes keep coming back to every block, and the end of the test holds it to that.
15898    #[test]
15899    fn string_kernels_read_a_stored_dictionary_without_keeping_its_blocks() {
15900        let spellings = (0..2_500)
15901            .map(|index| format!("HéLLo {index:05} {}", "Üß".repeat(index % 30)))
15902            .collect::<Vec<_>>();
15903        let (path, reader) = stored_spellings("string-kernels", &spellings);
15904        let page = reader.table.dictionaries[0].expect("a string column has one");
15905        let starved =
15906            open_global_dictionary(Arc::clone(&reader.file), page, &LogicalType::Varchar, 0)
15907                .expect("a dictionary opens whatever it may keep");
15908        let starved = Arc::new(starved);
15909        let (codes, valid) = scattered_rows(spellings.len());
15910        let rows = Vector::dictionary_over(codes.clone(), Arc::clone(&starved))
15911            .expect("every code is inside")
15912            .with_validity(Validity::from_run(&valid));
15913        let expected = |each: &dyn Fn(&str) -> String| {
15914            (0..rows.len())
15915                .map(|row| match valid[row] {
15916                    true => Value::Varchar(each(&spellings[codes[row] as usize])),
15917                    false => Value::Null,
15918                })
15919                .collect::<Vec<_>>()
15920        };
15921        let answers =
15922            |vector: &Vector| (0..vector.len()).map(|row| vector.value_at(row)).collect::<Vec<_>>();
15923
15924        // What a visit may add is the table of where every value ends, four bytes a value, which
15925        // reading every value this often makes worth building. A block is tens of bytes a value.
15926        let resting = starved.footprint();
15927        let ends = spellings.len() * size_of::<u32>();
15928        let lowered = rudb_kernels::call("lower", &[&rows], &LogicalType::Varchar, None)
15929            .expect("lower reads");
15930        assert_eq!(answers(&lowered), expected(&|text| text.to_lowercase()), "lower");
15931        assert!(starved.footprint() <= resting + ends, "lower kept a block it read");
15932
15933        let start = Vector::constant(LogicalType::BigInt, Value::BigInt(3), rows.len());
15934        let length = Vector::constant(LogicalType::BigInt, Value::BigInt(9), rows.len());
15935        let cut =
15936            rudb_kernels::call("substring", &[&rows, &start, &length], &LogicalType::Varchar, None)
15937                .expect("substring reads");
15938        let cut_of = |text: &str| text.chars().skip(2).take(9).collect::<String>();
15939        assert_eq!(answers(&cut), expected(&cut_of), "substring");
15940        assert!(starved.footprint() <= resting + ends, "substring kept a block it read");
15941
15942        // Every block has been read twice now and dropped the second time as well, which is a
15943        // column's worth dropped for want of a budget, so the next visit keeps what it reads.
15944        let raised = rudb_kernels::call("upper", &[&rows], &LogicalType::Varchar, None)
15945            .expect("upper reads");
15946        assert_eq!(answers(&raised), expected(&|text| text.to_uppercase()), "upper");
15947        let payload = spellings.iter().map(String::len).sum::<usize>();
15948        assert!(
15949            starved.footprint() >= resting + payload,
15950            "a visit that has dropped a column's worth of blocks keeps what it reads"
15951        );
15952        let again = rudb_kernels::call("upper", &[&rows], &LogicalType::Varchar, None)
15953            .expect("upper reads kept blocks");
15954        assert_eq!(answers(&again), answers(&raised), "the kept blocks answer the same");
15955        fs::remove_file(path).expect("remove scratch file");
15956    }
15957
15958    /// A sweep of the dictionary reads every value, and the second sweep keeps what it read, up to
15959    /// the budget.
15960    ///
15961    /// The point of the sweep is the resident size rather than the answer, so both are checked
15962    /// here. The first sweep keeps nothing, because a process that runs one statement never reads
15963    /// a block twice. A dictionary this small is well under [`TEXT_KEEP_BUDGET`], so the second
15964    /// sweep keeps everything and a third decodes nothing, which is what makes a session asking the
15965    /// same question again cost what it should. The ceiling is the other half of it and it has its own
15966    /// test below, because a ceiling that never binds is not a ceiling anybody checked.
15967    #[test]
15968    fn a_dictionary_sweep_reads_every_value_and_keeps_it_under_the_budget() {
15969        let path = path("dictionary-sweep");
15970        // Two thousand five hundred distinct values is two whole payload blocks and a part of a
15971        // third, so the sweep has to be called more than once and the last call has to stop short.
15972        let spellings = (0..2_500)
15973            .map(|index| Value::Varchar(format!("value {index:08} {}", "x".repeat(index % 40))))
15974            .collect::<Vec<_>>();
15975        let mut writer =
15976            Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
15977                .expect("new file");
15978        // A chunk is a part and a part is at most 1,024 rows, so the values go in three of them.
15979        // The dictionary is table wide and does not care where a value was written.
15980        for part in spellings.chunks(1_024) {
15981            writer
15982                .append(
15983                    &Chunk::new(vec![
15984                        Vector::from_values(LogicalType::Varchar, part).expect("strings"),
15985                    ])
15986                    .expect("one column"),
15987                )
15988                .expect("stripe written");
15989        }
15990        writer.finish().expect("commit");
15991
15992        let reader = Reader::open(&path).expect("valid directory");
15993        let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
15994        assert_eq!(dictionary.len(), spellings.len(), "every value is distinct");
15995        for first in [0, TEXT_PAYLOAD_VALUES, TEXT_PAYLOAD_VALUES * 2] {
15996            assert!(dictionary.text_block_might_contain(first, b"value").expect("signature"));
15997            assert!(!dictionary.text_block_might_contain(first, b"google").expect("signature"));
15998        }
15999
16000        let resting = dictionary.footprint();
16001        let sweep = || {
16002            let mut swept: Vec<Vec<u8>> = Vec::new();
16003            let mut at = 0;
16004            let mut calls = 0;
16005            while at < dictionary.len() {
16006                let stopped = dictionary
16007                    .sweep_text(at, dictionary.len(), &mut |index: usize, text: &[u8]| {
16008                        assert_eq!(index, swept.len(), "a sweep hands its values over in order");
16009                        swept.push(text.to_vec());
16010                        Ok(())
16011                    })
16012                    .expect("a sweep reads");
16013                assert!(stopped > at, "a sweep moves");
16014                at = stopped;
16015                calls += 1;
16016            }
16017            assert_eq!(calls, 3, "a sweep hands over one block at a time");
16018            swept
16019        };
16020        let swept = sweep();
16021        assert_eq!(dictionary.footprint(), resting, "a first sweep keeps nothing it decoded");
16022        assert_eq!(sweep(), swept, "a second sweep reads what the first did");
16023        let after = dictionary.footprint();
16024        assert!(after > resting, "a second sweep under the budget keeps what it decoded");
16025
16026        let read = (0..dictionary.len())
16027            .map(|code| dictionary.try_bytes_at(code).expect("read").expect("a value").to_vec())
16028            .collect::<Vec<_>>();
16029        assert_eq!(swept, read, "a sweep answers what a point read answers");
16030        // A read per value is about what makes the unpacked ends worth building, so whether they
16031        // are built here depends on how many reads the sweep made on the way. They are the one thing
16032        // allowed to grow, by four bytes a value, and nothing of the payload is.
16033        let grown = dictionary.footprint() - after;
16034        assert!(
16035            grown == 0 || grown == dictionary.len() * size_of::<u32>(),
16036            "a point read of a kept block decodes nothing, and {grown} bytes grew"
16037        );
16038        fs::remove_file(path).expect("remove scratch file");
16039    }
16040
16041    #[test]
16042    fn a_narrow_signature_of_an_older_file_answers_by_its_own_width() {
16043        let path = path("narrow-substring-signature");
16044        let blocks = [&b"https://google.com/"[..], b"https://example.org/", b"mail.google.com"];
16045        let mut grams = Vec::new();
16046        for text in blocks {
16047            let mut bits = vec![0_u8; NARROW_GRAM_BYTES];
16048            for gram in text.windows(4) {
16049                for bit in gram_bits(gram, NARROW_GRAM_BYTES) {
16050                    bits[bit / 8] |= 1 << (bit % 8);
16051                }
16052            }
16053            grams.extend(bits);
16054        }
16055        fs::write(&path, &grams).expect("scratch file");
16056        let file = File::open(&path).expect("open scratch file");
16057        let signatures = NativeGrams {
16058            start: 0,
16059            length: grams.len(),
16060            width: NARROW_GRAM_BYTES,
16061            hash: checksum(&grams),
16062            verdicts: Mutex::new(Vec::new()),
16063        };
16064        let verdict = signatures.verdicts(&file, b"google").expect("signatures read");
16065        assert_eq!(&verdict[..], &[true, false, true], "one verdict a block, at the narrow width");
16066        assert!(signatures.footprint() > 0, "a verdict is remembered");
16067        let again = signatures.verdicts(&file, b"google").expect("remembered");
16068        assert!(Arc::ptr_eq(&verdict, &again), "a second question about a literal reads nothing");
16069
16070        let damaged = NativeGrams {
16071            hash: signatures.hash ^ 1,
16072            verdicts: Mutex::new(Vec::new()),
16073            ..signatures
16074        };
16075        let error = damaged.verdicts(&file, b"google").expect_err("a damaged region is refused");
16076        assert!(error.to_string().contains("substring signatures checksum differs"), "{error}");
16077        fs::remove_file(path).expect("remove scratch file");
16078    }
16079
16080    #[test]
16081    fn a_damaged_substring_signature_is_checked_only_when_used() {
16082        let path = path("damaged-substring-signature");
16083        let mut writer =
16084            Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
16085                .expect("new file");
16086        let rows = [Value::Varchar("google".into()), Value::Varchar("example".into())];
16087        writer
16088            .append(
16089                &Chunk::new(vec![
16090                    Vector::from_values(LogicalType::Varchar, &rows).expect("strings"),
16091                ])
16092                .expect("one column"),
16093            )
16094            .expect("stripe written");
16095        writer.finish().expect("commit");
16096
16097        let reader = Reader::open(&path).expect("valid directory");
16098        let page = reader.table.dictionaries[0].expect("string dictionary page");
16099        let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
16100        file.seek(SeekFrom::Start(page.offset + u64::from(page.length) - 1))
16101            .expect("last signature byte");
16102        file.write_all(&[255]).expect("damage signature");
16103        let reader = Reader::open(&path).expect("the directory is still valid");
16104        let dictionary = reader.dictionary(0).expect("index is still valid").expect("dictionary");
16105        let error = dictionary
16106            .text_block_might_contain(0, b"goog")
16107            .expect_err("a used signature checks its own checksum");
16108        assert!(error.message().contains("substring signatures checksum differs"), "{error}");
16109        fs::remove_file(path).expect("remove scratch file");
16110    }
16111
16112    /// A sweep over a block whose second run of offsets is short reads the same values as a point
16113    /// read does.
16114    ///
16115    /// The sweep decodes the offsets of a whole run at a time rather than a value at a time, and a
16116    /// run holds half a block, so the count it asks for is the run length everywhere but at the end
16117    /// of the dictionary. Two thousand five hundred values, which is what the test above writes,
16118    /// never puts a short run second in its block: the last block there begins on a run boundary and
16119    /// holds one run. Two thousand eight hundred does, so the last block is a whole run of five
16120    /// hundred and twelve followed by two hundred and forty, and an off by one in either the count
16121    /// asked for or the slice taken out of the answer shows up as a wrong value or a refusal.
16122    #[test]
16123    fn a_sweep_over_a_block_with_a_short_second_run_reads_what_a_point_read_reads() {
16124        let path = path("dictionary-sweep-short-run");
16125        let spellings = (0..2_800)
16126            .map(|index| Value::Varchar(format!("value {index:08} {}", "x".repeat(index % 40))))
16127            .collect::<Vec<_>>();
16128        let mut writer =
16129            Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
16130                .expect("new file");
16131        for part in spellings.chunks(1_024) {
16132            writer
16133                .append(
16134                    &Chunk::new(vec![
16135                        Vector::from_values(LogicalType::Varchar, part).expect("strings"),
16136                    ])
16137                    .expect("one column"),
16138                )
16139                .expect("stripe written");
16140        }
16141        writer.finish().expect("commit");
16142
16143        let reader = Reader::open(&path).expect("valid directory");
16144        let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
16145        assert_eq!(dictionary.len(), spellings.len(), "every value is distinct");
16146        let last = dictionary.len() % TEXT_PAYLOAD_VALUES;
16147        assert!(last > TEXT_OFFSET_RUN, "the last block has to reach into a second run of offsets");
16148        assert!(last < TEXT_PAYLOAD_VALUES, "and that second run has to be short of a whole one");
16149
16150        let mut swept: Vec<Vec<u8>> = Vec::new();
16151        let mut at = 0;
16152        while at < dictionary.len() {
16153            let stopped = dictionary
16154                .sweep_text(at, dictionary.len(), &mut |index: usize, text: &[u8]| {
16155                    assert_eq!(index, swept.len(), "a sweep hands its values over in order");
16156                    swept.push(text.to_vec());
16157                    Ok(())
16158                })
16159                .expect("a sweep reads");
16160            assert!(stopped > at, "a sweep moves");
16161            at = stopped;
16162        }
16163        let read = (0..dictionary.len())
16164            .map(|code| dictionary.try_bytes_at(code).expect("read").expect("a value").to_vec())
16165            .collect::<Vec<_>>();
16166        assert_eq!(swept, read, "a sweep answers what a point read answers");
16167        fs::remove_file(path).expect("remove scratch file");
16168    }
16169
16170    /// The unpacked ends answer what the packed ends answer, on both sides of the switch.
16171    ///
16172    /// A column asked for one offset at a time reads them out of the packed form until the reads
16173    /// are worth a table and out of the table after that, so every value here is read twice and the
16174    /// two passes are compared against the spellings and against each other. Two thousand eight
16175    /// hundred values is two payload blocks and a bit, which puts the switch in the middle of the
16176    /// first pass and means the pass straddles a block boundary, where the start of a value is zero
16177    /// rather than the end of the value before it.
16178    #[test]
16179    fn the_unpacked_ends_answer_what_the_packed_ends_answer() {
16180        let path = path("dictionary-unpacked-ends");
16181        let spellings = (0..2_800)
16182            .map(|index| Value::Varchar(format!("value {index:08} {}", "x".repeat(index % 40))))
16183            .collect::<Vec<_>>();
16184        let mut writer =
16185            Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
16186                .expect("new file");
16187        for part in spellings.chunks(1_024) {
16188            writer
16189                .append(
16190                    &Chunk::new(vec![
16191                        Vector::from_values(LogicalType::Varchar, part).expect("strings"),
16192                    ])
16193                    .expect("one column"),
16194                )
16195                .expect("stripe written");
16196        }
16197        writer.finish().expect("commit");
16198
16199        let reader = Reader::open(&path).expect("valid directory");
16200        let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
16201        assert_eq!(dictionary.len(), spellings.len(), "every value is distinct");
16202        let wanted = (0..spellings.len())
16203            .map(|index| format!("value {index:08} {}", "x".repeat(index % 40)).into_bytes())
16204            .collect::<Vec<_>>();
16205
16206        let pass = |what: &str| {
16207            for (index, value) in wanted.iter().enumerate() {
16208                let len = dictionary.try_bytes_len_at(index).expect("read").expect("a value");
16209                assert_eq!(len, value.len(), "{what} has the wrong length at {index}");
16210                let bytes = dictionary.try_bytes_at(index).expect("read").expect("a value");
16211                assert_eq!(bytes, value.as_slice(), "{what} has the wrong value at {index}");
16212            }
16213        };
16214        pass("the first pass");
16215        pass("the second pass");
16216
16217        // The whole vector in one call, over the text and through codes into it, which is how a
16218        // scan of a stored column hands it out. The codes run backwards and repeat so that they are
16219        // neither the positions nor in order.
16220        let lens = wanted.iter().map(|value| value.len() as i64).collect::<Vec<_>>();
16221        let mut whole = vec![0i64; wanted.len()];
16222        assert!(dictionary.try_bytes_lens(&mut whole).expect("read"), "the text answers whole");
16223        assert_eq!(whole, lens, "a vector of lengths answers what a length at a time answers");
16224        let codes = (0..4_000_u32).map(|row| (7 * (4_000 - row)) % 2_800).collect::<Vec<_>>();
16225        let coded = Vector::dictionary_over(codes.clone(), dictionary).expect("codes in range");
16226        let mut through = vec![0i64; codes.len()];
16227        assert!(coded.try_bytes_lens(&mut through).expect("read"), "the codes answer whole");
16228        for (row, &code) in codes.iter().enumerate() {
16229            assert_eq!(through[row], lens[code as usize], "row {row} reads code {code}");
16230            let one = coded.try_bytes_len_at(row).expect("read").expect("a value");
16231            assert_eq!(through[row], one as i64, "row {row} a row at a time");
16232        }
16233
16234        // A handful of codes over a column nobody has read yet is short of the table, so the same
16235        // call answers out of the packed ends instead, and has to answer the same.
16236        let fresh = Reader::open(&path).expect("valid directory");
16237        let untouched = fresh.dictionary(0).expect("read").expect("a string column has one");
16238        let few = vec![2_799_u32, 0, 1_024, 1_023, 511, 512];
16239        let coded = Vector::dictionary_over(few.clone(), untouched).expect("in range");
16240        let mut short = vec![0i64; few.len()];
16241        assert!(coded.try_bytes_lens(&mut short).expect("read"), "the codes answer whole");
16242        let expected = few.iter().map(|&code| lens[code as usize]).collect::<Vec<_>>();
16243        assert_eq!(short, expected, "the packed ends answer what the table answers");
16244        fs::remove_file(path).expect("remove scratch file");
16245    }
16246
16247    /// Narrowing a page takes what fits and refuses the page for anything that does not.
16248    ///
16249    /// The edges of the range on both sides and one step past each of them, for every type, because
16250    /// checking a page separately from converting it is only right if the check refuses exactly what
16251    /// `TryFrom` would have refused, and off by one there is a file that reads back a different
16252    /// number than it was given. The check is a bit pattern rather than a comparison, so it is not
16253    /// the shape a reader would guess from the bounds, which is why all six are here. The empty page
16254    /// is here because a check written the obvious way starts with the extremes the wrong way round
16255    /// and refuses it.
16256    #[test]
16257    fn narrowing_a_page_takes_what_fits_and_refuses_what_does_not() {
16258        assert_eq!(fit::<i8>(&[]).expect("an empty page fits anything"), Vec::<i8>::new());
16259        assert_eq!(fit::<i8>(&[-128, 0, 127]).expect("the edges fit"), vec![-128_i8, 0, 127]);
16260        fit::<i8>(&[128]).expect_err("one past the top does not fit");
16261        fit::<i8>(&[-129]).expect_err("one past the bottom does not fit");
16262        assert_eq!(fit::<u8>(&[0, 255]).expect("the edges fit"), vec![0_u8, 255]);
16263        fit::<u8>(&[256]).expect_err("one past the top does not fit");
16264        fit::<u8>(&[-1]).expect_err("a negative does not fit an unsigned page");
16265        assert_eq!(
16266            fit::<i16>(&[-32_768, 0, 32_767]).expect("the edges fit"),
16267            vec![-32_768_i16, 0, 32_767]
16268        );
16269        fit::<i16>(&[32_768]).expect_err("one past the top does not fit");
16270        fit::<i16>(&[-32_769]).expect_err("one past the bottom does not fit");
16271        assert_eq!(fit::<u16>(&[0, 65_535]).expect("the edges fit"), vec![0_u16, 65_535]);
16272        fit::<u16>(&[65_536]).expect_err("one past the top does not fit");
16273        fit::<u16>(&[-1]).expect_err("a negative does not fit an unsigned page");
16274        assert_eq!(
16275            fit::<i32>(&[i64::from(i32::MIN), 0, i64::from(i32::MAX)]).expect("the edges fit"),
16276            vec![i32::MIN, 0, i32::MAX]
16277        );
16278        fit::<i32>(&[i64::from(i32::MAX) + 1]).expect_err("one past the top does not fit");
16279        fit::<i32>(&[i64::from(i32::MIN) - 1]).expect_err("one past the bottom does not fit");
16280        assert_eq!(
16281            fit::<u32>(&[0, 4_294_967_295]).expect("the edges fit"),
16282            vec![0_u32, 4_294_967_295]
16283        );
16284        fit::<u32>(&[4_294_967_296]).expect_err("one past the top does not fit");
16285        fit::<u32>(&[-1]).expect_err("a negative does not fit an unsigned page");
16286
16287        // One value in a page that fits is still a page that does not, which is the thing an or
16288        // into an accumulator could get wrong in a way a page of one value would never show.
16289        fit::<i8>(&[0, 1, 2, 128, 3]).expect_err("one bad value spoils the page");
16290    }
16291
16292    /// The residue says yes to exactly what `TryFrom` says yes to.
16293    ///
16294    /// The edges above are the cases anyone would think to write down. This is the argument that
16295    /// there are no others, made by asking both questions about every value either narrow type could
16296    /// have an opinion about, and then about the values around the wide edges and the ends of an
16297    /// `i64`, which a range that size cannot reach.
16298    #[test]
16299    fn the_residue_agrees_with_a_checked_conversion_everywhere() {
16300        for value in -70_000_i64..70_000 {
16301            assert_eq!(fit::<i8>(&[value]).is_ok(), i8::try_from(value).is_ok(), "{value} as i8");
16302            assert_eq!(fit::<u8>(&[value]).is_ok(), u8::try_from(value).is_ok(), "{value} as u8");
16303            assert_eq!(fit::<i16>(&[value]).is_ok(), i16::try_from(value).is_ok(), "{value} i16");
16304            assert_eq!(fit::<u16>(&[value]).is_ok(), u16::try_from(value).is_ok(), "{value} u16");
16305        }
16306        let wide = [i64::MIN, i64::MIN + 1, i64::from(i32::MIN), 0, i64::from(u32::MAX), i64::MAX];
16307        for edge in wide {
16308            for step in -2_i64..=2 {
16309                let value = edge.saturating_add(step);
16310                assert_eq!(
16311                    fit::<i32>(&[value]).is_ok(),
16312                    i32::try_from(value).is_ok(),
16313                    "{value} as i32"
16314                );
16315                assert_eq!(
16316                    fit::<u32>(&[value]).is_ok(),
16317                    u32::try_from(value).is_ok(),
16318                    "{value} as u32"
16319                );
16320            }
16321        }
16322    }
16323
16324    /// All three block layouts come back as the same values in the same order.
16325    ///
16326    /// Blocks outside the page are what every file this build writes holds. Blocks that say where
16327    /// they are but sit inside the page behind the order are format 26, and blocks behind one
16328    /// another with only their ends recorded are older still. Nothing in the writer produces the
16329    /// last two any more, so the only way to find out whether the reader still understands those
16330    /// files is to write them here. The
16331    /// bytes go straight into a file with no directory around them, because what is under test is
16332    /// [`open_global_dictionary`], which is handed a page and a file and asks the directory for
16333    /// nothing.
16334    ///
16335    /// Three thousand values so that there are three payload blocks and a partial fourth, which is
16336    /// what makes the last block the one place where a length and an end disagree about what they
16337    /// are counting.
16338    #[test]
16339    fn a_dictionary_reads_the_same_whether_its_blocks_say_where_they_are() {
16340        let spellings = (0..3_000)
16341            .map(|index| format!("value {index:08} {}", "y".repeat(index % 40)))
16342            .collect::<Vec<_>>();
16343        let mut read = Vec::new();
16344        for layout in ["outside", "inside", "behind"] {
16345            let mut dictionary = GlobalDictionary::new();
16346            for text in &spellings {
16347                dictionary.code(text).expect("a code for every spelling");
16348            }
16349            dictionary.finish_blocks().expect("the last block encodes");
16350            let order = dictionary.ranked(None).expect("a sorted order");
16351            // Where the blocks go if they start at `from` and follow one another.
16352            let laid = |from: u64| {
16353                let mut at = from;
16354                dictionary
16355                    .blocks
16356                    .iter()
16357                    .map(|block| {
16358                        let place =
16359                            Placed { start: at, length: block.len() as u64, hash: checksum(block) };
16360                        at += block.len() as u64;
16361                        place
16362                    })
16363                    .collect::<Vec<_>>()
16364            };
16365            let payload = dictionary.blocks.concat();
16366            let scattered = layout != "behind";
16367            let (bytes, encoded, offset, length) = if layout == "outside" {
16368                let mut bytes = vec![0; HEADER as usize];
16369                bytes.extend_from_slice(&payload);
16370                let encoded = encode_global_dictionary(&dictionary, &order, &laid(HEADER), true)
16371                    .expect("an encoding");
16372                let offset = bytes.len() as u64;
16373                bytes.extend_from_slice(&encoded.index);
16374                bytes.extend_from_slice(&encoded.ranks);
16375                bytes.extend_from_slice(&encoded.grams);
16376                let length = encoded.index.len() + encoded.ranks.len() + encoded.grams.len();
16377                (bytes, encoded, offset, length)
16378            } else {
16379                // The index is the same length wherever the blocks are, so a first pass says where
16380                // the page ends and the second writes the places that follow it.
16381                let first = encode_global_dictionary(&dictionary, &order, &laid(0), scattered)
16382                    .expect("an encoding");
16383                let body = (first.index.len() + first.ranks.len() + first.grams.len()) as u64;
16384                let encoded = encode_global_dictionary(&dictionary, &order, &laid(body), scattered)
16385                    .expect("an encoding");
16386                let mut bytes = encoded.index.clone();
16387                bytes.extend_from_slice(&encoded.ranks);
16388                bytes.extend_from_slice(&encoded.grams);
16389                bytes.extend_from_slice(&payload);
16390                let length = bytes.len();
16391                (bytes, encoded, 0, length)
16392            };
16393            let path = path(&format!("blocks-{layout}"));
16394            fs::write(&path, &bytes).expect("the dictionary is written on its own");
16395            let file = Arc::new(File::open(&path).expect("it opens again"));
16396            let page = Page {
16397                offset,
16398                length: u32::try_from(length).expect("a test dictionary is small"),
16399                hash: checksum(&encoded.index),
16400            };
16401            let opened =
16402                open_global_dictionary(file, page, &LogicalType::Varchar, TEXT_KEEP_BUDGET)
16403                    .expect("a dictionary laid out either way opens");
16404            let mut swept: Vec<Vec<u8>> = Vec::new();
16405            let mut at = 0;
16406            while at < opened.len() {
16407                at = opened
16408                    .sweep_text(at, opened.len(), &mut |_index: usize, text: &[u8]| {
16409                        swept.push(text.to_vec());
16410                        Ok(())
16411                    })
16412                    .expect("a sweep reads");
16413            }
16414            fs::remove_file(&path).expect("clean up");
16415            read.push(swept);
16416        }
16417        let wanted =
16418            spellings.iter().map(|text| text.as_bytes().to_vec()).collect::<Vec<Vec<u8>>>();
16419        assert_eq!(read[0], wanted, "the blocks outside the page hold the values");
16420        assert_eq!(read[1], read[0], "the blocks inside the page hold the same values");
16421        assert_eq!(read[2], read[0], "the blocks behind one another hold the same values");
16422    }
16423
16424    /// A dictionary at its budget sweeps without keeping, and still answers what it answered.
16425    ///
16426    /// The budget is a quarter of a gigabyte in a running database, which is a fine size for a real
16427    /// column and no size at all for a test, so this opens the same dictionary a second time with a
16428    /// budget of zero. That is the shape of the hundred million row case: `URL` fills the budget
16429    /// somewhere in the middle of itself and everything past that point is read and dropped, which
16430    /// costs the decode again and holds none of it.
16431    #[test]
16432    fn a_dictionary_at_its_budget_sweeps_without_keeping() {
16433        let path = path("dictionary-budget");
16434        let spellings = (0..2_500)
16435            .map(|index| Value::Varchar(format!("value {index:08} {}", "y".repeat(index % 40))))
16436            .collect::<Vec<_>>();
16437        let mut writer =
16438            Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
16439                .expect("new file");
16440        for part in spellings.chunks(1_024) {
16441            writer
16442                .append(
16443                    &Chunk::new(vec![
16444                        Vector::from_values(LogicalType::Varchar, part).expect("strings"),
16445                    ])
16446                    .expect("one column"),
16447                )
16448                .expect("stripe written");
16449        }
16450        writer.finish().expect("commit");
16451
16452        let reader = Reader::open(&path).expect("valid directory");
16453        let page = reader.table.dictionaries[0].expect("a string column has one");
16454        let file = Arc::clone(&reader.file);
16455        let starved = open_global_dictionary(file, page, &LogicalType::Varchar, 0)
16456            .expect("a dictionary opens whatever it may keep");
16457
16458        let resting = starved.footprint();
16459        let mut swept: Vec<Vec<u8>> = Vec::new();
16460        let mut at = 0;
16461        while at < starved.len() {
16462            at = starved
16463                .sweep_text(at, starved.len(), &mut |_index: usize, text: &[u8]| {
16464                    swept.push(text.to_vec());
16465                    Ok(())
16466                })
16467                .expect("a sweep reads");
16468        }
16469        assert_eq!(swept.len(), spellings.len(), "a starved sweep still reads every value");
16470        assert_eq!(starved.footprint(), resting, "and keeps no block it decoded");
16471
16472        let generous = reader.dictionary(0).expect("read").expect("a string column has one");
16473        let read = (0..generous.len())
16474            .map(|code| generous.try_bytes_at(code).expect("read").expect("a value").to_vec())
16475            .collect::<Vec<_>>();
16476        assert_eq!(swept, read, "a starved sweep answers what a point read answers");
16477        fs::remove_file(path).expect("remove scratch file");
16478    }
16479
16480    #[test]
16481    fn damaged_membership_cannot_skip_a_string_page() {
16482        let path = path("damaged-membership");
16483        let mut writer = Writer::create(
16484            &path,
16485            "items",
16486            vec![
16487                Field::required("id", LogicalType::Integer),
16488                Field::new("text", LogicalType::Varchar),
16489            ],
16490        )
16491        .expect("new file");
16492        writer.append(&sample()).expect("stripe written");
16493        writer.finish().expect("commit");
16494
16495        let reader = Reader::open(&path).expect("valid directory");
16496        let membership = reader.table.stripes[0].memberships.get(1).expect("string membership");
16497        let mut file = OpenOptions::new().write(true).open(&path).expect("open membership page");
16498        file.seek(SeekFrom::Start(membership.offset)).expect("membership start");
16499        file.write_all(&[255]).expect("damage membership");
16500        let error = reader.skips_codes(0, 1, &[3]).expect_err("corruption must not skip rows");
16501        assert!(error.message().contains("membership page checksum differs"), "{error}");
16502        fs::remove_file(path).expect("remove scratch file");
16503    }
16504
16505    #[test]
16506    fn membership_delta_stream_is_sorted_exact_and_bounded() {
16507        let unique = unique_codes(&[900, 4, 4, 72, 9, u32::MAX]);
16508        assert_eq!(unique, [4, 9, 72, 900, u32::MAX]);
16509        let encoded = encode_membership(&unique);
16510        assert_eq!(
16511            decode_membership(&encoded).expect("valid membership"),
16512            [4, 9, 72, 900, u32::MAX]
16513        );
16514        // A stripe's index is the union of its parts', so a code in two of them is in it once and
16515        // the result is still one ascending run of deltas.
16516        let merged = merged_codes(vec![vec![4, 900], vec![9, 900, u32::MAX], vec![72]]);
16517        assert_eq!(merged, [4, 9, 72, 900, u32::MAX]);
16518        assert_eq!(
16519            decode_membership(&encode_membership(&merged)).expect("valid membership"),
16520            unique
16521        );
16522        assert!(decode_membership(&[1, 0x80]).is_err(), "a truncated varint is invalid");
16523        assert!(
16524            decode_membership(&[1, 0xff, 0xff, 0xff, 0xff, 0x10]).is_err(),
16525            "a value past u32 is invalid"
16526        );
16527    }
16528
16529    #[test]
16530    fn a_global_dictionary_may_be_larger_than_one_column_page() {
16531        let dictionary = Page {
16532            offset: HEADER,
16533            length: u32::try_from(MAX_PAGE + 1).expect("the page bound fits on disk"),
16534            hash: 0,
16535        };
16536        let table = Table {
16537            name: "items".to_owned(),
16538            fields: vec![Field::new("text", LogicalType::Varchar)],
16539            stripes: Vec::new(),
16540            rows: 0,
16541            dictionaries: vec![Some(dictionary)],
16542            dictionary_payloads: Vec::new(),
16543            demoted: Vec::new(),
16544            distincts: vec![None],
16545            frequencies: vec![None],
16546            pair_frequencies: Vec::new(),
16547            frequency_texts: Vec::new(),
16548            host_groups: None,
16549            clustering: None,
16550            generation: 1,
16551            sections: Vec::new(),
16552        };
16553        let directory = encode_directory(&table).expect("directory");
16554        let file_size = dictionary.offset + u64::from(dictionary.length) + 1;
16555
16556        let decoded = decode_directory(&directory, file_size).expect("large lazy dictionary");
16557        assert_eq!(decoded.dictionaries[0].expect("dictionary").length, dictionary.length);
16558    }
16559
16560    #[test]
16561    fn a_column_with_one_value_everywhere_costs_almost_nothing_a_row() {
16562        let path = path("constant-codes");
16563        let mut writer =
16564            Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
16565                .expect("new file");
16566        let empty = vec![Value::Varchar(String::new()); 1024];
16567        for _ in 0..4 {
16568            let column = Vector::from_values(LogicalType::Varchar, &empty).expect("strings");
16569            writer.append(&Chunk::new(vec![column]).expect("one column")).expect("a part");
16570        }
16571        writer.finish().expect("commit");
16572
16573        let reader = Reader::open(&path).expect("valid directory");
16574        let pages = reader.layout().columns.first().expect("one column").pages;
16575        // This column used to cost four bytes a row, 16,384 of them, the same as a column of four
16576        // thousand distinct URLs would. The cascade calls each part a constant, so what is left is
16577        // a tag, a count and the value, and the row count stops being what drives the number.
16578        assert!(pages < 256, "{pages} bytes of pages for 4,096 rows of one value");
16579        let read = reader.read(3, &[0]).expect("the last part back");
16580        assert_eq!(read.value_at(0, 0), Value::Varchar(String::new()));
16581        assert_eq!(read.value_at(1023, 0), Value::Varchar(String::new()));
16582        fs::remove_file(path).expect("remove scratch file");
16583    }
16584
16585    #[test]
16586    fn a_cascade_value_too_wide_for_its_column_is_refused_rather_than_cut() {
16587        // What a damaged page looks like from here: the cascade decoded, so the bytes are not
16588        // truncated, but the values do not belong to the column the directory says they do.
16589        let over = vec![i64::from(i32::MAX) + 1];
16590        let error = narrowed(&LogicalType::Integer, over).expect_err("a page that disagrees");
16591        assert!(format!("{error}").contains("not of its type"), "{error}");
16592        assert!(narrowed(&LogicalType::BigInt, vec![i64::MIN]).is_ok(), "bigint holds all of i64");
16593        assert!(narrowed(&LogicalType::Varchar, vec![0]).is_err(), "strings are not integers");
16594    }
16595
16596    #[test]
16597    fn a_code_stream_the_cascade_cannot_shrink_is_left_alone() {
16598        // A shift register rather than a run, because an arithmetic run is the one wide shape the
16599        // cascade does shrink. This is what a column with tens of millions of distinct values hands
16600        // over: full width codes with no order to them.
16601        let mut state: u32 = 0x9e37_79b9;
16602        let spread: Vec<u32> = (0..1024)
16603            .map(|_| {
16604                state ^= state << 13;
16605                state ^= state >> 17;
16606                state ^= state << 5;
16607                state
16608            })
16609            .collect();
16610        assert_eq!(encoded_codes(&spread).expect("no failure"), None);
16611        let near: Vec<u32> = (0..1024).collect();
16612        let coded = encoded_codes(&near).expect("no failure").expect("counting up is packable");
16613        assert!(coded.len() < near.len() * 4, "{} bytes for a run of 1,024", coded.len());
16614    }
16615
16616    /// The columns of a stripe are encoded on whichever thread got to them, so the one thing that
16617    /// must not depend on which thread that was is the file. Two writes of the same rows are
16618    /// compared byte for byte rather than value for value, because a dictionary that two columns
16619    /// somehow shared would still read back correctly and would hand out its codes in the order the
16620    /// threads happened to run in, which is exactly what this is here to catch.
16621    #[test]
16622    fn two_writes_of_the_same_rows_give_the_same_bytes() {
16623        fn written(path: &PathBuf) {
16624            let fields = (0..40)
16625                .map(|column| {
16626                    let ty =
16627                        if column % 4 == 0 { LogicalType::Varchar } else { LogicalType::BigInt };
16628                    Field::new(format!("c{column}"), ty)
16629                })
16630                .collect::<Vec<_>>();
16631            let mut writer = Writer::create(path, "wide", fields).expect("new file");
16632            for part in 0..70_u64 {
16633                let columns = (0..40)
16634                    .map(|column| {
16635                        let values = (0..64_u64)
16636                            .map(|row| {
16637                                let seed = part.wrapping_mul(31).wrapping_add(row);
16638                                if column % 4 == 0 {
16639                                    Value::Varchar(format!("v{}", seed % 17))
16640                                } else {
16641                                    Value::BigInt(i64::try_from(seed % 97).expect("small"))
16642                                }
16643                            })
16644                            .collect::<Vec<_>>();
16645                        let ty = if column % 4 == 0 {
16646                            LogicalType::Varchar
16647                        } else {
16648                            LogicalType::BigInt
16649                        };
16650                        Vector::from_values(ty, &values).expect("a column")
16651                    })
16652                    .collect::<Vec<_>>();
16653                writer.append(&Chunk::new(columns).expect("forty columns")).expect("a part");
16654            }
16655            writer.finish().expect("commit");
16656        }
16657
16658        let first = path("repeatable-one");
16659        let second = path("repeatable-two");
16660        written(&first);
16661        written(&second);
16662        let left = fs::read(&first).expect("the first file");
16663        let right = fs::read(&second).expect("the second file");
16664        assert_eq!(left.len(), right.len(), "two writes of the same rows differ in length");
16665        assert!(left == right, "two writes of the same rows differ in their bytes");
16666
16667        // And the rows are still there, since a pair of identically wrong files would pass the
16668        // comparison above on its own.
16669        let reader = Reader::open(&first).expect("valid directory");
16670        assert_eq!(reader.table().rows(), 70 * 64);
16671        let read = reader.read(0, &[0, 1]).expect("the first part back");
16672        assert_eq!(read.value_at(0, 0), Value::Varchar("v0".to_owned()));
16673        assert_eq!(read.value_at(0, 1), Value::BigInt(0));
16674        fs::remove_file(first).expect("remove scratch file");
16675        fs::remove_file(second).expect("remove scratch file");
16676    }
16677
16678    /// Three tables of different shapes in one file, read back by name.
16679    fn three_tables(path: &PathBuf) {
16680        let writer = Writer::create(
16681            path,
16682            "region",
16683            vec![
16684                Field::new("r_key", LogicalType::Integer),
16685                Field::new("r_name", LogicalType::Varchar),
16686            ],
16687        )
16688        .expect("new file");
16689        let mut writer = writer;
16690        writer
16691            .append(
16692                &Chunk::new(vec![
16693                    Vector::from_values(
16694                        LogicalType::Integer,
16695                        &[Value::Integer(0), Value::Integer(1)],
16696                    )
16697                    .expect("keys"),
16698                    Vector::from_values(
16699                        LogicalType::Varchar,
16700                        &[Value::Varchar("AFRICA".to_owned()), Value::Varchar("ASIA".to_owned())],
16701                    )
16702                    .expect("names"),
16703                ])
16704                .expect("two columns"),
16705            )
16706            .expect("a part");
16707        let mut writer = writer
16708            .next("empty", vec![Field::new("nothing", LogicalType::BigInt)])
16709            .expect("a second table");
16710        writer
16711            .append(
16712                &Chunk::new(vec![
16713                    Vector::from_values(LogicalType::BigInt, &[Value::BigInt(7)]).expect("a row"),
16714                ])
16715                .expect("one column"),
16716            )
16717            .expect("a part");
16718        let mut writer =
16719            writer.next("wide", vec![Field::new("n", LogicalType::BigInt)]).expect("a third table");
16720        for part in 0..70_i64 {
16721            let values = (0..64).map(|row| Value::BigInt(part * 64 + row)).collect::<Vec<_>>();
16722            writer
16723                .append(
16724                    &Chunk::new(vec![
16725                        Vector::from_values(LogicalType::BigInt, &values).expect("a column"),
16726                    ])
16727                    .expect("one column"),
16728                )
16729                .expect("a part");
16730        }
16731        writer.finish().expect("commit");
16732    }
16733
16734    #[test]
16735    fn three_tables_in_one_file_read_back_by_name() {
16736        let file = path("three-tables");
16737        three_tables(&file);
16738        let catalog = Catalog::open(&file).expect("a committed catalog");
16739        assert_eq!(catalog.names().collect::<Vec<_>>(), ["region", "empty", "wide"]);
16740
16741        let region = catalog.table("region").expect("the first table");
16742        assert_eq!(region.table().rows(), 2);
16743        assert_eq!(
16744            region.read(0, &[1]).expect("names").value_at(1, 0),
16745            Value::Varchar("ASIA".to_owned())
16746        );
16747
16748        let wide = catalog.table("wide").expect("the third table");
16749        assert_eq!(wide.table().rows(), 70 * 64);
16750        assert_eq!(wide.read(0, &[0]).expect("the first part").value_at(0, 0), Value::BigInt(0));
16751
16752        // The middle table is reached without the one after it having been touched, which is what
16753        // a directory per table buys over one directory of everything.
16754        let empty = catalog.table("empty").expect("the second table");
16755        assert_eq!(empty.table().rows(), 1);
16756        assert_eq!(empty.read(0, &[0]).expect("the row").value_at(0, 0), Value::BigInt(7));
16757
16758        fs::remove_file(file).expect("remove scratch file");
16759    }
16760
16761    #[test]
16762    fn a_name_the_file_does_not_hold_is_an_error_rather_than_the_first_table() {
16763        let file = path("three-tables-missing");
16764        three_tables(&file);
16765        let catalog = Catalog::open(&file).expect("a committed catalog");
16766        let error = catalog.table("nation").expect_err("no such table");
16767        assert!(error.message().contains("nation"), "{}", error.message());
16768        fs::remove_file(file).expect("remove scratch file");
16769    }
16770
16771    #[test]
16772    fn a_file_of_three_tables_will_not_open_as_one() {
16773        let file = path("three-tables-unnamed");
16774        three_tables(&file);
16775        let error = Reader::open(&file).expect_err("more than one table");
16776        assert!(error.message().contains("more than one table"), "{}", error.message());
16777        fs::remove_file(file).expect("remove scratch file");
16778    }
16779
16780    /// One column per storage width, because the width is what decides how many bytes a row costs.
16781    #[test]
16782    fn decimals_of_every_storage_width_round_trip() {
16783        let file = path("decimals");
16784        let widths = [(4_u8, 2_u8), (9, 2), (18, 4), (38, 6)];
16785        let fields = widths
16786            .iter()
16787            .enumerate()
16788            .map(|(index, (width, scale))| {
16789                Field::new(
16790                    format!("d{index}"),
16791                    LogicalType::decimal(*width, *scale).expect("a decimal type"),
16792                )
16793            })
16794            .collect::<Vec<_>>();
16795        let mut writer = Writer::create(&file, "money", fields).expect("new file");
16796        let rows: [i128; 3] = [-1234, 0, 999];
16797        let columns = widths
16798            .iter()
16799            .map(|(width, scale)| {
16800                let values = rows
16801                    .iter()
16802                    .map(|unscaled| Value::Decimal {
16803                        unscaled: *unscaled,
16804                        width: *width,
16805                        scale: *scale,
16806                    })
16807                    .collect::<Vec<_>>();
16808                Vector::from_values(
16809                    LogicalType::decimal(*width, *scale).expect("a decimal type"),
16810                    &values,
16811                )
16812                .expect("a decimal column")
16813            })
16814            .collect::<Vec<_>>();
16815        writer.append(&Chunk::new(columns).expect("four columns")).expect("a part");
16816        writer.finish().expect("commit");
16817
16818        let reader = Reader::open(&file).expect("a committed file");
16819        for (index, (width, scale)) in widths.iter().enumerate() {
16820            assert_eq!(
16821                reader.table().fields()[index].ty,
16822                LogicalType::decimal(*width, *scale).expect("a decimal type"),
16823                "column {index} came back as another type"
16824            );
16825            let column = reader.read(0, &[index]).expect("the column");
16826            for (row, unscaled) in rows.iter().enumerate() {
16827                assert_eq!(
16828                    column.value_at(row, 0),
16829                    Value::Decimal { unscaled: *unscaled, width: *width, scale: *scale },
16830                    "column {index} row {row}"
16831                );
16832            }
16833        }
16834        fs::remove_file(file).expect("remove scratch file");
16835    }
16836
16837    #[test]
16838    fn two_tables_of_one_name_are_refused_before_anything_is_committed() {
16839        let file = path("two-of-a-name");
16840        let writer = Writer::create(&file, "t", vec![Field::new("a", LogicalType::BigInt)])
16841            .expect("new file");
16842        let error = writer
16843            .next("t", vec![Field::new("a", LogicalType::BigInt)])
16844            .expect_err("the same name twice");
16845        assert!(error.message().contains("same name"), "{}", error.message());
16846        fs::remove_file(file).expect("remove scratch file");
16847    }
16848
16849    #[test]
16850    fn integer_tally_counts_encoded_rows_and_declines_null_parts() {
16851        let file = path("integer-tally");
16852        let mut writer =
16853            Writer::create(&file, "events", vec![Field::new("source", LogicalType::SmallInt)])
16854                .expect("new file");
16855        let mut values = vec![Value::SmallInt(0); 1024];
16856        values[7] = Value::SmallInt(3);
16857        values[99] = Value::SmallInt(-2);
16858        values[1001] = Value::SmallInt(3);
16859        let column = Vector::from_values(LogicalType::SmallInt, &values).expect("integer values");
16860        writer.append(&Chunk::new(vec![column]).expect("one column")).expect("first part");
16861        values[0] = Value::Null;
16862        let column = Vector::from_values(LogicalType::SmallInt, &values).expect("nullable values");
16863        writer.append(&Chunk::new(vec![column]).expect("one column")).expect("second part");
16864        writer.finish().expect("commit");
16865
16866        let reader = Reader::open(&file).expect("read file");
16867        assert_eq!(
16868            reader.integer_tally(0, 0).expect("valid part"),
16869            Some(vec![(-2, 1), (0, 1021), (3, 2)])
16870        );
16871        assert!(reader.integer_tally(1, 0).expect("valid null part").is_none());
16872        let catalog = Catalog::open(&file).expect("catalog");
16873        assert_eq!(
16874            catalog.integer_tally("events", 0).expect("nullable column"),
16875            Some(vec![(-2, 2), (0, 2041), (3, 4)])
16876        );
16877        fs::remove_file(file).expect("remove scratch file");
16878    }
16879
16880    #[test]
16881    fn catalog_tallies_one_integer_column_without_opening_the_whole_table() {
16882        let file = path("catalog-integer-tally");
16883        let mut writer = Writer::create(
16884            &file,
16885            "events",
16886            vec![
16887                Field::new("noise", LogicalType::SmallInt),
16888                Field::new("source", LogicalType::SmallInt),
16889            ],
16890        )
16891        .expect("new file");
16892        let noise = vec![Value::SmallInt(9); 1024];
16893        let mut source = vec![Value::SmallInt(0); 1024];
16894        source[7] = Value::SmallInt(3);
16895        source[99] = Value::SmallInt(-2);
16896        let chunk = Chunk::new(vec![
16897            Vector::from_values(LogicalType::SmallInt, &noise).expect("noise"),
16898            Vector::from_values(LogicalType::SmallInt, &source).expect("source"),
16899        ])
16900        .expect("two columns");
16901        writer.append(&chunk).expect("append");
16902        writer.finish().expect("commit");
16903
16904        let catalog = Catalog::open(&file).expect("catalog");
16905        assert_eq!(
16906            catalog.integer_tally("events", 1).expect("selected column"),
16907            Some(vec![(-2, 1), (0, 1022), (3, 1)])
16908        );
16909        assert_eq!(
16910            catalog.integer_tally("events", 0).expect("other column"),
16911            Some(vec![(9, 1024)])
16912        );
16913        fs::remove_file(file).expect("remove scratch file");
16914    }
16915
16916    #[test]
16917    fn opening_the_catalog_reads_no_table_directory() {
16918        let file = path("catalog-only");
16919        three_tables(&file);
16920        let catalog = Catalog::open(&file).expect("a committed catalog");
16921        // The header and one slot, and nothing under it. The third table's directory covers seventy
16922        // stripes and reading it here would be the whole point of the two levels thrown away.
16923        assert_eq!(catalog.opening.reads, 2, "opening the catalog read more than the slot");
16924        assert_eq!(catalog.names().len(), 3);
16925        fs::remove_file(file).expect("remove scratch file");
16926    }
16927
16928    /// The checksum answers what it has always answered, at every length its branches split on.
16929    ///
16930    /// This is a compatibility test rather than a correctness one. Nothing about the hash has to be
16931    /// any particular function, but a file already on disk carries the answers the version that
16932    /// wrote it gave, so a change here is a change that makes every stored file fail to verify. The
16933    /// lengths are the ones the code makes decisions about: nothing, under a block, a block exactly,
16934    /// a block and a word, a word and a half word, and a half word and a byte.
16935    ///
16936    /// The empty answer is the published xxHash64 vector for an empty input at seed zero, which is
16937    /// also a check that this is the function it says it is.
16938    #[test]
16939    fn the_checksum_answers_what_it_has_always_answered() {
16940        let bytes: Vec<u8> =
16941            (0..1000_u32).map(|at| (at.wrapping_mul(31).wrapping_add(7) % 251) as u8).collect();
16942        for (length, expected) in [
16943            (0, 0xef46_db37_51d8_e999),
16944            (1, 0xa96c_7f0c_e858_bbb7),
16945            (3, 0x56e6_9576_32a4_87f9),
16946            (4, 0xc60d_15b1_e3ff_8f04),
16947            (5, 0x8088_1585_8624_dd4e),
16948            (7, 0xafbe_fc3d_6c6f_9a8e),
16949            (8, 0x3da5_c7aa_2696_83e0),
16950            (9, 0x465e_c429_b13c_3892),
16951            (15, 0xdee8_9d8a_065a_6233),
16952            (16, 0x1330_489a_7767_9c80),
16953            (31, 0x3391_303d_485e_846e),
16954            (32, 0x40b7_aff7_5d45_bbc8),
16955            (33, 0x4997_cae4_951c_17a5),
16956            (39, 0x5807_28fd_5c14_5739),
16957            (40, 0xf95c_f6f5_c08a_3d3b),
16958            (63, 0x2944_b4da_fc69_b206),
16959            (64, 0xbb76_f6ef_19bd_5a1b),
16960            (65, 0x814e_0c65_4a9f_d640),
16961            (127, 0x00de_aab1_31cf_f89b),
16962            (1000, 0x9e33_00c1_cde3_c58d),
16963        ] {
16964            assert_eq!(checksum(&bytes[..length]), expected, "the checksum of {length} bytes");
16965        }
16966        assert_eq!(checksum(b"the quick brown fox jumps over the lazy dog"), 0xed71_4233_c5a9_a792);
16967    }
16968    /// A declared order survives the file, and a table that declared none stays as it was.
16969    ///
16970    /// The second half is the one worth a test. The clustering section is written only when there
16971    /// is a declaration, so a file of two tables where one is clustered exercises both the present
16972    /// and the absent branch of the decoder in one directory, which is where a length bug would
16973    /// show up as one table reading the other's bytes.
16974    #[test]
16975    fn a_declared_order_comes_back_out_of_the_file() {
16976        let path = path("clustered");
16977        let shipped = vec![
16978            Field::new("key", LogicalType::BigInt),
16979            Field::new("line", LogicalType::Integer),
16980            Field::new("shipdate", LogicalType::Date),
16981        ];
16982        let plain = vec![Field::new("a", LogicalType::Integer)];
16983        let stage_zero = Clustering::new(vec![2, 0, 1], Width::Month, &shipped).expect("valid");
16984
16985        let mut writer = Writer::create(&path, "lineitem", shipped)
16986            .expect("new file")
16987            .declare(stage_zero.clone())
16988            .expect("the columns are the table's");
16989        let column = |ty: LogicalType, values: &[Value]| {
16990            Vector::from_values(ty, values).expect("the values match the type")
16991        };
16992        writer
16993            .append(
16994                &Chunk::new(vec![
16995                    column(
16996                        LogicalType::BigInt,
16997                        &[Value::BigInt(0), Value::BigInt(1), Value::BigInt(2), Value::BigInt(3)],
16998                    ),
16999                    column(
17000                        LogicalType::Integer,
17001                        &[
17002                            Value::Integer(1),
17003                            Value::Integer(1),
17004                            Value::Integer(1),
17005                            Value::Integer(1),
17006                        ],
17007                    ),
17008                    column(
17009                        LogicalType::Date,
17010                        &[Value::Date(0), Value::Date(1), Value::Date(2), Value::Date(3)],
17011                    ),
17012                ])
17013                .expect("three columns"),
17014            )
17015            .expect("four rows");
17016        let mut writer = writer.next("nation", plain).expect("a second table");
17017        writer
17018            .append(
17019                &Chunk::new(vec![column(LogicalType::Integer, &[Value::Integer(7)])])
17020                    .expect("one column"),
17021            )
17022            .expect("one row");
17023        writer.finish().expect("commit");
17024
17025        let catalog = Catalog::open(&path).expect("reopen");
17026        let lineitem = catalog.table("lineitem").expect("the clustered table");
17027        assert_eq!(lineitem.table().clustering(), Some(&stage_zero));
17028        let nation = catalog.table("nation").expect("the plain table");
17029        assert_eq!(nation.table().clustering(), None, "nobody declared one here");
17030
17031        // And the rows are still the rows, because the section goes on the end of the directory
17032        // and the easy way to break that is to leave the cursor somewhere the next read trusts.
17033        assert_eq!(lineitem.table().rows(), 4);
17034        assert_eq!(nation.table().rows(), 1);
17035        fs::remove_file(&path).ok();
17036    }
17037
17038    /// A declaration naming a column the table does not have is refused where it is made.
17039    #[test]
17040    fn a_declaration_off_the_end_of_the_table_never_reaches_the_file() {
17041        let path = path("clustered-bad");
17042        let writer = Writer::create(&path, "items", vec![Field::new("a", LogicalType::Integer)])
17043            .expect("new file");
17044        let four =
17045            (0..4).map(|at| Field::new(format!("c{at}"), LogicalType::Integer)).collect::<Vec<_>>();
17046        let wrong = Clustering::new(vec![3], Width::Exact, &four).expect("valid against four");
17047        assert!(writer.declare(wrong).is_err(), "the table has one column, not four");
17048        fs::remove_file(&path).ok();
17049    }
17050
17051    /// The sorted order is the byte order, whatever the values do before they differ.
17052    ///
17053    /// A block handed out of the writer's lock to be encoded, and given back in whatever order the
17054    /// stripes happen to finish in, is the same block with the same signature as one encoded in
17055    /// place, and lands in the same position.
17056    #[test]
17057    fn blocks_handed_out_and_given_back_out_of_order_are_the_blocks_encoded_in_place() {
17058        let values = (0..PAYLOAD_SAMPLE_BLOCKS * TEXT_PAYLOAD_VALUES * 2 + 100)
17059            .map(|at| format!("http://example{}.test/page/{at:06}", at % 7))
17060            .collect::<Vec<_>>();
17061        let filled = || {
17062            let mut dictionary = GlobalDictionary::new();
17063            for value in &values {
17064                dictionary.code(value).expect("a code for every value");
17065            }
17066            dictionary.settle().expect("a shape");
17067            dictionary
17068        };
17069        let mut in_place = filled();
17070        in_place.finish_blocks().expect("every block encodes");
17071
17072        let mut handed = filled();
17073        let out = handed.hand_out(3);
17074        assert_eq!(out.len(), PAYLOAD_SAMPLE_BLOCKS * 2, "every sealed block goes out");
17075        assert!(handed.waiting.is_empty(), "and none is left to be encoded under the lock");
17076        for job in out.iter().rev() {
17077            assert_eq!(job.place().0, 3, "a block goes back to the column it came from");
17078            handed.take_back(job.place().1, job.encode().expect("encodes")).expect("taken back");
17079        }
17080        assert!(handed.early.is_empty(), "nothing is waiting on a gap");
17081        handed.finish_blocks().expect("the last block encodes");
17082
17083        assert_eq!(handed.blocks, in_place.blocks, "the same blocks in the same order");
17084        assert_eq!(handed.grams, in_place.grams, "with the same signatures");
17085    }
17086
17087    /// A block given back twice is a bug in whoever gave it, and is said rather than written twice.
17088    #[test]
17089    fn a_block_given_back_twice_is_refused() {
17090        let mut dictionary = GlobalDictionary::new();
17091        for at in 0..PAYLOAD_SAMPLE_BLOCKS * TEXT_PAYLOAD_VALUES {
17092            dictionary.code(&format!("value {at}")).expect("a code");
17093        }
17094        dictionary.settle().expect("a shape");
17095        let out = dictionary.hand_out(0);
17096        let last = out.last().expect("blocks went out");
17097        let at = last.place().1;
17098        dictionary.take_back(at, last.encode().expect("encodes")).expect("taken back once");
17099        assert!(dictionary.take_back(at, last.encode().expect("encodes")).is_err());
17100    }
17101
17102    /// The values here are the shape the sort is built for and the shape a comparison sort is worst
17103    /// at: a common scheme, a handful of hosts, and a path that only decides the pair thirty bytes
17104    /// in. They also cover what the bucketing has to get right at the edges, which is a value that
17105    /// has run out where another carries on, the empty value, and enough entries to take the range
17106    /// down through several passes and out the bottom into the comparison that finishes it.
17107    #[test]
17108    fn the_dictionary_order_is_the_byte_order_however_deep_the_values_agree() {
17109        let mut values = vec![String::new(), "http://".to_owned()];
17110        for host in 0..7 {
17111            for path in 0..30 {
17112                values.push(format!("http://example{host}.test/page/{path:04}/index.html"));
17113                values.push(format!("http://example{host}.test/page/{path:04}"));
17114            }
17115        }
17116        values.push("http://example0.test/page/0000/index.htmlx".to_owned());
17117
17118        let mut dictionary = GlobalDictionary::new();
17119        for value in &values {
17120            dictionary.code(value).expect("a code for every value");
17121        }
17122        dictionary.finish_blocks().expect("the last block encodes");
17123        let ranked = dictionary.ranked(None).expect("a sorted order");
17124        assert_eq!(ranked.len(), values.len(), "one entry a distinct value");
17125
17126        let spellings = dictionary_values(&dictionary);
17127        let seen = ranked
17128            .iter()
17129            .map(|&(_, code)| {
17130                String::from_utf8(spellings[code as usize].clone()).expect("text in, text out")
17131            })
17132            .collect::<Vec<_>>();
17133        let mut wanted = values.clone();
17134        wanted.sort_unstable();
17135        assert_eq!(seen, wanted, "the order is the order the bytes give");
17136
17137        for &(carried, code) in &ranked {
17138            let value = &spellings[code as usize];
17139            assert_eq!(carried, head(value), "the head belongs to the value it is filed with");
17140        }
17141    }
17142
17143    /// Picking the commonest entries leaves exactly what sorting all of them and cutting left.
17144    ///
17145    /// The counts here are deliberately full of ties, including a tie that straddles the cut, which
17146    /// is where a partition and a sort can disagree if the comparison they are given is not total.
17147    #[test]
17148    fn the_commonest_entries_are_the_ones_a_full_sort_would_have_kept() {
17149        let entry =
17150            |value: u32, count: u64| FrequencyEntry { value: FrequencyValue::Code(value), count };
17151        let mut all = (0..FREQUENCY_ENTRIES as u32 * 3)
17152            .map(|code| entry(code, u64::from(code % 7) + 1))
17153            .collect::<Vec<_>>();
17154        all.push(FrequencyEntry { value: FrequencyValue::Null, count: 4 });
17155
17156        let mut sorted = all.clone();
17157        sorted.sort_unstable_by(|left, right| {
17158            right.count.cmp(&left.count).then_with(|| frequency_order(left.value, right.value))
17159        });
17160        let wanted_omitted = sorted[FREQUENCY_ENTRIES].count;
17161        sorted.truncate(FREQUENCY_ENTRIES);
17162
17163        let mut picked = all.clone();
17164        let omitted = keep_most_frequent(&mut picked);
17165        assert_eq!(omitted, wanted_omitted, "the largest count that did not make the cut");
17166        assert_eq!(picked.len(), FREQUENCY_ENTRIES, "the cut is where it says it is");
17167        assert!(
17168            picked
17169                .iter()
17170                .zip(&sorted)
17171                .all(|(one, two)| one.value == two.value && one.count == two.count),
17172            "the same entries in the same order"
17173        );
17174
17175        let mut short = all[..FREQUENCY_ENTRIES - 1].to_vec();
17176        let omitted = keep_most_frequent(&mut short);
17177        assert_eq!(omitted, 0, "nothing is omitted when everything fits");
17178        assert!(short.windows(2).all(|pair| pair[0].count >= pair[1].count), "still in order");
17179    }
17180
17181    /// A dictionary too small to bucket, and one with nothing in it, come back in order too.
17182    #[test]
17183    fn a_short_dictionary_sorts_without_a_bucketing_pass() {
17184        let empty = GlobalDictionary::new();
17185        assert!(empty.ranked(None).expect("an empty order").is_empty(), "nothing in, nothing out");
17186
17187        let mut dictionary = GlobalDictionary::new();
17188        for value in ["pear", "apple", "", "apples", "app"] {
17189            dictionary.code(value).expect("a code for every value");
17190        }
17191        dictionary.finish_blocks().expect("the one block encodes");
17192        let spellings = dictionary_values(&dictionary);
17193        let seen = dictionary
17194            .ranked(None)
17195            .expect("a sorted order")
17196            .iter()
17197            .map(|&(_, code)| spellings[code as usize].clone())
17198            .collect::<Vec<_>>();
17199        let wanted: Vec<Vec<u8>> =
17200            [&b""[..], b"app", b"apple", b"apples", b"pear"].iter().map(|v| v.to_vec()).collect();
17201        assert_eq!(seen, wanted, "shorter first where one runs out inside another");
17202    }
17203
17204    /// A demoted dictionary gives back what it kept for looking values up, the load profile is told,
17205    /// and it refuses any value after that.
17206    #[test]
17207    fn a_demoted_dictionary_holds_less_and_takes_no_more_values() {
17208        let profile = LoadProfile::begin("demoted");
17209        let mut dictionary = GlobalDictionary::new();
17210        for value in 0..50_000 {
17211            dictionary.code(&format!("https://example.com/page/{value}")).expect("a code");
17212        }
17213        let (_, grown) = dictionary.recharge(Some(&profile));
17214        assert_eq!(profile.held(), grown, "the profile holds what the dictionary does");
17215
17216        dictionary.demote();
17217        let (before, after) = dictionary.recharge(Some(&profile));
17218        assert_eq!(before, grown);
17219        // What stays is the ends, the counts and the blocks not yet written, which a load writes
17220        // as it goes, so here the drop is the hash tables and the check hashes.
17221        assert!(after < grown - grown / 4, "the lookup is let go of: {after} of {grown}");
17222        assert_eq!(profile.held(), after, "the profile was told about the drop");
17223        assert!(dictionary.code("one more").is_err(), "a demoted dictionary takes no values");
17224
17225        dictionary.demote();
17226        assert_eq!(
17227            dictionary.recharge(Some(&profile)),
17228            (after, after),
17229            "demoting twice is a no-op"
17230        );
17231        assert_eq!(dictionary.values(), 50_000, "the values coded before stay");
17232    }
17233}