Skip to main content

rudb_native/
lib.rs

1//! Rudb's single-file columnar snapshot format.
2//!
3//! A committed directory names independently readable column pages. It has two levels: a catalog
4//! directory naming every table in the file, which is what a footer slot points at and what opening
5//! a database reads, and one directory per table under it holding that table's stripes, pages and
6//! statistics. One slot write publishes all of them, so a commit is atomic across tables.
7//!
8//! This version handles scalar columns; the file header has two generation slots so an unfinished
9//! replacement directory cannot hide the last complete one. See
10//! `spec/storage-v3/12-many-tables-in-one-file.md`.
11//!
12//! # Parts and stripes
13//!
14//! A part is one appended chunk, which is a thousand rows, and it is the unit a scan decodes and
15//! hands to the pipeline. A stripe is sixty four parts, and it is the unit the directory describes
16//! and the unit the file is laid out in: one page per column per stripe, holding that column's
17//! sixty four part payloads end to end.
18//!
19//! The two are separate because they are sized by different pressures. A part wants to be small
20//! because it is a vector and vectors live in cache. A stripe wants to be large because everything
21//! the directory holds is per stripe and the directory is one buffer that has to be read and
22//! decoded before a single row can be answered. A hundred million rows of the hundred and five
23//! column ClickBench table is ninety seven thousand parts, and a directory with a page entry and a
24//! pair of bounds per part per column is several hundred megabytes, which is what made that load
25//! fail before this split existed. Sixty four parts to a stripe divides that by sixty four.
26//!
27//! Where the parts of a page start is not in the directory either, for the same reason. Each
28//! stripe writes one index page holding a length and a checksum per part per column, and a reader
29//! preads the sixty four entries belonging to the column it wants. A scan reads the whole column
30//! page once and slices it; a sparse row fetch reads the index entries and then only the part it
31//! needs.
32
33#![forbid(unsafe_code)]
34
35use std::borrow::Cow;
36use std::cmp::Ordering;
37use std::collections::{BTreeMap, HashMap, VecDeque};
38use std::fs::{File, OpenOptions};
39use std::io::{Read, Seek, SeekFrom};
40use std::mem::{size_of, size_of_val};
41use std::path::Path;
42use std::slice;
43use std::sync::atomic::{AtomicBool, AtomicUsize, Ordering as Atomic};
44use std::sync::{Arc, Condvar, Mutex, OnceLock, PoisonError, Weak};
45
46use rudb_common::bounds::{self, Bound, Op, scaled_as};
47use rudb_common::{Clustering, Error, Field, LogicalType, PhysicalType, Result, Value, Width};
48use rudb_encoding::{bitpack, chooser, integer, string};
49use rudb_metrics::{LoadProfile, Stage};
50use rudb_storage::sieve::Sieve;
51use rudb_storage::{Probe, Range, Zone};
52use rudb_vector::string::StringColumn;
53use rudb_vector::validity::Validity;
54use rudb_vector::{Buffer, Chunk, Data, Packed, TextSource, Vector, search_below};
55
56mod distinct;
57pub mod graph;
58pub mod host;
59mod prepare;
60use prepare::Lent;
61pub mod section;
62pub mod stats;
63mod zones;
64
65pub use prepare::{Merged, Merger, Paged, Prepared, Preparer};
66pub use section::Section;
67pub use zones::{Common, Stripes, ascending, distincts};
68
69const MAGIC: &[u8; 8] = b"RUDBNV10";
70const DIRECTORY: &[u8; 8] = b"RUDBDI10";
71const CATALOG: &[u8; 8] = b"RUDBCA10";
72const NONZERO_COUNTS: &[u8; 8] = b"RUDBNZ10";
73const AGGREGATE_SUMS: &[u8; 8] = b"RUDBAG10";
74const DISTINCT_COUNTS: &[u8; 8] = b"RUDBDC10";
75const INTEGER_EXTREMES: &[u8; 8] = b"RUDBEX10";
76const COMPLETE_FREQUENCIES: &[u8; 8] = b"RUDBFQ10";
77const MAX_CATALOG_FREQUENCIES: usize = 64;
78const FORMAT: u32 = 28;
79
80/// Formats this build can open.
81///
82/// More than one, for the first time, and the reason is spec/graph/10-milestones.md's G1 exit
83/// criterion: a build with the section table in it has to open a file written before the section
84/// table existed, unchanged and without a rewrite. Formats 22 and 23 are those files, and both read
85/// as a table with an empty section table, which is exactly what section 3.1 says a table with no
86/// graph sections is.
87///
88/// All three of the older ones are readable for the same reason. What took the format from 22 to 23
89/// was tags for fourteen more column types, and a file written before that has none of them in it,
90/// so nothing in an older file is a tag this build cannot read. What took it from 23 to 24 is the
91/// section table, which a file written before it simply does not have. What takes it from 24 to 25
92/// is the view section on the end of the catalog, which an older file does not have either, and a
93/// catalog that ends where the tables end reads as a catalog with no views in it. What takes it
94/// from 25 to 26 is that a global dictionary's payload blocks now say where they are, and a file
95/// written before that has them behind one another, which [`open_global_dictionary`] reads by
96/// turning the ends it finds into the same places the newer files name outright. What takes it
97/// from 26 to 27 is that those blocks are written into the file as the load goes, between the
98/// stripes, rather than behind the dictionary's index at the end, so the dictionary's page is the
99/// index and the sorted order and nothing else. A format 26 file has its blocks inside the page,
100/// and the reader tells the two apart by whether the page has room left over for them.
101///
102/// Format 28 adds per-payload-block substring signatures to global string dictionaries. Older
103/// files have no signatures and use the ordinary exact string filter.
104///
105/// This is not a general compatibility promise. Seven formats are readable because there was a
106/// specific reason for each, and the list shrinks again the moment the older ones stop being worth
107/// carrying.
108const READABLE: &[u32] = &[22, 23, 24, 25, 26, 27, FORMAT];
109
110const HEADER: u64 = 80;
111const SLOT_BYTES: usize = 28;
112const MAX_PAGE: usize = 256 * 1024 * 1024;
113const MAX_DIRECTORY: usize = 128 * 1024 * 1024;
114const FREQUENCIES_V2: &[u8; 8] = b"RUDBFQ2\0";
115const FREQUENCIES: &[u8; 8] = b"RUDBFQ3\0";
116/// Inline spellings for string entries in the bounded frequency synopsis.
117///
118/// A planner usually asks about one literal such as the empty string. Without this block it opens
119/// a multi-million-value global dictionary and visits the payload blocks of every retained entry
120/// merely to compare that literal with at most 512 heavy hitters. The spellings are already in
121/// memory while the writer sorts the dictionary, so storing this bounded copy makes planning a
122/// directory read and leaves the dictionary unopened.
123const FREQUENCY_TEXTS: &[u8; 8] = b"RUDBFT1\0";
124/// Certified host aggregate state for the version-one anchored replacement expression.
125const HOST_GROUPS: &[u8; 8] = b"RUDBHG1\0";
126/// Exact leading counts for a bounded pair of dictionary-backed grouping keys.
127///
128/// This is a separate optional directory block rather than another frequency format. Readers that
129/// predate it still understand every earlier directory, and a table without a pair worth keeping
130/// writes no block at all.
131const PAIR_FREQUENCIES: &[u8; 8] = b"RUDBPF1\0";
132/// The clustering declaration, written after the frequencies and only when there is one.
133///
134/// No format bump for this, which is the convention the frequency section set in #728: a new
135/// optional trailing section with its own magic leaves every file that does not use it byte for
136/// byte what it was, and the version is bumped for a change to a layout that already exists, as
137/// #1029 did. A file with no declaration is the same bytes this build wrote yesterday.
138///
139/// The width byte in this block gained a fifth value for #1285, for a declaration that leaves the
140/// bucket to the row count, and that did not bump the format either. It is the one case where the
141/// reasoning needs saying out loud, because it is a new value in a layout that already exists
142/// rather than a new section. A build without it reading one of these says `clustering width
143/// tag differs` and refuses the table, which is what that message was written for. Bumping the
144/// format instead would have made every file this build writes unreadable to an older one, whether
145/// it has a declaration in it or not, to warn about a case that only arises when it does.
146const CLUSTERING: &[u8; 8] = b"RUDBCL1\0";
147/// The graph section table, written after the clustering declaration and written even when empty.
148///
149/// Same convention and the same reason as the block above it, with one difference: this one is
150/// always there, so a file written by this build says which sections it has rather than leaving a
151/// reader to infer it from where the bytes ran out. Section 3.1 of the graph spec is what makes
152/// that safe to add without a format bump, because a table with no sections answers every query
153/// the way it did before, only without the graph path.
154const SECTIONS: &[u8; 8] = b"RUDBSE1\0";
155/// How many bytes of each column's global dictionary live outside its page, written only when any do.
156///
157/// From format 27 a dictionary's payload blocks are written into the file while the load runs, so
158/// they sit between the stripes and the dictionary's page covers only its index and sorted order.
159/// Nothing needs the total to read the file, because the index names every block. It is here for
160/// what a file costs a column, which [`Reader::layout`] and the statistics budget both report, and
161/// which would otherwise lose most of the bytes of every large string column.
162const DICTIONARY_PAYLOADS: &[u8; 8] = b"RUDBDP1\0";
163
164/// The most sections one table's directory may name.
165///
166/// A relationship contributes at most three sections, so this bounds a table at a few thousand
167/// relationships, which is far past anything a schema has. The bound is here so that a torn
168/// directory naming four billion of them is refused at decode rather than turned into an
169/// allocation, the same reason the extent count has one.
170const MAX_SECTIONS: usize = 4096;
171const FREQUENCY_CANDIDATES: usize = 32_768;
172const FREQUENCY_ENTRIES: usize = 512;
173const FREQUENCY_BUILD_RANK: usize = 10;
174const FREQUENCY_ORDINALS: usize = 131_072;
175const MAX_PAIR_FREQUENCIES: usize = 1024;
176/// The most exact heavy-hitter text one column may copy into the directory.
177///
178/// A column with unusually large leading values keeps the old code-only synopsis instead. The
179/// optimization must never turn a valid load into a directory-size failure.
180const FREQUENCY_TEXT_BUDGET: usize = 1024 * 1024;
181/// The most threads the two per column passes at the end of a commit are spread over.
182///
183/// A table like `hits` has ninety numeric columns, so on a machine with more cores than this the
184/// cap is what decides how long the frequencies take rather than the columns are. It is here at all
185/// because each worker holds a candidate table and a decoded part, and a hundred of those at once
186/// on a narrow machine would be worse than waiting.
187const MAX_FREQUENCY_WORKERS: usize = 32;
188
189/// How many threads the passes at the end of a commit are spread over on this machine.
190fn close_workers() -> usize {
191    std::thread::available_parallelism().map_or(1, usize::from).min(MAX_FREQUENCY_WORKERS)
192}
193
194/// How many decoded bytes the global dictionaries closing at the same time may hold between them.
195///
196/// Closing a dictionary decodes every value it holds, sorts them and drops them, and #1356 took the
197/// columns one at a time so that five of them decoded at once were not the peak of a load. On the
198/// ClickBench `hits` 10M load that made the dictionaries 1.9 s of a 6.3 s load on the 32 core box,
199/// with `Referer`, `Title` and `URL` each most of a second on their own. A column is taken while the
200/// ones already closing leave room for it under this, and always when nothing else is closing, so
201/// every column of `hits` at 10M rows closes at once and `URL` at 100M, which is past this alone,
202/// still closes on its own.
203const CLOSE_DICTIONARY_BYTES: usize = 1 << 30;
204
205/// The most threads one stripe's encode is spread over.
206///
207/// Higher than the frequency cap because this is the load itself rather than a pass at the end of
208/// it, and the work is one column of sixty four parts, which is large enough that a thread that
209/// takes one is not a thread that was started for nothing. A machine with more cores than this has
210/// the rest of them on the Parquet read, which is still one thread and is the other half of #808.
211const MAX_ENCODE_WORKERS: usize = 32;
212
213/// How much a writer appends before it asks the kernel to start writing it to the device.
214///
215/// Without it every byte of a load waits in the page cache for the sync at the commit, and that
216/// sync was 1.3 to 1.7 s of a ClickBench `hits` 10M load of 8 to 9 s on the 32 core box. With it
217/// the device writes while the load is still encoding. Thirty two megabytes is a few stripes of
218/// `hits`, big enough that the call costs nothing next to the write, and small enough that what
219/// is left for the commit is one stretch.
220const WRITEBACK_STRETCH: u64 = 32 << 20;
221
222/// The most bytes one column of one part may spend on a membership sieve.
223///
224/// A part is a thousand rows, so a filter sized for every one of them being distinct is about
225/// thirteen hundred bytes and this never binds in practice. It is here so that a part that somehow
226/// arrives much wider than a vector cannot put an unbounded index in the file. What does bind is the
227/// rule in `Writer::encode_pages` that a sieve may not be as large as the part it indexes, which is a cap
228/// per column rather than one number for the whole file.
229const SIEVE_BUDGET: usize = 8 * 1024;
230
231/// The most bytes one end of a per part range may spend on a string.
232///
233/// A bound is allowed to be wider than the truth and never narrower, so a long string is cut down to
234/// this many bytes for the low end and cut down and then stepped up for the high end. The reason for
235/// a cap at all is that there are nine hundred and seventy four parts of a hundred and five columns
236/// in a million rows of ClickBench and `URL` runs to hundreds of bytes, so keeping every end whole
237/// would put more in the directory than the skipping is worth. Twenty four bytes is past the point
238/// where two URLs of the same site still look alike.
239const PART_BOUND_BYTES: usize = 24;
240
241fn io(error: std::io::Error) -> Error {
242    Error::io(error.to_string())
243}
244
245fn invalid(message: &str) -> Error {
246    Error::invalid_input(format!("invalid rudb native file: {message}"))
247}
248
249/// Adds a sequence of byte counts without an overflow the caller has to think about.
250fn sum(counts: impl Iterator<Item = u64>) -> u64 {
251    counts.fold(0, u64::saturating_add)
252}
253
254/// One column's span out of a per column list, or zero when the list is shorter than the column.
255fn span_bytes(spans: &[Span], at: usize) -> u64 {
256    spans.get(at).map_or(0, |span| u64::from(span.length))
257}
258
259/// One column's page out of a per column list, or zero when that column has no page at all.
260fn page_bytes(pages: &[Option<Page>], at: usize) -> u64 {
261    pages.get(at).and_then(Option::as_ref).map_or(0, Page::bytes)
262}
263
264/// Everything one column's global dictionary costs the file, its page and the blocks outside it.
265fn dictionary_bytes(table: &Table, at: usize) -> u64 {
266    page_bytes(&table.dictionaries, at)
267        .saturating_add(table.dictionary_payloads.get(at).copied().unwrap_or(0))
268}
269
270/// The xxHash64 of `bytes`, which is what every span this format stores is checked against.
271///
272/// It walks the input as chunks rather than as offsets into it, and that is the only thing about it
273/// worth a comment. The offset form reads `bytes[at..at + 8]`, and neither the slicing nor the
274/// `try_into` behind it can be proved in range by a compiler that does not know where `at` stopped,
275/// so each of the four lanes paid for a bounds check and a length check on every thirty two bytes.
276/// A chunk carries its own length, so both fold away and the loop is the multiplies and rotates it
277/// was meant to be. That loop runs over every byte of every span a query reads, which on ClickBench
278/// 8 is about five percent of the query.
279fn checksum(bytes: &[u8]) -> u64 {
280    seeded_checksum(bytes, 0)
281}
282
283/// A hundred and twenty eight bit name for `bytes`, as two xxHash64 walks under different seeds,
284/// with the format this build writes folded in so that a name made by one format is never taken
285/// for the name of a file in another.
286///
287/// For a caller outside this crate that has to name a file by what went into it, which is what a
288/// Parquet mirror's key is. See the global dictionary's use of the same pair for the arithmetic.
289#[must_use]
290pub fn content_name(bytes: &[u8]) -> u128 {
291    let seed = u64::from(FORMAT);
292    u128::from(seeded_checksum(bytes, seed)) << 64 | u128::from(seeded_checksum(bytes, !seed))
293}
294
295/// [`content_name`] of bytes that arrive in pieces, which gives the same name as the pieces joined.
296///
297/// A Parquet mirror is named for the file's footer, and that is 930 KB on the ten million row
298/// ClickBench file. Read whole to be hashed it is a freed megabyte in every process that opens the
299/// mirror, which the allocator keeps. Read a window at a time it is a window.
300#[derive(Debug, Clone)]
301pub struct ContentNamer {
302    seeds: [u64; 2],
303    lanes: [[u64; 4]; 2],
304    held: [u8; 32],
305    filled: usize,
306    length: u64,
307}
308
309impl Default for ContentNamer {
310    fn default() -> Self {
311        let seed = u64::from(FORMAT);
312        let seeds = [seed, !seed];
313        let lanes = seeds.map(|seed| {
314            [
315                seed.wrapping_add(XXH_P1).wrapping_add(XXH_P2),
316                seed.wrapping_add(XXH_P2),
317                seed,
318                seed.wrapping_sub(XXH_P1),
319            ]
320        });
321        Self { seeds, lanes, held: [0; 32], filled: 0, length: 0 }
322    }
323}
324
325impl ContentNamer {
326    /// Takes the next piece.
327    pub fn update(&mut self, mut bytes: &[u8]) {
328        self.length += bytes.len() as u64;
329        if self.filled > 0 {
330            let take = (32 - self.filled).min(bytes.len());
331            self.held[self.filled..self.filled + take].copy_from_slice(&bytes[..take]);
332            self.filled += take;
333            bytes = &bytes[take..];
334            if self.filled < 32 {
335                return;
336            }
337            let block = self.held;
338            self.lanes.iter_mut().for_each(|lanes| checksum_block(lanes, &block));
339            self.filled = 0;
340        }
341        let mut blocks = bytes.chunks_exact(32);
342        for block in blocks.by_ref() {
343            self.lanes.iter_mut().for_each(|lanes| checksum_block(lanes, block));
344        }
345        let rest = blocks.remainder();
346        self.held[..rest.len()].copy_from_slice(rest);
347        self.filled = rest.len();
348    }
349
350    /// The name of everything taken so far.
351    #[must_use]
352    pub fn finish(&self) -> u128 {
353        let rest = &self.held[..self.filled];
354        let [first, second] = [0, 1].map(|at| {
355            if self.length < 32 {
356                checksum_tail(self.seeds[at].wrapping_add(XXH_P5).wrapping_add(self.length), rest)
357            } else {
358                finish_checksum(self.lanes[at], rest, self.length)
359            }
360        });
361        u128::from(first) << 64 | u128::from(second)
362    }
363}
364
365/// The xxHash64 of `bytes` started from `seed`, which is the same walk with a different beginning.
366///
367/// A seed is here for one caller: a global dictionary decides whether two values are the same by
368/// their hashes rather than by their bytes, and one sixty four bit hash is not enough to do that
369/// with. Twenty million distinct values collide on sixty four bits about once in a hundred thousand
370/// loads, which for a wrong answer is far too often. Two hashes of the same value under different
371/// seeds are independent, so the pair is a hundred and twenty eight bits and the same arithmetic
372/// puts that at around one in 1e24.
373fn seeded_checksum(bytes: &[u8], seed: u64) -> u64 {
374    // Asked for before the loop rather than after it, because a `ChunksExact` settles what it
375    // cannot divide when it is built and hands back the same tail whether it has been walked or not.
376    let mut blocks = bytes.chunks_exact(32);
377    let rest = blocks.remainder();
378    if bytes.len() < 32 {
379        return checksum_tail(seed.wrapping_add(XXH_P5).wrapping_add(bytes.len() as u64), rest);
380    }
381    let mut lanes = [
382        seed.wrapping_add(XXH_P1).wrapping_add(XXH_P2),
383        seed.wrapping_add(XXH_P2),
384        seed,
385        seed.wrapping_sub(XXH_P1),
386    ];
387    for block in blocks.by_ref() {
388        checksum_block(&mut lanes, block);
389    }
390    finish_checksum(lanes, rest, bytes.len() as u64)
391}
392
393const XXH_P1: u64 = 11_400_714_785_074_694_791;
394const XXH_P2: u64 = 14_029_467_366_897_019_727;
395const XXH_P3: u64 = 1_609_587_929_392_839_161;
396const XXH_P4: u64 = 9_650_029_242_287_828_579;
397const XXH_P5: u64 = 2_870_177_450_012_600_261;
398
399fn checksum_round(state: u64, word: u64) -> u64 {
400    state.wrapping_add(word.wrapping_mul(XXH_P2)).rotate_left(31).wrapping_mul(XXH_P1)
401}
402
403fn checksum_word(chunk: &[u8]) -> u64 {
404    u64::from_le_bytes(chunk.try_into().expect("eight checksum bytes"))
405}
406
407/// One thirty two byte block into the four lanes.
408fn checksum_block(lanes: &mut [u64; 4], block: &[u8]) {
409    for (lane, chunk) in lanes.iter_mut().zip(block.chunks_exact(8)) {
410        *lane = checksum_round(*lane, checksum_word(chunk));
411    }
412}
413
414/// The lanes after every whole block, folded together with what was left over and the length.
415fn finish_checksum(lanes: [u64; 4], rest: &[u8], length: u64) -> u64 {
416    let merge = |state: u64, lane: u64| {
417        (state ^ checksum_round(0, lane)).wrapping_mul(XXH_P1).wrapping_add(XXH_P4)
418    };
419    let [one, two, three, four] = lanes;
420    let combined = one
421        .rotate_left(1)
422        .wrapping_add(two.rotate_left(7))
423        .wrapping_add(three.rotate_left(12))
424        .wrapping_add(four.rotate_left(18));
425    let hash = merge(merge(merge(merge(combined, one), two), three), four);
426    checksum_tail(hash.wrapping_add(length), rest)
427}
428
429/// The fewer than thirty two bytes after the last whole block, and the final mix.
430fn checksum_tail(mut hash: u64, mut rest: &[u8]) -> u64 {
431    let mut words = rest.chunks_exact(8);
432    for chunk in words.by_ref() {
433        hash ^= checksum_round(0, checksum_word(chunk));
434        hash = hash.rotate_left(27).wrapping_mul(XXH_P1).wrapping_add(XXH_P4);
435    }
436    rest = words.remainder();
437    if rest.len() >= 4 {
438        let (head, tail) = rest.split_at(4);
439        let quarter = u32::from_le_bytes(head.try_into().expect("four checksum bytes"));
440        hash ^= u64::from(quarter).wrapping_mul(XXH_P1);
441        hash = hash.rotate_left(23).wrapping_mul(XXH_P2).wrapping_add(XXH_P3);
442        rest = tail;
443    }
444    for &byte in rest {
445        hash ^= u64::from(byte).wrapping_mul(XXH_P5);
446        hash = hash.rotate_left(11).wrapping_mul(XXH_P1);
447    }
448    hash ^= hash >> 33;
449    hash = hash.wrapping_mul(XXH_P2);
450    hash ^= hash >> 29;
451    hash = hash.wrapping_mul(XXH_P3);
452    hash ^ (hash >> 32)
453}
454
455/// The checksum of `length` bytes of `file` from `offset`, read [`DIRECTORY_WINDOW`] at a time.
456///
457/// The same xxHash64 as [`checksum`], carried across reads rather than over one buffer, so that a
458/// directory can be checked without all of it being in memory at once. The four lanes take whole
459/// thirty two byte blocks, and a read that ends partway through one keeps the tail for the next.
460fn file_checksum(file: &File, offset: u64, length: usize) -> Result<u64> {
461    if length < 32 {
462        let mut bytes = vec![0; length];
463        read_at(file, offset, &mut bytes)?;
464        return Ok(checksum(&bytes));
465    }
466    let mut lanes = [XXH_P1.wrapping_add(XXH_P2), XXH_P2, 0, 0_u64.wrapping_sub(XXH_P1)];
467    let mut buffer = vec![0; DIRECTORY_WINDOW.min(length)];
468    let mut kept = 0;
469    let mut read = 0;
470    while read < length {
471        let want = (buffer.len() - kept).min(length - read);
472        read_at(file, offset + read as u64, &mut buffer[kept..kept + want])?;
473        read += want;
474        let filled = kept + want;
475        let whole = filled / 32 * 32;
476        for block in buffer[..whole].chunks_exact(32) {
477            checksum_block(&mut lanes, block);
478        }
479        buffer.copy_within(whole..filled, 0);
480        kept = filled - whole;
481    }
482    Ok(finish_checksum(lanes, &buffer[..kept], length as u64))
483}
484
485#[derive(Debug, Clone, Copy)]
486struct Slot {
487    offset: u64,
488    length: u32,
489    generation: u64,
490    hash: u64,
491}
492
493impl Slot {
494    fn bytes(self) -> [u8; SLOT_BYTES] {
495        let mut result = [0; SLOT_BYTES];
496        result[..8].copy_from_slice(&self.offset.to_le_bytes());
497        result[8..12].copy_from_slice(&self.length.to_le_bytes());
498        result[12..20].copy_from_slice(&self.generation.to_le_bytes());
499        result[20..28].copy_from_slice(&self.hash.to_le_bytes());
500        result
501    }
502
503    fn read(bytes: &[u8]) -> Self {
504        Self {
505            offset: u64::from_le_bytes(bytes[..8].try_into().expect("eight bytes")),
506            length: u32::from_le_bytes(bytes[8..12].try_into().expect("four bytes")),
507            generation: u64::from_le_bytes(bytes[12..20].try_into().expect("eight bytes")),
508            hash: u64::from_le_bytes(bytes[20..28].try_into().expect("eight bytes")),
509        }
510    }
511}
512
513#[derive(Debug, Clone, Copy)]
514struct Page {
515    offset: u64,
516    length: u32,
517    hash: u64,
518}
519
520impl Page {
521    /// How much of the file this page takes, for [`Reader::layout`].
522    fn bytes(&self) -> u64 {
523        u64::from(self.length)
524    }
525}
526
527#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
528enum FrequencyValue {
529    Null,
530    Integer(i128),
531    Code(u32),
532}
533
534/// A table keyed by the sixty four bits of the values the numeric frequency pass counts.
535///
536/// Every integer of every numeric column goes through one of these at least once when a table
537/// closes, and with the standard hasher that was a fifth of the close on its own, all of it SipHash
538/// guarding against an attacker who would have to choose the rows of the file being written.
539type FrequencyMap<V> = HashMap<u64, V, Spread>;
540
541/// The first pass of [`Writer::numeric_frequency`]: a Misra-Gries candidate table keyed by a value's
542/// sixty four bits, with the null counted beside it.
543#[derive(Debug, Default)]
544struct Candidates {
545    counts: FrequencyMap<u32>,
546    nulls: u32,
547    decrements: u64,
548}
549
550impl Candidates {
551    /// Counts `times` rows of `bits` and ends in the state `times` rows counted one at a time would.
552    ///
553    /// A value already held, or one there is room to hold, takes the whole run at once, because
554    /// every row after the first would find it held. A value the full table turns away goes a row
555    /// at a time, because each of its rows decrements every candidate and one of those decrements
556    /// can free the place the next row takes.
557    fn add(&mut self, bits: Option<u64>, mut times: u32) {
558        while times > 0 {
559            let held = match bits {
560                Some(bits) => self.counts.get_mut(&bits),
561                None if self.nulls != 0 => Some(&mut self.nulls),
562                None => None,
563            };
564            if let Some(count) = held {
565                *count = count.saturating_add(times);
566                return;
567            }
568            if self.counts.len() + usize::from(self.nulls != 0) < FREQUENCY_CANDIDATES {
569                match bits {
570                    Some(bits) => {
571                        self.counts.insert(bits, times);
572                    }
573                    None => self.nulls = times,
574                }
575                return;
576            }
577            self.counts.retain(|_, count| {
578                *count -= 1;
579                *count != 0
580            });
581            self.nulls = self.nulls.saturating_sub(1);
582            self.decrements = self.decrements.saturating_add(1);
583            times -= 1;
584        }
585    }
586}
587
588/// Equal rows in a row, gathered so they are counted once.
589#[derive(Debug, Default)]
590struct Run {
591    bits: Option<u64>,
592    times: u32,
593}
594
595impl Run {
596    /// Adds one row, and hands back the run it ended if it was not the same value.
597    fn push(&mut self, bits: Option<u64>) -> Option<(Option<u64>, u32)> {
598        if self.times != 0 && self.bits == bits && self.times < u32::MAX {
599            self.times += 1;
600            return None;
601        }
602        let ended = self.take();
603        self.bits = bits;
604        self.times = 1;
605        ended
606    }
607
608    /// The run being gathered, if there is one, leaving none.
609    fn take(&mut self) -> Option<(Option<u64>, u32)> {
610        let times = std::mem::take(&mut self.times);
611        (times != 0).then_some((self.bits, times))
612    }
613}
614
615/// Builds the hasher for [`FrequencyMap`].
616#[derive(Debug, Default, Clone, Copy)]
617struct Spread;
618
619impl std::hash::BuildHasher for Spread {
620    type Hasher = SpreadHasher;
621
622    fn build_hasher(&self) -> SpreadHasher {
623        SpreadHasher(0)
624    }
625}
626
627/// Folds each word in with a full width multiply whose two halves are xored together.
628///
629/// A plain multiply leaves the low bits of the hash as poor as the low bits of the key, and the
630/// table picks its bucket from the low bits, so a timestamp column, whose values are all multiples
631/// of a million microseconds, would pile into a sixty fourth of the buckets. Folding the high half
632/// of the product back in is what gives the low bits the whole word.
633#[derive(Debug)]
634struct SpreadHasher(u64);
635
636impl SpreadHasher {
637    fn mix(&mut self, word: u64) {
638        let product = u128::from(self.0 ^ word) * 0x9E37_79B9_7F4A_7C15_u128;
639        self.0 = (product as u64) ^ ((product >> 64) as u64);
640    }
641}
642
643impl std::hash::Hasher for SpreadHasher {
644    fn write(&mut self, bytes: &[u8]) {
645        for part in bytes.chunks(8) {
646            let mut word = [0; 8];
647            word[..part.len()].copy_from_slice(part);
648            self.mix(u64::from_le_bytes(word));
649        }
650    }
651
652    fn write_u32(&mut self, value: u32) {
653        self.mix(u64::from(value));
654    }
655
656    fn write_u64(&mut self, value: u64) {
657        self.mix(value);
658    }
659
660    fn write_i128(&mut self, value: i128) {
661        self.mix(value as u64);
662        self.mix((value >> 64) as u64);
663    }
664
665    fn write_isize(&mut self, value: isize) {
666        self.mix(value as u64);
667    }
668
669    fn finish(&self) -> u64 {
670        self.0
671    }
672}
673
674#[derive(Debug, Clone)]
675struct FrequencyEntry {
676    value: FrequencyValue,
677    count: u64,
678}
679
680/// Exact leading frequencies for one column.
681///
682/// Values outside `entries` occur at most `omitted_max` times. This lets a count-descending TopN
683/// use the synopsis only when its last winner is strictly above every omitted value.
684#[derive(Debug, Clone)]
685struct FrequencySummary {
686    entries: Vec<FrequencyEntry>,
687    omitted_max: u64,
688    ordinals: Vec<u64>,
689    ordinal_entries: Vec<u16>,
690}
691
692#[derive(Debug, Clone)]
693struct PairFrequencyEntry {
694    first_entry: u16,
695    second: Option<u32>,
696    count: u64,
697}
698
699/// Exact leading counts for one numeric frequency anchor and one stable string code space.
700///
701/// `omitted_max` covers both first-key values outside the numeric synopsis and pairs below the
702/// retained prefix. A TopN may therefore use the entries only when its boundary strictly exceeds
703/// this number.
704#[derive(Debug, Clone)]
705struct PairFrequencySummary {
706    first: u16,
707    second: u16,
708    entries: Vec<PairFrequencyEntry>,
709    omitted_max: u64,
710}
711
712/// One column's frequency synopsis, in memory or left where it is in the file.
713///
714/// A writer holds what it counted. A reader leaves every synopsis in the file and reads one back
715/// when a query asks about its column, because they are the largest thing in a directory once they
716/// are decoded, forty eight bytes an entry and nearly twenty thousand entries over `hits`, and
717/// most queries ask about none of them. Where one sits is found at open, by reading it through and
718/// checking it, so a torn synopsis is still refused when the table is opened.
719#[derive(Debug, Clone)]
720enum Frequencies {
721    Held(FrequencySummary),
722    /// Where the synopsis sits, and whether it was written with the value of each ordinal, which
723    /// is what the directory's frequency magic says and the synopsis itself does not.
724    Stored {
725        span: Span,
726        values: bool,
727    },
728}
729
730/// The values one column's frequency synopsis lists, with a bound on everything it left out.
731///
732/// What [`Reader::frequency_prefix`] answers. The counts are exact, and `omitted_max` is how many
733/// rows any value not in the list can hold, which is zero when nothing was left out at all.
734#[derive(Debug, Clone)]
735pub struct FrequencyPrefix {
736    /// Every value the synopsis lists, with the number of rows holding it, count descending.
737    pub entries: Vec<(Value, u64)>,
738    /// How many rows the most common value outside the list holds, and zero for a complete list.
739    pub omitted_max: u64,
740}
741
742/// Sparse row ordinals covered by a numeric frequency candidate set.
743#[derive(Debug, Clone, PartialEq)]
744pub struct FrequencyOccurrences {
745    /// Upper bound for the frequency of every value absent from the fetched rows.
746    pub omitted_max: u64,
747    /// Table-wide row ordinals in ascending order.
748    pub ordinals: Vec<u64>,
749    /// The retained heavy-hitter values named by `anchor_indices`.
750    pub anchors: Vec<Value>,
751    /// The index in `anchors` at each ordinal, or empty for a legacy FQ2 directory.
752    pub anchor_indices: Vec<u16>,
753}
754
755/// Exact grouped counts for a pair of values, in descending count order.
756pub type PairFrequencyCounts = Vec<(Vec<Value>, u64)>;
757
758/// Where one column's page for one stripe sits in the file.
759///
760/// A column page has no checksum of its own because every part inside it carries one, and the
761/// stripe's index page holds those. Checking a part on the way out of the page covers exactly the
762/// bytes a reader is about to decode, and covers them once whether the reader took the whole page
763/// or pulled one part out of the middle of it.
764#[derive(Debug, Clone, Copy, Default)]
765struct Span {
766    offset: u64,
767    length: u32,
768}
769
770/// One optional page for each column of a stripe, holding only the pages that are there.
771///
772/// A stripe has three of these, the membership, sieve and part range pages. As a
773/// `Vec<Option<Page>>` each was thirty two bytes a column whether the page was there or not, and
774/// over the ten million rows of `hits` that is half a megabyte at open for 7171 pages out of 16380
775/// slots. Kept sparse and packed, a page that is there is twenty four bytes and one that is not is
776/// nothing.
777#[derive(Debug, Clone, Default)]
778struct Pages {
779    columns: usize,
780    held: Box<[StripePage]>,
781}
782
783/// A page and the column it is for, packed so that the column sits where the padding was.
784#[derive(Debug, Clone, Copy)]
785struct StripePage {
786    offset: u64,
787    hash: u64,
788    length: u32,
789    column: u32,
790}
791
792impl Pages {
793    /// The pages of `columns` columns, one slot each in column order.
794    fn from_slots(slots: Vec<Option<Page>>) -> Result<Self> {
795        let mut held = Vec::with_capacity(slots.iter().flatten().count());
796        for (column, page) in slots.iter().enumerate() {
797            if let Some(page) = page {
798                let column =
799                    u32::try_from(column).map_err(|_| invalid("too many columns for a page"))?;
800                held.push(StripePage {
801                    offset: page.offset,
802                    hash: page.hash,
803                    length: page.length,
804                    column,
805                });
806            }
807        }
808        Ok(Self { columns: slots.len(), held: held.into_boxed_slice() })
809    }
810
811    /// The page of one column, if it has one.
812    fn get(&self, column: usize) -> Option<Page> {
813        let at = self.held.binary_search_by_key(&column, |placed| placed.column as usize).ok()?;
814        let placed = self.held[at];
815        Some(Page { offset: placed.offset, length: placed.length, hash: placed.hash })
816    }
817
818    /// One slot per column, in column order, the way the directory writes them.
819    fn slots(&self) -> impl Iterator<Item = Option<Page>> + '_ {
820        (0..self.columns).map(|column| self.get(column))
821    }
822
823    /// How much of the file one column's page takes, or zero when it has none.
824    fn bytes(&self, column: usize) -> u64 {
825        self.get(column).map_or(0, |page| page.bytes())
826    }
827}
828
829/// One independently readable stripe of a table.
830#[derive(Debug, Clone)]
831pub struct Stripe {
832    rows: usize,
833    /// Rows in each part, in source order. Kept in the directory so that mapping a row ordinal to a
834    /// part, which every sparse fetch does, never reads the file.
835    parts: Vec<u32>,
836    /// The index page: one section per column, holding a length and a checksum for every part and
837    /// then a checksum of the section itself, so that a reader can pread one column's section and
838    /// still know it is intact.
839    index: Span,
840    pages: Vec<Span>,
841    memberships: Pages,
842    /// One page per column holding the membership sieve of every part of the stripe, for the
843    /// columns that have one. A column whose parts all declined a sieve has no page at all.
844    sieves: Pages,
845    /// One page per column holding the two ends and the null count of every part of the stripe.
846    ///
847    /// The stripe's own `zone` below covers sixty four times as many rows, and on a column that is
848    /// not the one the rows are ordered by that is the difference between skipping half the file and
849    /// skipping all but three percent of it. On ClickBench 24 the cutoff the answer settles at
850    /// leaves eight stripes of sixteen alive and thirty parts of nine hundred and seventy four.
851    ///
852    /// A page per column rather than one page for the stripe, so that a query that compares one
853    /// column reads the ends of that column and not of the hundred and four beside it. Read lazily
854    /// for the same reason, like the sieves.
855    part_ranges: Pages,
856    zone: Zone,
857}
858
859impl Stripe {
860    /// Number of rows in this stripe.
861    #[must_use]
862    pub fn rows(&self) -> usize {
863        self.rows
864    }
865
866    /// Number of parts in this stripe.
867    #[must_use]
868    pub fn parts(&self) -> usize {
869        self.parts.len()
870    }
871
872    /// The two ends and the null count of every column over the whole stripe.
873    ///
874    /// In the directory and so in memory, which is what makes it the one a planner can ask. The
875    /// finer ones are a page per column per stripe in the file, read by [`Reader::skips`] when a
876    /// scan wants to know which parts to open.
877    #[must_use]
878    pub fn zone(&self) -> &Zone {
879        &self.zone
880    }
881}
882
883/// The committed table directory.
884#[derive(Debug, Clone)]
885pub struct Table {
886    name: String,
887    fields: Vec<Field>,
888    stripes: Vec<Stripe>,
889    rows: usize,
890    dictionaries: Vec<Option<Page>>,
891    /// Bytes of each column's dictionary payload that are outside its page, which is all of them
892    /// from format 27 and none of them before. See [`DICTIONARY_PAYLOADS`].
893    ///
894    /// Empty rather than a row of zeros on a table that has none, and read with `get` for that
895    /// reason, so that a table built by hand in a test does not have to know about it.
896    dictionary_payloads: Vec<u64>,
897    frequencies: Vec<Option<Frequencies>>,
898    pair_frequencies: Vec<PairFrequencySummary>,
899    /// String spellings aligned with each column's frequency entries.
900    ///
901    /// Empty for files written before `RUDBFT1`. A `None` entry is the null frequency entry; every
902    /// code entry in a column named by the block has its exact bytes here.
903    frequency_texts: Vec<Vec<Option<Vec<u8>>>>,
904    /// Exact candidate host aggregates and an upper bound for every omitted host.
905    host_groups: Option<host::HostSummary>,
906    /// How many distinct values each column holds, for the columns that know.
907    ///
908    /// A dictionary entry is made the first time a value is seen and nothing ever removes one, so
909    /// the size of the dictionary is the number of distinct values in the column. That is the whole
910    /// story for a column with no null in it, and the wrong number by one for a column with a null
911    /// in it, because a null row is written as the code for the empty string and makes an entry the
912    /// dictionary would not otherwise have. The writer knows which case it is, since it counts the
913    /// non-null rows that use each code while it builds the frequency summary, and the reader cannot
914    /// work it out from the dictionary alone. So the writer settles it here.
915    distincts: Vec<Option<u64>>,
916    /// The order the rows of this table are meant to be stored in, if anybody declared one.
917    ///
918    /// A declaration and not a measurement. Nothing here checks that the stripes actually arrived
919    /// in this order, and the reason it is worth storing anyway is that the order is the only thing
920    /// about a table that a rewrite destroys without anybody noticing. The fragment ranges prune on
921    /// whatever order the rows came in, so a table loaded sorted prunes and the same table after a
922    /// checkpoint that did not know to keep the order quietly stops pruning and nothing says why.
923    clustering: Option<Clustering>,
924    /// The file generation of the commit that last wrote this table's column pages.
925    ///
926    /// This is what spec/graph/03-the-file-format.md section 3.2 calls the table generation, and
927    /// the definition is deliberately about the pages rather than about the directory. A graph
928    /// section is a restatement of a column in terms of row ids, so what invalidates one is the
929    /// rows being renumbered, and nothing else. Adding a second table to the file, or attaching a
930    /// section to this one, commits a new file generation without touching a single row of this
931    /// table, and a definition that moved with those would declare every section in the file stale
932    /// for no reason.
933    ///
934    /// Zero on a table written before format 23, where nothing recorded it. Real generations start
935    /// at one, so zero can never match a section's stamp, and a table from format 22 has no
936    /// sections for it to match anyway.
937    generation: u64,
938    /// The graph sections this table carries, per spec/graph/03-the-file-format.md section 3.2.
939    ///
940    /// Empty for every table written before the section table existed, and empty is not a
941    /// degraded state: section 3.1 says deleting every graph section from a file changes no answer,
942    /// only the time, so a table with none here answers every query the same way and slower. That
943    /// is what lets this field arrive without a migration.
944    sections: Vec<Section>,
945}
946
947impl Table {
948    /// The SQL table name held by this snapshot.
949    #[must_use]
950    pub fn name(&self) -> &str {
951        &self.name
952    }
953
954    /// Columns in their SQL order.
955    #[must_use]
956    pub fn fields(&self) -> &[Field] {
957        &self.fields
958    }
959
960    /// Committed row count.
961    #[must_use]
962    pub fn rows(&self) -> usize {
963        self.rows
964    }
965
966    /// Independently readable stripes.
967    #[must_use]
968    pub fn stripes(&self) -> &[Stripe] {
969        &self.stripes
970    }
971
972    /// The order the rows are meant to be stored in, if this table was declared with one.
973    #[must_use]
974    pub fn clustering(&self) -> Option<&Clustering> {
975        self.clustering.as_ref()
976    }
977
978    /// The generation every section of this table is judged against.
979    ///
980    /// See the field. A caller deciding whether to read a section asks [`Section::usable`] with
981    /// this.
982    #[must_use]
983    pub fn generation(&self) -> u64 {
984        self.generation
985    }
986
987    /// Every graph section this table names, including the kinds this build does not know.
988    ///
989    /// Including them is the point. A caller that wants only the ones it can use asks
990    /// [`Section::usable`], and a caller rewriting the directory carries the rest through, so a
991    /// file opened by an older build and written again does not silently lose a section that build
992    /// had no name for.
993    #[must_use]
994    pub fn sections(&self) -> &[Section] {
995        &self.sections
996    }
997}
998
999/// One table's line in the catalog directory.
1000///
1001/// The small level of the two. It holds what opening a database needs and nothing else: the name to
1002/// bind, the shape to plan against, the row count, and where the table's own directory sits. A file
1003/// of eight tables is eight of these, and reading them costs the same whether the tables hold a
1004/// thousand rows or a billion.
1005///
1006/// The name, the fields and the row count are repeated here rather than pointed at inside the table
1007/// directory, which is the entire point of having two levels. A catalog that pointed at them would
1008/// have to read every table directory at open to answer what tables there are, which is the cost
1009/// this level exists to avoid.
1010#[derive(Debug, Clone)]
1011struct Entry {
1012    name: String,
1013    fields: Vec<Field>,
1014    rows: usize,
1015    /// Where this table's own directory sits, with the checksum it was committed under.
1016    directory: Page,
1017    /// Exact non-null, nonzero integer counts certified by the catalog checksum.
1018    nonzero: Vec<Option<u64>>,
1019    /// Exact sum and non-null count for signed integer columns.
1020    aggregates: Vec<Option<(i128, u64)>>,
1021    /// Exact non-null distinct values when the writer finished counting the column.
1022    distincts: Vec<Option<u64>>,
1023    /// Exact integer or date bounds; the inner `None` means every row is null.
1024    extremes: Vec<StoredIntegerExtremes>,
1025    /// Complete bounded numeric frequencies, including NULL when present.
1026    frequencies: Vec<StoredNumericFrequencies>,
1027}
1028
1029type StoredIntegerExtremes = Option<Option<(i128, i128)>>;
1030type StoredNumericFrequencies = Option<NumericFrequencies>;
1031
1032/// One view's line in the catalog directory.
1033///
1034/// A view has no pages, so unlike a table it is entirely here and there is no second level under it.
1035/// What it is made of is text: the body the binder binds again at every reference, and the whole
1036/// statement written back out, which is what `duckdb_views()` reports and nothing else reads.
1037///
1038/// The columns are a cache and they are written down anyway, which is worth saying out loud because
1039/// a cache in a file looks like a mistake. It is what the pin does. Create a view on a file, open
1040/// the file again in another process, and `duckdb_views()` answers `column_count` and `is_bound`
1041/// true without anything having bound the body, so the list survived the write. Not writing it
1042/// would answer null and false there, and the only way back would be to bind every view at open,
1043/// which is the thing the cache exists to avoid.
1044#[derive(Debug, Clone, PartialEq, Eq)]
1045pub struct ViewEntry {
1046    /// The view's own name, without the schema, the way a table entry holds its name.
1047    pub name: String,
1048    /// The query the view stands for, as the text that was written.
1049    pub sql: String,
1050    /// The whole `CREATE VIEW` written back out.
1051    pub statement: String,
1052    /// The column names the statement gave, which rename a prefix of what the body produces.
1053    pub aliases: Vec<String>,
1054    /// The columns the last bind of the body produced.
1055    pub columns: Vec<Field>,
1056}
1057
1058/// Where one column's bytes went, taken from the directory rather than by reading pages.
1059#[derive(Debug, Clone)]
1060pub struct ColumnLayout {
1061    /// The column's name, so a report does not have to carry the field list beside this.
1062    pub name: String,
1063    /// The type, spelled the way the catalog spells it.
1064    pub kind: String,
1065    /// Every stripe's page of this column added up, which is the encoded data itself.
1066    pub pages: u64,
1067    /// Every stripe's exact code membership page for this column.
1068    pub memberships: u64,
1069    /// Every stripe's membership sieve page for this column.
1070    pub sieves: u64,
1071    /// Every stripe's per part range page for this column.
1072    pub part_ranges: u64,
1073    /// The table wide dictionary of this column, if it has one.
1074    pub dictionary: u64,
1075}
1076
1077impl ColumnLayout {
1078    /// Everything this column costs, which is what the file would lose if the column went.
1079    #[must_use]
1080    pub fn total(&self) -> u64 {
1081        self.pages
1082            .saturating_add(self.memberships)
1083            .saturating_add(self.sieves)
1084            .saturating_add(self.part_ranges)
1085            .saturating_add(self.dictionary)
1086    }
1087}
1088
1089/// Where a whole file's bytes went.
1090///
1091/// Every number here comes out of the committed directory, so taking it costs one directory read
1092/// however large the file is. That is the point: a 45 GB table has to be able to say where it went
1093/// without being read, or nobody will ask.
1094///
1095/// The parts that are not a column are kept apart rather than shared out over the columns. The
1096/// stripe index page holds a section per column and could be split, and the directory and the
1097/// header cannot be, so splitting one of the three and not the others would read as if the columns
1098/// accounted for everything. They do not, and the gap is the thing worth looking at.
1099#[derive(Debug, Clone)]
1100pub struct Layout {
1101    /// The size of the file on disk.
1102    pub file: u64,
1103    /// Committed rows.
1104    pub rows: usize,
1105    /// Committed stripes.
1106    pub stripes: usize,
1107    /// Committed parts, which is how many chunks a scan reads.
1108    pub parts: usize,
1109    /// One entry per column, in the table's column order.
1110    pub columns: Vec<ColumnLayout>,
1111    /// Every stripe's index page, which carries a length and a checksum for every part of every
1112    /// column and is charged per stripe rather than per column.
1113    pub indexes: u64,
1114    /// The committed directory itself, the one that was read to build this.
1115    pub directory: u64,
1116    /// The fixed header, which holds the magic, the format and the two directory slots.
1117    pub header: u64,
1118}
1119
1120impl Layout {
1121    /// Everything the columns cost together.
1122    #[must_use]
1123    pub fn columns_total(&self) -> u64 {
1124        self.columns.iter().map(ColumnLayout::total).fold(0, u64::saturating_add)
1125    }
1126
1127    /// What the file holds that this does not account for.
1128    ///
1129    /// A committed file is written once and never rewritten in place, so an earlier directory and
1130    /// the pages of an earlier snapshot are still in it. That is the honest place for them: they
1131    /// are bytes on disk that no column owns.
1132    #[must_use]
1133    pub fn unaccounted(&self) -> u64 {
1134        self.file
1135            .saturating_sub(self.columns_total())
1136            .saturating_sub(self.indexes)
1137            .saturating_sub(self.directory)
1138            .saturating_sub(self.header)
1139    }
1140}
1141
1142/// How one part of one column is stored, which is one row of `pragma_storage_info`.
1143///
1144/// Everything here is read off the file rather than worked out from the schema, because the whole
1145/// question this answers is what the encoder chose, and the encoder chooses per part. Two files
1146/// holding the same rows in a different order give different answers and that difference is the
1147/// reason to ask.
1148///
1149/// The encoding costs a read of the column's page, so this is not free the way [`Layout`] is. It is
1150/// one read per column per stripe rather than one per part, because a part is a few kilobytes out
1151/// of a page that is a quarter of a megabyte.
1152#[derive(Debug, Clone)]
1153pub struct StoredPart {
1154    /// Which stripe the part belongs to.
1155    pub stripe: usize,
1156    /// Which part of that stripe it is, counting from zero inside the stripe.
1157    pub part: usize,
1158    /// The table wide row number the part starts at.
1159    pub row: usize,
1160    /// How many rows it holds.
1161    pub rows: usize,
1162    /// What the encoder made of it, as a line of text like `DICT(PACKED, PACKED)`.
1163    pub encoding: String,
1164    /// The stored bytes of the part, which is what it costs in the file.
1165    pub bytes: u64,
1166    /// Where in the file the column page holding this part starts.
1167    pub page: u64,
1168    /// Where in that page the part starts.
1169    pub offset: u64,
1170    /// The smallest value the part holds, when the stored ranges say.
1171    pub low: Option<Value>,
1172    /// The largest, same.
1173    pub high: Option<Value>,
1174    /// How many of its rows are null, when the stored ranges say.
1175    pub nulls: Option<usize>,
1176}
1177
1178/// Seeds the second hash a global dictionary tells its values apart by.
1179///
1180/// Any value that is not zero does, since zero is the seed [`checksum`] already uses and the point
1181/// is only that the two hashes of one value are not the same number. This one is the fractional part
1182/// of the golden ratio in sixty four bits, which is the constant everything else here is built out
1183/// of and is as good a nothing-up-my-sleeve number as any.
1184const DICTIONARY_CHECK_SEED: u64 = 11_400_714_819_323_198_485;
1185
1186/// One column's table wide dictionary while the load is running.
1187///
1188/// The thing to understand about this is what it does not hold. A dictionary of `URL` at a hundred
1189/// million ClickBench rows has about eighteen million distinct values and 1.3 GB of bytes in them,
1190/// and five columns like it are twelve of the seventeen gigabytes a load of `hits` peaks at. So the
1191/// bytes are not kept. A value's bytes go into [`GlobalDictionary::filling`], and when that reaches
1192/// [`TEXT_PAYLOAD_VALUES`] values the block is sealed, handed out at the end of the merge that
1193/// sealed it to be encoded with the stripe's pages, and never seen in that form again. What is left
1194/// is the encoded block, which is two to three times smaller, and that is the same bytes the file
1195/// is going to hold anyway.
1196///
1197/// Two things needed the raw bytes and neither needs them now. Deciding whether a value has been
1198/// seen before was a hash lookup and then a comparison of the bytes, and is now a hash lookup and a
1199/// comparison of a second hash under a different seed, which is [`DICTIONARY_CHECK_SEED`] and the
1200/// argument for why that is sound. Sorting the values at the end needed all of them at once, and
1201/// now reads the blocks back through [`GlobalDictionary::decoded`] one column at a time, which is
1202/// one column's bytes rather than every column's.
1203///
1204/// The offsets going block relative comes free with it, and takes the four gigabyte wall with it.
1205/// They were `u32` into a per column payload, so a column could not hold more than four gigabytes of
1206/// values however much memory the machine had, and `URL` and `Referer` are within a small factor of
1207/// that at a hundred million rows. A `u32` into a block of 1,024 values is not a bound anything real
1208/// reaches. The stored form is unchanged, because [`encode_offsets`] was already subtracting a per
1209/// block base before writing.
1210#[derive(Debug)]
1211struct GlobalDictionary {
1212    primary: HashMap<u64, u32>,
1213    collisions: HashMap<u64, Vec<u32>>,
1214    /// Every value's hash under [`DICTIONARY_CHECK_SEED`], in code order.
1215    checks: Vec<u64>,
1216    /// Where every value ends inside the payload block it is in, in code order.
1217    ends: Vec<u32>,
1218    counts: Vec<u64>,
1219    nulls: u64,
1220    /// The values of the block being filled, back to back.
1221    filling: Vec<u8>,
1222    /// One conservative four-byte substring signature per encoded payload block, in block order.
1223    ///
1224    /// Made where the block is encoded rather than where it is sealed, because sealing is under the
1225    /// writer's lock and every byte of every value going through [`gram_bits`] was 2.9 of the 14
1226    /// seconds the 10m ClickBench load spent on the 32 core box.
1227    grams: Vec<[u8; TEXT_GRAM_BYTES]>,
1228    /// Blocks that have filled and not been handed out to be encoded yet, each with its block number.
1229    ///
1230    /// Empty except inside the merge that filled them, and while the column is still too small to
1231    /// settle a shape on.
1232    waiting: Vec<(usize, Vec<u8>)>,
1233    /// Blocks kept raw to settle a shape on, spread across the column, each with its number.
1234    ///
1235    /// At most [`PAYLOAD_SAMPLE_BLOCKS`] of them and so at most a few megabytes. Spread rather than
1236    /// taken off the front for the reason [`settle_shape`] gives, and kept rather than read back
1237    /// because reading back is a decode and this is a sample of a column that is still growing.
1238    sample: Vec<(usize, Vec<u8>)>,
1239    /// How far apart the blocks in `sample` are, which doubles every time there are too many.
1240    stride: usize,
1241    /// What the blocks encoded so far were encoded with, once the column is big enough to settle it.
1242    shape: Option<chooser::Settled>,
1243    /// How many blocks had filled when that shape was settled.
1244    settled: usize,
1245    /// The blocks that are encoded and not yet in the file, in block order, following `placed`.
1246    ///
1247    /// Empty between stripes, because [`Writer::place_blocks`] writes them the moment they come
1248    /// back. Only a dictionary that never meets a writer, which is a test's, keeps them here.
1249    blocks: Vec<Vec<u8>>,
1250    /// Blocks that came back encoded ahead of a block before them, by block number.
1251    ///
1252    /// Two stripes merged one after the other can have their pages built in the other order, and a
1253    /// block cannot go into `blocks` until every block before it is there. They wait here until the
1254    /// gap closes, which is at most until the stripe merged just before this one is written.
1255    early: BTreeMap<usize, EncodedBlock>,
1256    /// Where every block already written to the file is, in block order.
1257    placed: Vec<Placed>,
1258}
1259
1260/// Where one payload block of a global dictionary is in the file, and its checksum.
1261#[derive(Debug, Clone, Copy)]
1262struct Placed {
1263    start: u64,
1264    length: u64,
1265    hash: u64,
1266}
1267
1268/// Sorted `(head, code)` entries and the decoded bytes and block bases they were sorted over.
1269type RankedDictionary = (Vec<(u64, u32)>, Vec<u8>, Vec<u64>);
1270
1271impl GlobalDictionary {
1272    fn new() -> Self {
1273        Self {
1274            primary: HashMap::new(),
1275            collisions: HashMap::new(),
1276            checks: Vec::new(),
1277            ends: Vec::new(),
1278            counts: Vec::new(),
1279            nulls: 0,
1280            filling: Vec::new(),
1281            grams: Vec::new(),
1282            waiting: Vec::new(),
1283            sample: Vec::new(),
1284            stride: 1,
1285            shape: None,
1286            settled: 0,
1287            blocks: Vec::new(),
1288            early: BTreeMap::new(),
1289            placed: Vec::new(),
1290        }
1291    }
1292
1293    /// How many distinct values this dictionary holds, which is one past its largest code.
1294    fn values(&self) -> usize {
1295        self.ends.len()
1296    }
1297
1298    /// About how many bytes closing this dictionary holds at once: every value decoded, and a
1299    /// sort entry and a code for each.
1300    fn closing_bytes(&self) -> usize {
1301        let values = self.values();
1302        let decoded = (0..values.div_ceil(TEXT_PAYLOAD_VALUES))
1303            .map(|block| self.ends[((block + 1) * TEXT_PAYLOAD_VALUES).min(values) - 1] as usize)
1304            .sum::<usize>();
1305        decoded.saturating_add(values.saturating_mul(size_of::<(u64, u32)>() + size_of::<u32>()))
1306    }
1307
1308    /// How many blocks are encoded, written or not, which is the number the next one has to have.
1309    fn encoded(&self) -> usize {
1310        self.placed.len() + self.blocks.len()
1311    }
1312
1313    #[cfg(test)]
1314    fn code(&mut self, text: &str) -> Result<u32> {
1315        let bytes = text.as_bytes();
1316        self.code_hashed(bytes, checksum(bytes), seeded_checksum(bytes, DICTIONARY_CHECK_SEED))
1317    }
1318
1319    /// The code for a value whose two hashes the caller already has.
1320    ///
1321    /// A stripe prepared outside the writer's lock hashed every value it holds while it was coding
1322    /// them, and merging it into this dictionary is one of these a distinct value rather than two
1323    /// hashes of every row. See [`prepare`].
1324    fn code_hashed(&mut self, text: &[u8], hash: u64, check: u64) -> Result<u32> {
1325        if let Some(&code) = self.primary.get(&hash) {
1326            if self.checks.get(code as usize) == Some(&check) {
1327                return Ok(code);
1328            }
1329            if let Some(codes) = self.collisions.get(&hash) {
1330                if let Some(code) =
1331                    codes.iter().copied().find(|&code| self.checks[code as usize] == check)
1332                {
1333                    return Ok(code);
1334                }
1335            }
1336            let code = self.insert(text, check)?;
1337            self.collisions.entry(hash).or_default().push(code);
1338            return Ok(code);
1339        }
1340        let code = self.insert(text, check)?;
1341        self.primary.insert(hash, code);
1342        Ok(code)
1343    }
1344
1345    fn insert(&mut self, text: &[u8], check: u64) -> Result<u32> {
1346        let code = u32::try_from(self.ends.len())
1347            .map_err(|_| invalid("global dictionary has too many values"))?;
1348        self.filling.extend_from_slice(text);
1349        self.ends.push(
1350            u32::try_from(self.filling.len())
1351                .map_err(|_| invalid("a global dictionary value exceeds 4 GiB"))?,
1352        );
1353        self.checks.push(check);
1354        self.counts.push(0);
1355        if self.ends.len() % TEXT_PAYLOAD_VALUES == 0 {
1356            self.seal();
1357        }
1358        Ok(code)
1359    }
1360
1361    /// Closes the block being filled and puts it in the queue to be encoded.
1362    ///
1363    /// Also keeps a copy of it if it lands on the sample's stride, and halves the sample when that
1364    /// has left too many, which is what keeps the kept blocks spread evenly over however much of the
1365    /// column exists rather than bunched at whichever end was cheap to remember.
1366    fn seal(&mut self) {
1367        let at = self.ends.len().div_ceil(TEXT_PAYLOAD_VALUES) - 1;
1368        let bytes = std::mem::take(&mut self.filling);
1369        if at % self.stride == 0 {
1370            self.sample.push((at, bytes.clone()));
1371            if self.sample.len() > PAYLOAD_SAMPLE_BLOCKS {
1372                self.stride *= 2;
1373                let stride = self.stride;
1374                self.sample.retain(|(at, _)| at % stride == 0);
1375            }
1376        }
1377        self.waiting.push((at, bytes));
1378    }
1379
1380    /// The values of one block, as slices into the bytes the block was filled with.
1381    fn slices<'a>(&self, at: usize, bytes: &'a [u8]) -> Vec<&'a [u8]> {
1382        block_values(self.block_ends(at), bytes)
1383    }
1384
1385    /// Where every value of one block ends, relative to the block.
1386    fn block_ends(&self, at: usize) -> &[u32] {
1387        let first = (at * TEXT_PAYLOAD_VALUES).min(self.ends.len());
1388        let last = (first + TEXT_PAYLOAD_VALUES).min(self.ends.len());
1389        &self.ends[first..last]
1390    }
1391
1392    /// Takes every waiting block out to be encoded somewhere else, if the column has a shape to
1393    /// encode them with.
1394    ///
1395    /// This is what keeps the encoding out of the writer's lock. A block needs its bytes, where its
1396    /// values end and the shape, and nothing else of the dictionary, so it goes out with a copy of
1397    /// the four kilobytes of ends it has and comes back through [`GlobalDictionary::take_back`].
1398    fn hand_out(&mut self, column: usize) -> Vec<Unencoded> {
1399        let Some(shape) = &self.shape else { return Vec::new() };
1400        let waiting = std::mem::take(&mut self.waiting);
1401        waiting
1402            .into_iter()
1403            .map(|(at, bytes)| Unencoded {
1404                column,
1405                at,
1406                ends: self.block_ends(at).to_vec(),
1407                bytes,
1408                shape: shape.clone(),
1409            })
1410            .collect()
1411    }
1412
1413    /// Takes back one block that was handed out, and moves every block that is now next in line
1414    /// into `blocks`.
1415    fn take_back(&mut self, at: usize, block: EncodedBlock) -> Result<()> {
1416        if at < self.encoded() || self.early.insert(at, block).is_some() {
1417            return Err(Error::internal("a dictionary block came back twice"));
1418        }
1419        while let Some(block) = self.early.remove(&self.encoded()) {
1420            self.push_block(block);
1421        }
1422        Ok(())
1423    }
1424
1425    /// Appends the next encoded block and its signature.
1426    fn push_block(&mut self, (bytes, grams): EncodedBlock) {
1427        self.blocks.push(bytes);
1428        self.grams.push(*grams);
1429    }
1430
1431    /// Settles the shape the waiting blocks are about to be encoded with, if there is enough column
1432    /// to settle one on.
1433    ///
1434    /// Settled again once the column has grown fourfold, because the sample it was settled on then
1435    /// covered a quarter of what exists now and a dictionary in first seen order does not look the
1436    /// same at both ends. Blocks already encoded keep the shape they were encoded with. They can,
1437    /// because a block says what it is: nothing reading one asks the column what shape to expect.
1438    fn settle(&mut self) -> Result<()> {
1439        if self.sample.len() < PAYLOAD_SAMPLE_BLOCKS {
1440            return Ok(());
1441        }
1442        let complete = self.ends.len() / TEXT_PAYLOAD_VALUES;
1443        if self.shape.is_some() && complete < self.settled.saturating_mul(4) {
1444            return Ok(());
1445        }
1446        let sample =
1447            self.sample.iter().map(|(at, bytes)| self.slices(*at, bytes)).collect::<Vec<_>>();
1448        self.shape = Some(string::with_symbols(settle_shape(&sample)?, &sample));
1449        self.settled = complete;
1450        Ok(())
1451    }
1452
1453    /// Seals the part block at the end of the load, if there is one.
1454    fn seal_rest(&mut self) {
1455        // Asked of the values rather than of the bytes, because a block of empty strings has values
1456        // in it and no bytes, and a column of nulls is exactly that.
1457        if self.ends.len() % TEXT_PAYLOAD_VALUES != 0 {
1458            self.seal();
1459        }
1460    }
1461
1462    /// Encodes the waiting block at `at`, with the settled shape when there is one and by trying
1463    /// everything when the column was too small to settle one.
1464    fn encode_waiting(&self, at: usize) -> Result<EncodedBlock> {
1465        let (block, bytes) = &self.waiting[at];
1466        let values = self.slices(*block, bytes);
1467        let encoded = match &self.shape {
1468            Some(shape) => string::encode_with(&values, shape)?,
1469            None => string::encode(&values)?,
1470        };
1471        Ok((encoded, block_grams(&values)))
1472    }
1473
1474    /// [`finish_dictionaries`] for one dictionary on this thread, for the tests that hold one.
1475    #[cfg(test)]
1476    fn finish_blocks(&mut self) -> Result<()> {
1477        self.seal_rest();
1478        let made = (0..self.waiting.len())
1479            .map(|at| self.encode_waiting(at))
1480            .collect::<Result<Vec<_>>>()?;
1481        for ((at, _), block) in std::mem::take(&mut self.waiting).into_iter().zip(made) {
1482            if self.encoded() != at {
1483                return Err(Error::internal("a dictionary block was encoded out of order"));
1484            }
1485            self.push_block(block);
1486        }
1487        Ok(())
1488    }
1489
1490    /// Every value of this dictionary read back out of its encoded blocks, as the bytes back to back
1491    /// and where each block starts in them.
1492    ///
1493    /// This is the one place the whole column is in memory at once and the reason [`Writer::close`]
1494    /// takes the columns one at a time rather than across threads. One column's values is 1.3 GB on
1495    /// the worst ClickBench column, and five columns of that at once is the peak this was all meant
1496    /// to remove.
1497    ///
1498    /// The blocks are spread over threads instead. Each block's decoded length is already known from
1499    /// the ends of its values, so the answer is laid out before anything is decoded and every thread
1500    /// decodes its own run of blocks straight into its own part of it. On the 10m ClickBench sample
1501    /// this was a second of the close for `URL` alone, on one core of thirty two, and the close is
1502    /// what a load waits on once its stripes are written.
1503    ///
1504    /// The blocks already written are read back out of `file`, so what a thread holds beyond the
1505    /// answer is one encoded block. They were written moments or minutes ago and are almost always
1506    /// still in the page cache, so this is a copy rather than a read of the disk.
1507    fn decoded(&self, file: Option<&File>) -> Result<(Vec<u8>, Vec<u64>)> {
1508        let count = self.placed.len() + self.blocks.len();
1509        if count != self.values().div_ceil(TEXT_PAYLOAD_VALUES) {
1510            return Err(invalid("global dictionary blocks do not cover its values"));
1511        }
1512        let mut bases = Vec::with_capacity(count);
1513        let mut total = 0_usize;
1514        for block in 0..count {
1515            bases.push(total as u64);
1516            let last = ((block + 1) * TEXT_PAYLOAD_VALUES).min(self.values()) - 1;
1517            total = total
1518                .checked_add(self.ends[last] as usize)
1519                .ok_or_else(|| invalid("global dictionary does not fit in memory"))?;
1520        }
1521        let mut flat = vec![0_u8; total];
1522        let mut outs = Vec::with_capacity(count);
1523        let mut rest = flat.as_mut_slice();
1524        for block in 0..count {
1525            let end = bases.get(block + 1).map_or(total, |&base| base as usize);
1526            let (out, after) = rest.split_at_mut(end - bases[block] as usize);
1527            outs.push((block, out));
1528            rest = after;
1529        }
1530        let one = |run: &mut [(usize, &mut [u8])]| -> Result<()> {
1531            let mut stored = Vec::new();
1532            for (block, out) in run {
1533                let encoded = match self.placed.get(*block) {
1534                    Some(place) => {
1535                        let file = file.ok_or_else(|| {
1536                            Error::internal("a written dictionary block has no file")
1537                        })?;
1538                        let length = usize::try_from(place.length).map_err(|_| {
1539                            invalid("global dictionary block does not fit in memory")
1540                        })?;
1541                        stored.resize(length, 0);
1542                        read_at(file, place.start, &mut stored)?;
1543                        if checksum(&stored) != place.hash {
1544                            return Err(invalid(
1545                                "a global dictionary block did not read back as written",
1546                            ));
1547                        }
1548                        stored.as_slice()
1549                    }
1550                    None => &self.blocks[*block - self.placed.len()],
1551                };
1552                let decoded = string::decode_flat(encoded)?;
1553                if decoded.bytes().len() != out.len() {
1554                    return Err(invalid(
1555                        "a global dictionary block is not the length its ends say",
1556                    ));
1557                }
1558                out.copy_from_slice(decoded.bytes());
1559            }
1560            Ok(())
1561        };
1562        // Sixteen blocks a thread at the least, because a thread costs about what decoding a few
1563        // blocks does and most columns have one or two.
1564        let workers = close_workers().min(count / 16).max(1);
1565        if workers <= 1 {
1566            one(&mut outs)?;
1567        } else {
1568            let per = count.div_ceil(workers);
1569            std::thread::scope(|scope| {
1570                outs.chunks_mut(per)
1571                    .map(|run| scope.spawn(|| one(run)))
1572                    .collect::<Vec<_>>()
1573                    .into_iter()
1574                    .try_for_each(|handle| {
1575                        handle.join().map_err(|_| {
1576                            Error::internal("a global dictionary decode worker panicked")
1577                        })?
1578                    })
1579            })?;
1580        }
1581        drop(outs);
1582        Ok((flat, bases))
1583    }
1584
1585    /// Where the value at `code` sits in the bytes [`GlobalDictionary::decoded`] handed back.
1586    ///
1587    /// A block's first value starts at the block, and every other value starts where the one before
1588    /// it ended, which is what makes 1,024 values 1,024 numbers rather than 1,025.
1589    fn value_span(ends: &[u32], bases: &[u64], code: usize) -> (usize, usize) {
1590        let Some(&base) = bases.get(code / TEXT_PAYLOAD_VALUES) else { return (0, 0) };
1591        let Some(&end) = ends.get(code) else { return (0, 0) };
1592        let base = base as usize;
1593        let from = if code % TEXT_PAYLOAD_VALUES == 0 { 0 } else { ends[code - 1] as usize };
1594        (base + from, base + end as usize)
1595    }
1596
1597    /// This dictionary's values in sorted order, each as the first eight bytes of the value and the
1598    /// code that holds it, so entry `rank` describes the value that sits at `rank` when the values
1599    /// are sorted by their bytes.
1600    ///
1601    /// Codes themselves stay in first appearance order, which is what lets the writer hand one out
1602    /// the moment it sees a value rather than waiting for the last stripe, and which also keeps a
1603    /// stripe's codes close together because the data is clustered. This is what puts the values
1604    /// back in order for anything that needs it, and it is separate from the codes so that getting
1605    /// it costs a sort of the distinct values at the end rather than a rewrite of every code page.
1606    ///
1607    /// The order is the byte order of the values and nothing else. The heads are attached after the
1608    /// sort rather than sorted on, because padding with zero on the right is order preserving for
1609    /// byte strings and so sorting by head and then by bytes lands in the same place as sorting by
1610    /// bytes: a shorter value differs from a longer one that starts the same way at a position
1611    /// where the shorter one has run out, and zero is below every byte that could be there.
1612    ///
1613    /// The heads are kept because a reader searching this order wants a comparison it can make out
1614    /// of the index alone. What they buy there depends entirely on the column and is much less than
1615    /// it looks on the columns that cost the most, which [`sort_by_value`] measures.
1616    fn ranked_with_values(&self, file: Option<&File>) -> Result<RankedDictionary> {
1617        let (flat, bases) = self.decoded(file)?;
1618        let value = |code: u32| {
1619            let (from, to) = Self::value_span(&self.ends, &bases, code as usize);
1620            flat.get(from..to).unwrap_or_default()
1621        };
1622        let mut codes = (0..self.values() as u32).collect::<Vec<_>>();
1623        sort_by_value_across(&mut codes, value, close_workers());
1624        let order = codes.into_iter().map(|code| (head(value(code)), code)).collect();
1625        Ok((order, flat, bases))
1626    }
1627
1628    #[cfg(test)]
1629    fn ranked(&self, file: Option<&File>) -> Result<Vec<(u64, u32)>> {
1630        self.ranked_with_values(file).map(|(order, _, _)| order)
1631    }
1632}
1633
1634/// Appends pages and commits a new directory.
1635///
1636/// One writer covers a whole file rather than one table. [`Writer::next`] closes the table it is on
1637/// and opens another over the same file, and [`Writer::finish`] commits every table it has closed in
1638/// one generation. That is what makes a checkpoint atomic across tables: there is one slot write at
1639/// the end of it and a reader sees every table at the generation before it or every table at the
1640/// generation after it.
1641#[derive(Debug)]
1642pub struct Writer {
1643    file: File,
1644    /// Where the next write goes, counted here rather than asked of the file.
1645    ///
1646    /// The file's own cursor is not ours. Building the numeric frequencies reads pages back through
1647    /// [`read_at`], and a positional read is only positional about where it reads from: `pread`
1648    /// leaves the cursor alone, and the call Windows has for it moves the cursor to the end of what
1649    /// it read. A writer that asked the file where it was would then write the directory over a
1650    /// page it had already written, which is what it did.
1651    at: u64,
1652    /// How far into the file the kernel has been asked to start writing, see [`WRITEBACK_STRETCH`].
1653    written_back: u64,
1654    table: Table,
1655    generation: u64,
1656    /// The first and the last source position in every stripe, in the order the stripes were
1657    /// written.
1658    order: Vec<((u64, u64), (u64, u64))>,
1659    next_order: u64,
1660    dictionaries: Vec<Option<GlobalDictionary>>,
1661    /// Which columns still have a global dictionary, shared with every [`Preparer`] this writer
1662    /// hands out so that a stripe prepared after a column lost its dictionary is not coded for it.
1663    coded: Arc<[AtomicBool]>,
1664    /// One per column, folding the rows into a summary and a sketch as they go past.
1665    ///
1666    /// `None` for a column with no hash rule, which is the interval and the nested types. See
1667    /// [`stats::Gather`] for why the statistics are built here rather than by reading the file back
1668    /// once it is committed.
1669    gathers: Vec<Option<stats::Gather>>,
1670    /// The dictionaries and the statistics while a [`Merger`] has them, which is from
1671    /// [`Writer::merger`] until the table is closed. `dictionaries` and `gathers` are empty then.
1672    lent: Option<Arc<Lent>>,
1673    pending: Vec<PendingChunk>,
1674    /// The tables already closed in this generation, in the order they were written.
1675    closed: Vec<Entry>,
1676    /// The views the next commit writes down, which [`Writer::with_views`] sets.
1677    ///
1678    /// Carried forward from the committed generation by [`Writer::open`], so a writer that was only
1679    /// opened to append a table does not have to know about views to avoid dropping them.
1680    views: Vec<ViewEntry>,
1681    /// Where the stages this writer runs are charged, which [`Writer::with_profile`] sets.
1682    ///
1683    /// The writer runs the page builder, the dictionary blocks, the writes and the publish, and it
1684    /// charges them once per stripe and once per worker, never per chunk. See
1685    /// `rudb_metrics::LoadProfile` for why that is the grain.
1686    profile: Option<Arc<LoadProfile>>,
1687}
1688
1689/// A chunk that has arrived and is waiting for the rest of its stripe.
1690///
1691/// The rows are kept rather than the pages they encode to, which is the whole of #808's first half.
1692/// Encoding on arrival put every column of every part on the thread that called `append_at`, and
1693/// that thread is the only one the load has. Encoding at the flush instead means a stripe's worth
1694/// of work is on the table at once, and a stripe splits by column into a hundred and five pieces
1695/// that share nothing.
1696#[derive(Debug)]
1697struct PendingChunk {
1698    order: (u64, u64),
1699    chunk: Chunk,
1700}
1701
1702/// What the writer still needs of a part once its columns are encoded: where in the source it came
1703/// from, how many rows it has and how large those rows were.
1704///
1705/// A stripe waiting for the writer's lock carries these rather than its chunks, so its rows are
1706/// freed as soon as they are encoded and not after the stripe is written. See [`prepare`].
1707#[derive(Debug, Clone, Copy)]
1708struct Part {
1709    order: (u64, u64),
1710    rows: usize,
1711    footprint: usize,
1712}
1713
1714impl Part {
1715    fn of(pending: &PendingChunk) -> Self {
1716        Self {
1717            order: pending.order,
1718            rows: pending.chunk.len(),
1719            footprint: pending.chunk.footprint(),
1720        }
1721    }
1722}
1723
1724/// One column's share of a stripe, which is what one encode worker produces.
1725///
1726/// Indexed by part, so a stripe is a column of these and the write loop reads down one of them.
1727/// That is also the order the loop wanted: `flush_pending` walks a column at a time and lays its
1728/// parts next to each other, and it used to reach across a row of parts to do it.
1729#[derive(Debug)]
1730struct ColumnStripe {
1731    pages: Vec<Vec<u8>>,
1732    codes: Vec<Option<Vec<u32>>>,
1733    sieves: Vec<Option<Sieve>>,
1734    ranges: Vec<Range>,
1735}
1736
1737/// Roughly what encoding a column of this type costs, for ordering the encode queue.
1738///
1739/// Only the order matters and only roughly. A string column hashes and copies every value into a
1740/// dictionary and is in a different class from everything else, and among the fixed widths the wide
1741/// ones carry more bytes through the cascade than the narrow ones. Anything finer than that would
1742/// be a cost model, and the queue already absorbs a wrong guess: it only has to avoid finishing on
1743/// a column nobody else can help with.
1744fn weight(ty: &LogicalType) -> usize {
1745    match ty {
1746        LogicalType::Varchar | LogicalType::Blob | LogicalType::Bit => 64,
1747        LogicalType::HugeInt
1748        | LogicalType::UHugeInt
1749        | LogicalType::Uuid
1750        | LogicalType::Interval => 16,
1751        LogicalType::BigInt
1752        | LogicalType::UBigInt
1753        | LogicalType::Timestamp
1754        | LogicalType::Time
1755        | LogicalType::TimeTz
1756        | LogicalType::TimestampTz
1757        | LogicalType::TimestampS
1758        | LogicalType::TimestampMs
1759        | LogicalType::TimestampNs
1760        | LogicalType::Double
1761        | LogicalType::Decimal { .. } => 8,
1762        LogicalType::Integer | LogicalType::UInteger | LogicalType::Date | LogicalType::Float => 4,
1763        LogicalType::SmallInt | LogicalType::USmallInt => 2,
1764        _ => 1,
1765    }
1766}
1767
1768/// Parts in one stripe.
1769///
1770/// Sixty four thousand rows is the smallest stripe that keeps the ClickBench directory in single
1771/// digit megabytes at a hundred million rows, and it puts a four byte column's page at a quarter of
1772/// a megabyte, which is the size a sequential read wants. Larger stripes buy a smaller directory
1773/// and cost a sparse fetch, which has to read a page index before it can reach one part.
1774pub const STRIPE_PARTS: usize = 64;
1775
1776/// How many rows the writer wants to see before it decides whether a varchar column gets to keep
1777/// its global dictionary.
1778///
1779/// See [`prepare::drops_dictionary`]. A stripe is up to [`STRIPE_PARTS`] parts, so most tables give it
1780/// far more than this and it binds only on a table that is smaller than one stripe. A handful of
1781/// rows says nothing about whether a column repeats itself, and the answer that costs nothing when
1782/// the sample is that small is the one the writer has always given, which is to keep the dictionary.
1783const DICTIONARY_DECIDE_ROWS: usize = 4_096;
1784
1785/// Out of ten. A varchar column loses its dictionary when more than this many rows in ten of the
1786/// first stripe held a value that stripe had not seen before.
1787///
1788/// See [`prepare::drops_dictionary`]. Nine and not five, because the properties a dictionary buys are
1789/// worth keeping everywhere they are real. On ClickBench the widest string column is `Referer` at
1790/// 0.131 of its first stripe and every other one is below that, so nothing there is near this and
1791/// every one of them keeps its dictionary, which is what a group by on codes wants. On TPC-H
1792/// `o_comment` and `c_comment` are at 0.97 and are what this catches.
1793///
1794/// `l_comment` sits at 0.883 and so keeps its dictionary. Eight was built and measured rather than
1795/// argued about, and it is not a clear win: it takes `select l_comment from lineitem` from 4.335 G
1796/// instructions to 3.473 G and the file from 280.2 MB to 260.4 MB, and it takes a `like` over the
1797/// same column from 3.29 G to 4.27 G, because a dictionary runs the predicate once a distinct value
1798/// and there are 3.6 M of those to 6.0 M rows. The 22 query suite came out 6.91 s against 7.07 s in
1799/// favour of nine. So nine stays until there is a reason to prefer one of those shapes. See #1137.
1800const DICTIONARY_DISTINCT_IN_TEN: usize = 9;
1801
1802/// Bytes one part takes in a stripe's index page: four for the length, eight for the checksum.
1803const INDEX_ENTRY: usize = size_of::<u32>() + size_of::<u64>();
1804
1805/// Bytes one column's section of a stripe's index page takes, including its own trailing checksum.
1806fn index_section(parts: usize) -> Result<usize> {
1807    parts
1808        .checked_mul(INDEX_ENTRY)
1809        .and_then(|bytes| bytes.checked_add(size_of::<u64>()))
1810        .ok_or_else(|| invalid("index page length overflow"))
1811}
1812
1813impl Writer {
1814    /// Opens a committed file and starts a table in the generation after the one it holds.
1815    ///
1816    /// The tables already in the file are carried forward by name and by directory pointer, and
1817    /// their pages are not read. Nothing in the file is overwritten: the new table's pages and the
1818    /// new catalog go on the end, past the catalog the committed generation points at, and the one
1819    /// write that is not an append is the slot in the header that [`Writer::finish`] does last.
1820    ///
1821    /// That slot is the other one. A file committed at generation 1 is named by the slot at 16 and
1822    /// generation 2 writes the one at 44, so until the last four bytes of the commit land the file
1823    /// still reads as the generation before it, and a slot torn across a write fails its checksum
1824    /// and the reader falls back to the one beside it. This is what the second slot has always been
1825    /// for.
1826    ///
1827    /// # Errors
1828    ///
1829    /// If the file has no valid committed directory, is not this build's format, repeats the name
1830    /// of a table already in it that holds rows, has a field with no scalar encoding, or cannot be
1831    /// written.
1832    pub fn open(
1833        path: impl AsRef<Path>,
1834        name: impl Into<String>,
1835        fields: Vec<Field>,
1836    ) -> Result<Self> {
1837        for field in &fields {
1838            type_tag(&field.ty)?;
1839        }
1840        let name = name.into();
1841        let path = path.as_ref();
1842        let (_, size, slot, bytes, _) = slot_bytes(path)?;
1843        let (mut closed, views) = decode_catalog(&bytes, size)?;
1844        // A table already in the file under this name is only in the way if it holds rows. One that
1845        // holds none has no pages for this generation to carry and no reader that could lose
1846        // anything, so the table being started here takes its place in the catalog rather than
1847        // colliding with it, and `finish` writes the new entry where the old one was.
1848        //
1849        // That is not a corner. It is the shape every loading script writes: the schema goes in one
1850        // statement and the rows go in the next, and a checkpoint between them commits the empty
1851        // table. Before this, the second statement had to build the whole table in memory because
1852        // the first had already put the name in the file, which is how a load of a table larger
1853        // than memory became a load that needed memory the size of the table.
1854        if let Some(at) = closed.iter().position(|held| held.name == name) {
1855            if closed[at].rows > 0 {
1856                return Err(invalid("two tables in one native file have the same name"));
1857            }
1858            closed.remove(at);
1859        }
1860        // The generation of the slot whose bytes checksummed, and not the highest number in the
1861        // header. A slot torn across a write can hold any number at all, and taking that one would
1862        // be choosing which slot to overwrite from a value nothing has vouched for, which is how a
1863        // half written commit gets to destroy the one good copy beside it.
1864        let generation = slot
1865            .generation
1866            .checked_add(1)
1867            .ok_or_else(|| invalid("native file generation overflow"))?;
1868        let file = OpenOptions::new().write(true).read(true).open(path).map_err(io)?;
1869        Ok(Self {
1870            file,
1871            // The end of the file, so that the committed generation's catalog stays where its slot
1872            // says it is and keeps naming a file a reader can still open.
1873            at: size,
1874            written_back: size,
1875            dictionaries: fields
1876                .iter()
1877                .map(|field| (field.ty == LogicalType::Varchar).then(GlobalDictionary::new))
1878                .collect(),
1879            coded: fields
1880                .iter()
1881                .map(|field| AtomicBool::new(field.ty == LogicalType::Varchar))
1882                .collect(),
1883            gathers: fields.iter().map(|field| stats::Gather::new(&field.ty, generation)).collect(),
1884            lent: None,
1885            table: Table {
1886                name,
1887                dictionaries: vec![None; fields.len()],
1888                dictionary_payloads: Vec::new(),
1889                distincts: vec![None; fields.len()],
1890                fields,
1891                stripes: Vec::new(),
1892                rows: 0,
1893                frequencies: Vec::new(),
1894                pair_frequencies: Vec::new(),
1895                frequency_texts: Vec::new(),
1896                host_groups: None,
1897                clustering: None,
1898                generation,
1899                sections: Vec::new(),
1900            },
1901            generation,
1902            order: Vec::new(),
1903            next_order: 0,
1904            pending: Vec::with_capacity(STRIPE_PARTS),
1905            closed,
1906            views,
1907            profile: None,
1908        })
1909    }
1910
1911    /// Creates a new v10 file and its first table.
1912    ///
1913    /// # Errors
1914    ///
1915    /// If the file exists, a field has no scalar encoding, or the path cannot be written.
1916    pub fn create(
1917        path: impl AsRef<Path>,
1918        name: impl Into<String>,
1919        fields: Vec<Field>,
1920    ) -> Result<Self> {
1921        for field in &fields {
1922            type_tag(&field.ty)?;
1923        }
1924        let file =
1925            OpenOptions::new().write(true).read(true).create_new(true).open(path).map_err(io)?;
1926        let mut header = [0; HEADER as usize];
1927        header[..8].copy_from_slice(MAGIC);
1928        header[8..12].copy_from_slice(&FORMAT.to_le_bytes());
1929        write_at(&file, 0, &header)?;
1930        Ok(Self {
1931            file,
1932            at: HEADER,
1933            written_back: HEADER,
1934            dictionaries: fields
1935                .iter()
1936                .map(|field| (field.ty == LogicalType::Varchar).then(GlobalDictionary::new))
1937                .collect(),
1938            coded: fields
1939                .iter()
1940                .map(|field| AtomicBool::new(field.ty == LogicalType::Varchar))
1941                .collect(),
1942            gathers: fields.iter().map(|field| stats::Gather::new(&field.ty, 1)).collect(),
1943            lent: None,
1944            table: Table {
1945                name: name.into(),
1946                dictionaries: vec![None; fields.len()],
1947                dictionary_payloads: Vec::new(),
1948                distincts: vec![None; fields.len()],
1949                fields,
1950                stripes: Vec::new(),
1951                rows: 0,
1952                frequencies: Vec::new(),
1953                pair_frequencies: Vec::new(),
1954                frequency_texts: Vec::new(),
1955                host_groups: None,
1956                clustering: None,
1957                generation: 1,
1958                sections: Vec::new(),
1959            },
1960            generation: 1,
1961            order: Vec::new(),
1962            next_order: 0,
1963            pending: Vec::with_capacity(STRIPE_PARTS),
1964            closed: Vec::new(),
1965            views: Vec::new(),
1966            profile: None,
1967        })
1968    }
1969
1970    /// Creates a new file that holds no table at all, committed and ready to open.
1971    ///
1972    /// A database somebody dropped the last table out of is still a database, and until this there
1973    /// was no way to write one down. Every other way into this file goes through a table, because
1974    /// [`Writer::create`] takes the first one and [`Writer::finish`] commits the one it is on, so a
1975    /// catalog with nothing in it could be read and not written. The format already allowed it: the
1976    /// catalog is a count and that many entries, and a count of nought encodes and decodes the same
1977    /// way every other count does, which is why nothing here is a version change.
1978    ///
1979    /// It hands back nothing rather than a writer, because a writer with no table is a writer with
1980    /// nothing to append to. A file that is going to hold a table is [`Writer::create`], and one
1981    /// that is going to have a table added to it later is [`Writer::open`], which reads what this
1982    /// wrote the same way it reads any other generation.
1983    ///
1984    /// It takes the views anyway, because a database with no table can still have views in it. A
1985    /// view over `range` or over another view names no table, so dropping the last table out of a
1986    /// database does not have to leave the catalog with nothing worth writing down.
1987    ///
1988    /// # Errors
1989    ///
1990    /// If the file exists or the path cannot be written.
1991    pub fn empty(path: impl AsRef<Path>, views: &[ViewEntry]) -> Result<()> {
1992        let file =
1993            OpenOptions::new().write(true).read(true).create_new(true).open(path).map_err(io)?;
1994        let mut header = [0; HEADER as usize];
1995        header[..8].copy_from_slice(MAGIC);
1996        header[8..12].copy_from_slice(&FORMAT.to_le_bytes());
1997        write_at(&file, 0, &header)?;
1998        let catalog = encode_catalog(&[], views)?;
1999        write_at(&file, HEADER, &catalog)?;
2000        // The same two syncs in the same order as [`Writer::finish`], and for the same reason. The
2001        // catalog is on the disk before the slot names it, so a file this is interrupted in the
2002        // middle of is a header with no valid slot rather than a slot pointing at nothing.
2003        file.sync_all().map_err(io)?;
2004        let slot = Slot {
2005            offset: HEADER,
2006            length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
2007            generation: 1,
2008            hash: checksum(&catalog),
2009        };
2010        write_at(&file, slot_offset(1), &slot.bytes())?;
2011        file.sync_all().map_err(io)?;
2012        Ok(())
2013    }
2014
2015    /// Closes the table this writer is on and starts another one in the same file.
2016    ///
2017    /// Nothing is published here. The closed table's directory is written so that the bytes are on
2018    /// disk and its span is known, and the catalog that names it is only written by
2019    /// [`Writer::finish`], so a crash between two tables leaves the previous generation intact.
2020    ///
2021    /// # Errors
2022    ///
2023    /// If the name repeats a table already closed, a field has no scalar encoding, or the table
2024    /// being closed cannot be written.
2025    pub fn next(mut self, name: impl Into<String>, fields: Vec<Field>) -> Result<Self> {
2026        for field in &fields {
2027            type_tag(&field.ty)?;
2028        }
2029        let name = name.into();
2030        let entry = self.close()?;
2031        if self.closed.iter().chain(std::iter::once(&entry)).any(|held| held.name == name) {
2032            return Err(invalid("two tables in one native file have the same name"));
2033        }
2034        let Self { file, at, generation, mut closed, views, .. } = self;
2035        closed.push(entry);
2036        Ok(Self {
2037            file,
2038            written_back: at,
2039            at,
2040            generation,
2041            closed,
2042            views,
2043            profile: None,
2044            dictionaries: fields
2045                .iter()
2046                .map(|field| (field.ty == LogicalType::Varchar).then(GlobalDictionary::new))
2047                .collect(),
2048            coded: fields
2049                .iter()
2050                .map(|field| AtomicBool::new(field.ty == LogicalType::Varchar))
2051                .collect(),
2052            gathers: fields.iter().map(|field| stats::Gather::new(&field.ty, generation)).collect(),
2053            lent: None,
2054            table: Table {
2055                name,
2056                dictionaries: vec![None; fields.len()],
2057                dictionary_payloads: Vec::new(),
2058                distincts: vec![None; fields.len()],
2059                fields,
2060                stripes: Vec::new(),
2061                rows: 0,
2062                frequencies: Vec::new(),
2063                pair_frequencies: Vec::new(),
2064                frequency_texts: Vec::new(),
2065                host_groups: None,
2066                clustering: None,
2067                generation,
2068                sections: Vec::new(),
2069            },
2070            order: Vec::new(),
2071            next_order: 0,
2072            pending: Vec::with_capacity(STRIPE_PARTS),
2073        })
2074    }
2075
2076    /// Sets the views the next commit writes down, replacing whatever was carried forward.
2077    ///
2078    /// It replaces rather than adds because the caller has the whole catalog in front of it and the
2079    /// writer does not. A view that was dropped is a view that is not in the list any more, and
2080    /// there is no other way for the writer to hear about that, since nothing else it is told about
2081    /// mentions views at all.
2082    ///
2083    /// A writer that is never told anything writes back the views it read at [`Writer::open`], so a
2084    /// checkpoint that only had a table to append does not quietly drop them.
2085    #[must_use]
2086    pub fn with_views(mut self, views: Vec<ViewEntry>) -> Self {
2087        self.views = views;
2088        self
2089    }
2090
2091    /// Charges the stages this writer runs to `profile`.
2092    ///
2093    /// For the table being written now. [`Writer::next`] starts the next table without one,
2094    /// because a second table's stripes charged to the first table's load would be a profile of
2095    /// neither.
2096    #[must_use]
2097    pub fn with_profile(mut self, profile: Arc<LoadProfile>) -> Self {
2098        self.profile = Some(profile);
2099        self
2100    }
2101
2102    /// Records the order this table's rows are meant to be stored in.
2103    ///
2104    /// The declaration goes in the table directory and comes back out of
2105    /// [`Table::clustering`]. Nothing here sorts anything, and nothing here checks that the rows
2106    /// handed to [`Writer::append`] arrive in the order this claims. That is deliberate for now:
2107    /// the thing that was missing was a place to write the order down, and a loader that honours
2108    /// the declaration is the next piece rather than this one.
2109    ///
2110    /// The declaration applies to the table the writer is currently on, so it is set after
2111    /// [`Writer::next`] rather than once for the file.
2112    ///
2113    /// # Errors
2114    ///
2115    /// If the declaration names a column this table does not have.
2116    pub fn declare(mut self, clustering: Clustering) -> Result<Self> {
2117        // Rebuilt against this table's own column count rather than trusted, because the caller
2118        // built it against a catalog entry and the two could have drifted.
2119        self.table.clustering = Some(Clustering::new(
2120            clustering.columns().to_vec(),
2121            clustering.width(),
2122            &self.table.fields,
2123        )?);
2124        Ok(self)
2125    }
2126
2127    /// Appends bytes at the end of the file and moves the writer's own offset past them.
2128    ///
2129    /// Every write in here goes through this, so that [`Writer::at`] is the only answer to where
2130    /// anything is and the file's cursor is never consulted for it.
2131    fn put(&mut self, bytes: &[u8]) -> Result<()> {
2132        write_at(&self.file, self.at, bytes)?;
2133        self.at = self
2134            .at
2135            .checked_add(bytes.len() as u64)
2136            .ok_or_else(|| invalid("native file length overflow"))?;
2137        if self.at - self.written_back >= WRITEBACK_STRETCH {
2138            rudb_io::start_writeback(&self.file, self.written_back, self.at - self.written_back);
2139            self.written_back = self.at;
2140        }
2141        Ok(())
2142    }
2143
2144    /// Writes one chunk as independently readable column pages.
2145    ///
2146    /// # Errors
2147    ///
2148    /// If its width or types differ from the declared table, or a page exceeds its bound.
2149    pub fn append(&mut self, chunk: &Chunk) -> Result<()> {
2150        let order = (self.next_order, 0);
2151        self.next_order = self.next_order.saturating_add(1);
2152        self.append_at(order, chunk)
2153    }
2154
2155    /// Writes one chunk and records its source position for directory ordering.
2156    ///
2157    /// Pages may be encoded by parallel pipeline instances and reach the file in completion order.
2158    /// The stripe they land in is sorted by this key at commit, and [`Self::finish`] rejects a
2159    /// sequence whose parts do not come out in source order once the stripes are sorted, because a
2160    /// stripe groups whatever arrived together and cannot put a late part back where it belongs.
2161    ///
2162    /// # Errors
2163    ///
2164    /// The same as [`Self::append`].
2165    pub fn append_at(&mut self, order: (u64, u64), chunk: &Chunk) -> Result<()> {
2166        if chunk.is_empty() {
2167            return Ok(());
2168        }
2169        self.admit(chunk)?;
2170        if self.pending.last().is_some_and(|last| last.order > order) {
2171            self.flush_pending()?;
2172        }
2173        // Cloned rather than encoded, and a clone of a chunk that owns its buffers is a copy of
2174        // them. Sixty four parts of a hundred and five columns is tens of megabytes held for the
2175        // length of a stripe and a few seconds of memory traffic over a whole ClickBench load,
2176        // against the hundreds of seconds of encode this is what lets off one thread.
2177        self.pending.push(PendingChunk { order, chunk: chunk.clone() });
2178        if self.pending.len() == STRIPE_PARTS {
2179            self.flush_pending()?;
2180        }
2181        Ok(())
2182    }
2183
2184    /// Writes a run of chunks as one stripe of its own.
2185    ///
2186    /// [`Self::append_at`] decides where a stripe ends by watching the orders go past, which works
2187    /// when one caller hands over every chunk in source order and does not when several do. A
2188    /// writer being fed by more than one pipeline instance sees the orders interleave, and a stripe
2189    /// that ends every time two of them cross is a stripe of one or two parts.
2190    ///
2191    /// So the grouping moves to the caller. Whoever is buffering hands over a run it already knows
2192    /// is contiguous and in order, and gets a stripe holding exactly that run. The orders still
2193    /// have to come out in source order once the stripes are sorted, which [`Self::finish`] checks,
2194    /// so the runs from different callers may interleave with each other but may not overlap.
2195    ///
2196    /// # Errors
2197    ///
2198    /// The same as [`Self::append`], and if the run is longer than [`STRIPE_PARTS`].
2199    pub fn append_stripe(&mut self, parts: Vec<((u64, u64), Chunk)>) -> Result<()> {
2200        if parts.len() > STRIPE_PARTS {
2201            return Err(invalid("a stripe was handed more parts than it holds"));
2202        }
2203        // Whatever an earlier caller left behind is its own stripe rather than the front of this
2204        // one, because the two runs are from different places in the source and a stripe is a run.
2205        self.flush_pending()?;
2206        for (order, chunk) in parts {
2207            if chunk.is_empty() {
2208                continue;
2209            }
2210            self.admit(&chunk)?;
2211            self.pending.push(PendingChunk { order, chunk });
2212        }
2213        self.flush_pending()
2214    }
2215
2216    /// Checks a chunk against the declared table and counts its rows in.
2217    fn admit(&mut self, chunk: &Chunk) -> Result<()> {
2218        if chunk.width() != self.table.fields.len() {
2219            return Err(invalid("chunk width differs from table schema"));
2220        }
2221        for (index, field) in self.table.fields.iter().enumerate() {
2222            if chunk.column(index)?.logical_type() != &field.ty {
2223                return Err(invalid("chunk type differs from table schema"));
2224            }
2225        }
2226        self.table.rows = self
2227            .table
2228            .rows
2229            .checked_add(chunk.len())
2230            .ok_or_else(|| invalid("row count overflow"))?;
2231        Ok(())
2232    }
2233
2234    /// One column's parts of a stripe as pages, for a column with no global dictionary.
2235    fn encode_pages(columns: &[&Vector]) -> Result<ColumnStripe> {
2236        let mut stripe = ColumnStripe {
2237            pages: Vec::with_capacity(columns.len()),
2238            codes: Vec::with_capacity(columns.len()),
2239            sieves: Vec::with_capacity(columns.len()),
2240            ranges: Vec::with_capacity(columns.len()),
2241        };
2242        let mut settling = Settling::default();
2243        for &column in columns {
2244            let bytes = encode(column, &mut settling)?;
2245            if bytes.len() > MAX_PAGE {
2246                return Err(invalid("column page exceeds the configured bound"));
2247            }
2248            // The range is built first because the sieve reads it rather than walking the column a
2249            // second time to find out how wide it is.
2250            let range = Range::of(column);
2251            // A sieve at least as large as the part it indexes is not written. A reader reads the
2252            // sieve to decide whether to read the part, so when the sieve is the larger of the two
2253            // it has already spent more than the read it is trying to avoid, and that holds even if
2254            // it rejects every time. It is a necessary condition rather than the whole rule, which
2255            // is that a sieve pays when its bytes are under the rejection rate times the part's,
2256            // but the rejection rate depends on what a query probes for and the writer does not
2257            // know that. The necessary half needs two numbers that are both in hand here.
2258            //
2259            // A column with a global dictionary gets none, because it already has an exact
2260            // membership index per stripe. Those do not come through here. See [`prepare`].
2261            let sieve =
2262                Sieve::of(column, &range, SIEVE_BUDGET).filter(|sieve| sieve.len() < bytes.len());
2263            stripe.pages.push(bytes);
2264            stripe.codes.push(None);
2265            stripe.sieves.push(sieve);
2266            stripe.ranges.push(range);
2267        }
2268        Ok(stripe)
2269    }
2270
2271    /// Writes every encoded dictionary block that is not in the file yet and forgets its bytes.
2272    ///
2273    /// This is what keeps a load from holding its dictionaries' payload. The blocks land between
2274    /// stripes wherever the writer is, which is fine because the index says where each one is.
2275    fn place_blocks(&mut self) -> Result<()> {
2276        if let Some(lent) = self.lent.clone() {
2277            return self.place_lent_blocks(&lent);
2278        }
2279        let mut dictionaries = std::mem::take(&mut self.dictionaries);
2280        let placed = dictionaries.iter_mut().flatten().try_for_each(|dictionary| {
2281            for block in std::mem::take(&mut dictionary.blocks) {
2282                let start = self.at;
2283                self.put(&block)?;
2284                dictionary.placed.push(Placed {
2285                    start,
2286                    length: block.len() as u64,
2287                    hash: checksum(&block),
2288                });
2289            }
2290            Ok(())
2291        });
2292        self.dictionaries = dictionaries;
2293        placed
2294    }
2295
2296    /// [`Writer::place_blocks`] while a [`Merger`] has the dictionaries.
2297    ///
2298    /// A column whose merge is running is passed over rather than waited for, because the writer's
2299    /// lock is held here and a merge of `URL` can take tens of milliseconds. Its blocks go out with
2300    /// a later stripe, or at the close.
2301    fn place_lent_blocks(&mut self, lent: &Lent) -> Result<()> {
2302        for column in lent.columns() {
2303            let Ok(mut held) = column.try_lock() else { continue };
2304            let Some(dictionary) = held.dictionary.as_mut() else { continue };
2305            for block in std::mem::take(&mut dictionary.blocks) {
2306                let start = self.at;
2307                self.put(&block)?;
2308                dictionary.placed.push(Placed {
2309                    start,
2310                    length: block.len() as u64,
2311                    hash: checksum(&block),
2312                });
2313            }
2314        }
2315        Ok(())
2316    }
2317
2318    /// Takes the dictionaries and the statistics back from the [`Merger`] that has them.
2319    ///
2320    /// A merge that starts after this is refused, since whatever it merged would be lost.
2321    fn reclaim(&mut self) -> Result<()> {
2322        let Some(lent) = self.lent.take() else { return Ok(()) };
2323        let (dictionaries, gathers) = lent.reclaim()?;
2324        self.dictionaries = dictionaries;
2325        self.gathers = gathers;
2326        Ok(())
2327    }
2328
2329    /// Writes the buffered parts as one stripe, each column's parts contiguous on disk.
2330    ///
2331    /// The same four steps a caller holding this writer behind a lock takes, with nobody else
2332    /// waiting between them. See [`prepare`].
2333    fn flush_pending(&mut self) -> Result<()> {
2334        if self.pending.is_empty() {
2335            return Ok(());
2336        }
2337        let held = std::mem::take(&mut self.pending);
2338        let prepared = self.preparer().prepare_held(held)?;
2339        let merged = self.merge_held(prepared)?;
2340        let paged = merged.pages()?;
2341        self.write_paged(paged)
2342    }
2343
2344    /// Writes one stripe whose pages are built, each column's parts contiguous on disk.
2345    fn write_stripe(&mut self, held: &[Part], encoded: Vec<ColumnStripe>) -> Result<()> {
2346        let width = self.table.fields.len();
2347        let parts = held.len();
2348        if encoded.len() != width {
2349            return Err(Error::internal("a stripe came to the writer with the wrong columns"));
2350        }
2351        let profile = self.profile.clone();
2352        if let Some(profile) = &profile {
2353            let rows = held.iter().map(|part| part.rows as u64).sum();
2354            let raw = held.iter().map(|part| part.footprint as u64).sum();
2355            let pages =
2356                encoded.iter().flat_map(|stripe| &stripe.pages).map(|page| page.len() as u64).sum();
2357            profile.moved(Stage::Pages, raw, pages, rows);
2358        }
2359        // Before a byte of the stripe is written, so that the blocks the stripe's pages were built
2360        // with, and any that were waiting on them, are let go of now rather than a stripe later.
2361        let timing = profile.as_deref().map(|profile| profile.span(Stage::Dictionary));
2362        let before = self.at;
2363        self.place_blocks()?;
2364        drop(timing);
2365        if let Some(profile) = &profile {
2366            profile.moved(Stage::Dictionary, 0, self.at - before, 0);
2367        }
2368        let timing = profile.as_deref().map(|profile| profile.span(Stage::Write));
2369        let before = self.at;
2370        let mut pages = Vec::with_capacity(width);
2371        let mut memberships = vec![None; width];
2372        let mut ranges = Vec::with_capacity(width);
2373        let mut index = Vec::with_capacity(width.saturating_mul(index_section(parts)?));
2374        for stripe in &encoded {
2375            let offset = self.at;
2376            let section = index.len();
2377            let mut length = 0_usize;
2378            for bytes in &stripe.pages {
2379                write_at(&self.file, self.at + length as u64, bytes)?;
2380                put_u32(
2381                    &mut index,
2382                    u32::try_from(bytes.len()).map_err(|_| invalid("part length overflow"))?,
2383                );
2384                put_u64(&mut index, checksum(bytes));
2385                length = length
2386                    .checked_add(bytes.len())
2387                    .ok_or_else(|| invalid("column page length overflow"))?;
2388            }
2389            let hash = checksum(&index[section..]);
2390            put_u64(&mut index, hash);
2391            if length > MAX_PAGE {
2392                return Err(invalid("column page exceeds the configured bound"));
2393            }
2394            self.at = self
2395                .at
2396                .checked_add(length as u64)
2397                .ok_or_else(|| invalid("native file length overflow"))?;
2398            pages.push(Span {
2399                offset,
2400                length: u32::try_from(length).map_err(|_| invalid("page length overflow"))?,
2401            });
2402            ranges.push(merged_range(stripe.ranges.iter().cloned()));
2403        }
2404        for (membership, stripe) in memberships.iter_mut().zip(&encoded) {
2405            if stripe.codes.iter().all(Option::is_none) {
2406                continue;
2407            }
2408            let lists = stripe
2409                .codes
2410                .iter()
2411                .map(|codes| codes.clone().unwrap_or_default())
2412                .collect::<Vec<_>>();
2413            let bytes = encode_membership(&merged_codes(lists));
2414            let offset = self.at;
2415            self.put(&bytes)?;
2416            *membership = Some(Page {
2417                offset,
2418                length: u32::try_from(bytes.len())
2419                    .map_err(|_| invalid("membership page length overflow"))?,
2420                hash: checksum(&bytes),
2421            });
2422        }
2423        let mut sieves = vec![None; width];
2424        for (page, stripe) in sieves.iter_mut().zip(&encoded) {
2425            if stripe.sieves.iter().all(Option::is_none) {
2426                continue;
2427            }
2428            let bytes = encode_sieves(stripe.sieves.iter())?;
2429            let offset = self.at;
2430            self.put(&bytes)?;
2431            *page = Some(Page {
2432                offset,
2433                length: u32::try_from(bytes.len())
2434                    .map_err(|_| invalid("sieve page length overflow"))?,
2435                hash: checksum(&bytes),
2436            });
2437        }
2438        // A stripe of one part has the same rows in it as that part, so its own bounds are already
2439        // the part's and a page here would say what the directory says. Everywhere else the page is
2440        // written unless it comes to more than the column it indexes, which is the rule the sieves
2441        // go by and for the same reason: a reader reads this to decide whether to read the column,
2442        // so a page larger than the column has spent more than the read it is avoiding.
2443        let mut part_ranges = vec![None; width];
2444        if parts > 1 {
2445            for ((page, stripe), span) in part_ranges.iter_mut().zip(&encoded).zip(&pages) {
2446                let bytes = encode_part_ranges(&stripe.ranges)?;
2447                if bytes.len() >= span.length as usize {
2448                    continue;
2449                }
2450                let offset = self.at;
2451                self.put(&bytes)?;
2452                *page = Some(Page {
2453                    offset,
2454                    length: u32::try_from(bytes.len())
2455                        .map_err(|_| invalid("part range page length overflow"))?,
2456                    hash: checksum(&bytes),
2457                });
2458            }
2459        }
2460        let offset = self.at;
2461        self.put(&index)?;
2462        let index = Span {
2463            offset,
2464            length: u32::try_from(index.len())
2465                .map_err(|_| invalid("index page length overflow"))?,
2466        };
2467        let mut rows = 0_usize;
2468        let mut lengths = Vec::with_capacity(parts);
2469        let mut span = None;
2470        for part in held {
2471            rows = rows.checked_add(part.rows).ok_or_else(|| invalid("row count overflow"))?;
2472            lengths.push(u32::try_from(part.rows).map_err(|_| invalid("part row count overflow"))?);
2473            span = Some(span.map_or((part.order, part.order), |(first, _)| (first, part.order)));
2474        }
2475        self.order.push(span.ok_or_else(|| invalid("a stripe was flushed with no parts"))?);
2476        self.table.stripes.push(Stripe {
2477            rows,
2478            parts: lengths,
2479            index,
2480            pages,
2481            memberships: Pages::from_slots(memberships)?,
2482            sieves: Pages::from_slots(sieves)?,
2483            part_ranges: Pages::from_slots(part_ranges)?,
2484            zone: Zone::from_ranges(ranges),
2485        });
2486        drop(timing);
2487        if let Some(profile) = &profile {
2488            profile.moved(Stage::Write, 0, self.at - before, rows as u64);
2489        }
2490        Ok(())
2491    }
2492
2493    /// Finds exact heavy hitters without keeping a hash table for every numeric column while the
2494    /// load is live. The pages are already in the target file, so one column at a time uses a
2495    /// bounded Misra-Gries candidate table and then recounts only those candidates.
2496    ///
2497    /// The first of those passes also counts the column's distinct values exactly, up to the cap in
2498    /// [`distinct`], which is the number a string column gets from its dictionary. It comes back
2499    /// beside the summary because a column whose heavy hitters cannot be proved can still have been
2500    /// counted.
2501    ///
2502    /// The tables are keyed by a value's sixty four bits rather than by [`FrequencyValue`], and a
2503    /// null is counted beside them. Every integer type the format stores fits in those bits, so
2504    /// within one column two values share bits only if they are the same value, and a sixteen byte
2505    /// entry keeps the whole candidate table in the second level cache where the forty eight byte
2506    /// one did not. The null takes part in the candidate table exactly as a key would: it holds a
2507    /// place while its count is above zero, and it is decremented with the rest.
2508    fn numeric_frequency(&self, column: usize) -> Result<(Option<FrequencySummary>, Option<u64>)> {
2509        let signed = match self.table.fields[column].ty {
2510            LogicalType::TinyInt
2511            | LogicalType::SmallInt
2512            | LogicalType::Integer
2513            | LogicalType::BigInt
2514            | LogicalType::Date
2515            | LogicalType::Timestamp => true,
2516            LogicalType::UTinyInt
2517            | LogicalType::USmallInt
2518            | LogicalType::UInteger
2519            | LogicalType::UBigInt => false,
2520            _ => return Ok((None, None)),
2521        };
2522        let value_of = |bits: Option<u64>| match bits {
2523            None => FrequencyValue::Null,
2524            Some(bits) if signed => FrequencyValue::Integer(i128::from(bits as i64)),
2525            Some(bits) => FrequencyValue::Integer(i128::from(bits)),
2526        };
2527        // Rows arrive a run of equal values at a time, because a sorted column is runs and a flag
2528        // column is mostly one value, so a run is counted and inserted once rather than per row.
2529        let mut first = Candidates::default();
2530        let mut distinct = distinct::ExactDistinct::new();
2531        let mut run = Run::default();
2532        self.visit_numeric(column, signed, |_, bits| {
2533            if let Some((bits, times)) = run.push(bits) {
2534                first.add(bits, times);
2535            }
2536            if run.times == 1 {
2537                if let Some(bits) = bits {
2538                    distinct.insert(bits);
2539                }
2540            }
2541        })?;
2542        if let Some((bits, times)) = run.take() {
2543            first.add(bits, times);
2544        }
2545        let Candidates { counts: candidates, nulls, decrements } = first;
2546        let (exact, null_count) = if decrements == 0 {
2547            let exact = candidates
2548                .into_iter()
2549                .map(|(bits, count)| (bits, u64::from(count)))
2550                .collect::<FrequencyMap<_>>();
2551            (exact, (nulls != 0).then_some(u64::from(nulls)))
2552        } else {
2553            let mut lower = candidates.values().copied().collect::<Vec<_>>();
2554            if nulls != 0 {
2555                lower.push(nulls);
2556            }
2557            lower.sort_unstable_by(|left, right| right.cmp(left));
2558            if lower.len() < FREQUENCY_BUILD_RANK
2559                || u64::from(lower[FREQUENCY_BUILD_RANK - 1]) <= decrements
2560            {
2561                return Ok((None, distinct.count()));
2562            }
2563            let mut exact =
2564                candidates.into_keys().map(|bits| (bits, 0_u64)).collect::<FrequencyMap<_>>();
2565            let mut null_count = (nulls != 0).then_some(0_u64);
2566            let mut recount = |bits: Option<u64>, times: u32| {
2567                let held = match bits {
2568                    Some(bits) => exact.get_mut(&bits),
2569                    None => null_count.as_mut(),
2570                };
2571                if let Some(count) = held {
2572                    *count = count.saturating_add(u64::from(times));
2573                }
2574            };
2575            let mut run = Run::default();
2576            self.visit_numeric(column, signed, |_, bits| {
2577                if let Some((bits, times)) = run.push(bits) {
2578                    recount(bits, times);
2579                }
2580            })?;
2581            if let Some((bits, times)) = run.take() {
2582                recount(bits, times);
2583            }
2584            (exact, null_count)
2585        };
2586        let mut entries = exact
2587            .into_iter()
2588            .map(|(bits, count)| FrequencyEntry { value: value_of(Some(bits)), count })
2589            .chain(null_count.map(|count| FrequencyEntry { value: FrequencyValue::Null, count }))
2590            .collect::<Vec<_>>();
2591        let omitted_max = keep_most_frequent(&mut entries).max(decrements);
2592        let kept_rows = entries.iter().try_fold(0_u64, |total, entry| {
2593            total.checked_add(entry.count).filter(|&total| total <= FREQUENCY_ORDINALS as u64)
2594        });
2595        let mut ordinals = Vec::new();
2596        let mut ordinal_entries = Vec::new();
2597        if let Some(kept_rows) = kept_rows {
2598            let mut kept = FrequencyMap::default();
2599            let mut null_kept = None;
2600            for (at, entry) in entries.iter().enumerate() {
2601                let at = u16::try_from(at)
2602                    .map_err(|_| invalid("too many retained frequency entries"))?;
2603                match entry.value {
2604                    FrequencyValue::Integer(value) => {
2605                        kept.insert(value as u64, at);
2606                    }
2607                    FrequencyValue::Null => null_kept = Some(at),
2608                    FrequencyValue::Code(_) => {}
2609                }
2610            }
2611            ordinals.reserve(usize::try_from(kept_rows).unwrap_or(FREQUENCY_ORDINALS));
2612            ordinal_entries.reserve(usize::try_from(kept_rows).unwrap_or(FREQUENCY_ORDINALS));
2613            self.visit_numeric(column, signed, |ordinal, bits| {
2614                let held = match bits {
2615                    Some(bits) => kept.get(&bits).copied(),
2616                    None => null_kept,
2617                };
2618                if let Some(entry) = held {
2619                    ordinals.push(ordinal);
2620                    ordinal_entries.push(entry);
2621                }
2622            })?;
2623        }
2624        Ok((
2625            Some(FrequencySummary { entries, omitted_max, ordinals, ordinal_entries }),
2626            distinct.count(),
2627        ))
2628    }
2629
2630    /// Hands every row of an integer column to `visit` as its ordinal and its sixty four bits, or
2631    /// `None` for a null.
2632    ///
2633    /// `signed` says which of the two readings the column has. A packed unsigned column would come
2634    /// back from `signed_block` as a base plus a code in `i64`, which wraps for a value past the top
2635    /// of `BIGINT`, so only a signed column takes the block path.
2636    fn visit_numeric(
2637        &self,
2638        column: usize,
2639        signed: bool,
2640        mut visit: impl FnMut(u64, Option<u64>),
2641    ) -> Result<()> {
2642        let ty = &self.table.fields[column].ty;
2643        let mut start = 0_u64;
2644        let mut block = Vec::new();
2645        for stripe in &self.table.stripes {
2646            let spans = read_index(&self.file, stripe, column)?;
2647            let page = stripe.pages[column];
2648            let mut bytes = vec![0; page.length as usize];
2649            read_at(&self.file, page.offset, &mut bytes)?;
2650            for (span, &rows) in spans.iter().zip(&stripe.parts) {
2651                let part = part_bytes(&bytes, *span)?;
2652                if checksum(part) != span.hash {
2653                    return Err(invalid("column page checksum differs while building frequencies"));
2654                }
2655                let rows = rows as usize;
2656                let vector = decode(ty, rows, part, None)?;
2657                // Every signed layout a numeric column decodes to, which is every column of `hits`,
2658                // comes out as one run of `i64` and is walked as a slice. The row path below is for
2659                // the unsigned types and anything else that cannot be handed over that way.
2660                if signed && vector.signed_block(&mut block) && block.len() == rows {
2661                    if vector.none_null() {
2662                        for (row, &value) in block.iter().enumerate() {
2663                            visit(start.saturating_add(row as u64), Some(value as u64));
2664                        }
2665                    } else {
2666                        for (row, &value) in block.iter().enumerate() {
2667                            let bits = (!vector.is_null_at(row)).then_some(value as u64);
2668                            visit(start.saturating_add(row as u64), bits);
2669                        }
2670                    }
2671                    start = start.saturating_add(rows as u64);
2672                    continue;
2673                }
2674                // row at a time: frequency construction visits decoded values to update bounded candidates.
2675                for row in 0..rows {
2676                    let bits = if vector.is_null_at(row) {
2677                        None
2678                    } else {
2679                        // An unsigned column has no signed reading, and the documented fallback is
2680                        // the value itself. Every width the format stores fits in sixty four bits,
2681                        // so nothing is lost on the way through.
2682                        let widened = match vector.signed_at(row) {
2683                            Some(value) => Some(value as u64),
2684                            None => match vector.value_at(row) {
2685                                Value::UTinyInt(value) => Some(u64::from(value)),
2686                                Value::USmallInt(value) => Some(u64::from(value)),
2687                                Value::UInteger(value) => Some(u64::from(value)),
2688                                Value::UBigInt(value) => Some(value),
2689                                _ => None,
2690                            },
2691                        };
2692                        Some(widened.ok_or_else(|| {
2693                            invalid("numeric frequency page did not contain an integer value")
2694                        })?)
2695                    };
2696                    visit(start.saturating_add(row as u64), bits);
2697                }
2698                start = start.saturating_add(rows as u64);
2699            }
2700        }
2701        Ok(())
2702    }
2703
2704    /// Builds independent numeric synopses concurrently after all column pages are committed.
2705    ///
2706    /// The columns go through a queue rather than being cut into equal runs, because they are not
2707    /// equally expensive and they are not shuffled. A `BIGINT` column carries eight times the bytes
2708    /// of a `TINYINT` through the decode, and a run of them sits together in a schema the way it
2709    /// sits together in `hits`, so a worker that was handed the wrong six columns finishes long
2710    /// after one that was handed the right six and the whole phase waits for it.
2711    fn numeric_frequencies(&self) -> Result<Vec<(Option<FrequencySummary>, Option<u64>)>> {
2712        let mut columns = self
2713            .table
2714            .fields
2715            .iter()
2716            .enumerate()
2717            .filter_map(|(column, field)| {
2718                matches!(
2719                    field.ty,
2720                    LogicalType::TinyInt
2721                        | LogicalType::SmallInt
2722                        | LogicalType::Integer
2723                        | LogicalType::BigInt
2724                        | LogicalType::UTinyInt
2725                        | LogicalType::USmallInt
2726                        | LogicalType::UInteger
2727                        | LogicalType::UBigInt
2728                        | LogicalType::Date
2729                        | LogicalType::Timestamp
2730                )
2731                .then_some(column)
2732            })
2733            .collect::<Vec<_>>();
2734        let workers = std::thread::available_parallelism()
2735            .map_or(1, usize::from)
2736            .min(MAX_FREQUENCY_WORKERS)
2737            .min(columns.len());
2738        let profile = self.profile.as_deref();
2739        if workers <= 1 {
2740            let _timing = profile.map(|profile| profile.span(Stage::Publish));
2741            let mut frequencies = vec![(None, None); self.table.fields.len()];
2742            for column in columns {
2743                frequencies[column] = self.numeric_frequency(column)?;
2744            }
2745            return Ok(frequencies);
2746        }
2747        // Popped from the back, so the expensive columns are the ones taken first and the cheap ones
2748        // are what is left to fill in behind them.
2749        columns.sort_by_key(|&column| weight(&self.table.fields[column].ty));
2750        let queue = Mutex::new(columns);
2751        let pieces = std::thread::scope(|scope| {
2752            (0..workers)
2753                .map(|_| {
2754                    scope.spawn(|| {
2755                        let _timing = profile.map(|profile| profile.span(Stage::Publish));
2756                        let mut mine = Vec::new();
2757                        loop {
2758                            let taken = queue
2759                                .lock()
2760                                .map_err(|_| Error::internal("a native frequency worker panicked"))?
2761                                .pop();
2762                            let Some(column) = taken else { break };
2763                            mine.push((column, self.numeric_frequency(column)?));
2764                        }
2765                        Ok(mine)
2766                    })
2767                })
2768                .collect::<Vec<_>>()
2769                .into_iter()
2770                .map(|handle| {
2771                    handle
2772                        .join()
2773                        .map_err(|_| Error::internal("a native frequency worker panicked"))?
2774                })
2775                .collect::<Result<Vec<_>>>()
2776        })?;
2777        let mut frequencies = vec![(None, None); self.table.fields.len()];
2778        for piece in pieces {
2779            for (column, summary) in piece {
2780                frequencies[column] = summary;
2781            }
2782        }
2783        Ok(frequencies)
2784    }
2785
2786    /// Reads one stable dictionary code column only at sorted table-wide row ordinals.
2787    fn stable_codes_at(&self, column: usize, ordinals: &[u64]) -> Result<Option<Vec<Option<u32>>>> {
2788        if self.dictionaries.get(column).and_then(Option::as_ref).is_none() {
2789            return Ok(None);
2790        }
2791        if ordinals.windows(2).any(|pair| pair[0] >= pair[1]) {
2792            return Err(invalid("frequency ordinals are not sorted and unique"));
2793        }
2794        let mut out = Vec::with_capacity(ordinals.len());
2795        let mut wanted = 0;
2796        let mut stripe_start = 0_u64;
2797        for stripe in &self.table.stripes {
2798            let stripe_end = stripe_start.saturating_add(stripe.rows as u64);
2799            if wanted == ordinals.len() || ordinals[wanted] >= stripe_end {
2800                stripe_start = stripe_end;
2801                continue;
2802            }
2803            let spans = read_index(&self.file, stripe, column)?;
2804            let page = stripe.pages[column];
2805            let mut bytes = vec![0; page.length as usize];
2806            read_at(&self.file, page.offset, &mut bytes)?;
2807            let mut part_start = stripe_start;
2808            for (span, &rows) in spans.iter().zip(&stripe.parts) {
2809                let part_end = part_start.saturating_add(u64::from(rows));
2810                if wanted < ordinals.len() && ordinals[wanted] < part_end {
2811                    let part = part_bytes(&bytes, *span)?;
2812                    if checksum(part) != span.hash {
2813                        return Err(invalid(
2814                            "column page checksum differs while building pair frequencies",
2815                        ));
2816                    }
2817                    let upto = ordinals.partition_point(|&ordinal| ordinal < part_end);
2818                    let positions = ordinals[wanted..upto]
2819                        .iter()
2820                        .map(|&ordinal| {
2821                            usize::try_from(ordinal.saturating_sub(part_start))
2822                                .map_err(|_| invalid("frequency row offset does not fit in memory"))
2823                        })
2824                        .collect::<Result<Vec<_>>>()?;
2825                    if !decode_selected_stable_codes(rows as usize, part, &positions, &mut out)? {
2826                        return Ok(None);
2827                    }
2828                    wanted = upto;
2829                }
2830                part_start = part_end;
2831            }
2832            stripe_start = stripe_end;
2833        }
2834        if wanted != ordinals.len() {
2835            return Err(invalid("frequency ordinal is outside the table"));
2836        }
2837        Ok(Some(out))
2838    }
2839
2840    /// Derives bounded two-key leaders from numeric anchor ordinals and stable string codes.
2841    fn pair_frequencies(
2842        &self,
2843        frequencies: &[Option<Frequencies>],
2844    ) -> Result<Vec<PairFrequencySummary>> {
2845        let anchors = frequencies
2846            .iter()
2847            .enumerate()
2848            .filter_map(|(column, summary)| {
2849                // A writer holds every synopsis it counted, so there is nothing stored to skip.
2850                match summary {
2851                    Some(Frequencies::Held(summary)) => Some(summary),
2852                    _ => None,
2853                }
2854                .filter(|summary| {
2855                    !summary.ordinals.is_empty()
2856                        && summary.ordinal_entries.len() == summary.ordinals.len()
2857                })
2858                .cloned()
2859                .map(|summary| (column, summary))
2860            })
2861            .collect::<Vec<_>>();
2862        let strings = self
2863            .dictionaries
2864            .iter()
2865            .enumerate()
2866            .filter_map(|(column, dictionary)| dictionary.as_ref().map(|_| column))
2867            .collect::<Vec<_>>();
2868        let mut summaries = Vec::new();
2869        for (first, anchors) in anchors {
2870            for &second in &strings {
2871                if summaries.len() == MAX_PAIR_FREQUENCIES {
2872                    return Ok(summaries);
2873                }
2874                let Some(codes) = self.stable_codes_at(second, &anchors.ordinals)? else {
2875                    continue;
2876                };
2877                if codes.len() != anchors.ordinal_entries.len() {
2878                    return Err(invalid("pair frequency columns have different lengths"));
2879                }
2880                let mut counts = HashMap::<(u16, Option<u32>), u64>::new();
2881                for (&anchor, code) in anchors.ordinal_entries.iter().zip(codes) {
2882                    *counts.entry((anchor, code)).or_default() += 1;
2883                }
2884                let mut entries = counts
2885                    .into_iter()
2886                    .map(|((first_entry, second), count)| PairFrequencyEntry {
2887                        first_entry,
2888                        second,
2889                        count,
2890                    })
2891                    .collect::<Vec<_>>();
2892                entries.sort_unstable_by(|left, right| {
2893                    right
2894                        .count
2895                        .cmp(&left.count)
2896                        .then_with(|| left.first_entry.cmp(&right.first_entry))
2897                        .then_with(|| left.second.cmp(&right.second))
2898                });
2899                let pair_omitted = entries.get(FREQUENCY_ENTRIES).map_or(0, |entry| entry.count);
2900                entries.truncate(FREQUENCY_ENTRIES);
2901                summaries.push(PairFrequencySummary {
2902                    first: u16::try_from(first)
2903                        .map_err(|_| invalid("pair frequency column index overflows"))?,
2904                    second: u16::try_from(second)
2905                        .map_err(|_| invalid("pair frequency column index overflows"))?,
2906                    entries,
2907                    omitted_max: anchors.omitted_max.max(pair_omitted),
2908                });
2909            }
2910        }
2911        Ok(summaries)
2912    }
2913
2914    /// Writes the directory of the table this writer is on and says where it went.
2915    ///
2916    /// Everything [`Writer::finish`] used to do except the two writes that publish. Pulling it out
2917    /// is what lets a second table follow a first: the bytes of a closed table are complete and
2918    /// addressable while nothing yet points at them, and the pointer is the last write of the
2919    /// commit.
2920    ///
2921    /// # Errors
2922    ///
2923    /// If directory encoding or writing fails.
2924    fn close(&mut self) -> Result<Entry> {
2925        self.reclaim()?;
2926        self.flush_pending()?;
2927        // The rest of a table is its statistics, its dictionaries and its directory. The dictionary
2928        // work is charged as its own stage, because ranking a global dictionary can be most of what
2929        // this costs, and the rest as publish.
2930        let profile = self.profile.clone();
2931        let timing = profile.as_deref().map(|profile| profile.span(Stage::Publish));
2932        let before = self.at;
2933        let mut stripes = std::mem::take(&mut self.order)
2934            .into_iter()
2935            .zip(std::mem::take(&mut self.table.stripes))
2936            .collect::<Vec<_>>();
2937        stripes.sort_by_key(|(order, _)| order.0);
2938        let mut previous: Option<(u64, u64)> = None;
2939        for ((first, last), _) in &stripes {
2940            if previous.is_some_and(|previous| previous >= *first) {
2941                return Err(invalid("chunks did not arrive in source order"));
2942            }
2943            previous = Some(*last);
2944        }
2945        self.table.stripes = stripes.into_iter().map(|(_, stripe)| stripe).collect();
2946        drop(timing);
2947        let timing = profile.as_deref().map(|profile| profile.span(Stage::Dictionary));
2948        let placing = self.at;
2949        finish_dictionaries(&mut self.dictionaries)?;
2950        self.place_blocks()?;
2951        // The numeric frequencies and the global dictionaries read what is already written and
2952        // write nothing, so they run at the same time. Each was most of a second on `hits` with the
2953        // other waiting for it, and neither keeps every core busy on its own: each is as long as
2954        // its longest column. Both charge themselves, one span to each thread that works, because
2955        // they run on threads of their own and a span on this one would see their wall time and
2956        // none of their CPU.
2957        let this = &*self;
2958        let (numeric, closed) = std::thread::scope(|scope| {
2959            let numeric = scope.spawn(|| {
2960                let (frequencies, distincts): (Vec<Option<FrequencySummary>>, Vec<_>) =
2961                    this.numeric_frequencies()?.into_iter().unzip();
2962                let frequencies = frequencies
2963                    .into_iter()
2964                    .map(|held| held.map(Frequencies::Held))
2965                    .collect::<Vec<_>>();
2966                let pairs = this.pair_frequencies(&frequencies)?;
2967                Ok::<_, Error>((frequencies, distincts, pairs))
2968            });
2969            let closed = this.close_dictionaries();
2970            let numeric =
2971                numeric.join().map_err(|_| Error::internal("the native frequency thread panicked"));
2972            (numeric, closed)
2973        });
2974        let (frequencies, distincts, pairs) = numeric??;
2975        let closed = closed?;
2976        self.table.frequencies = frequencies;
2977        self.table.distincts = distincts;
2978        self.table.pair_frequencies = pairs;
2979        self.dictionaries = Vec::new();
2980        self.table.dictionary_payloads = vec![0; self.table.fields.len()];
2981        self.table.frequency_texts = vec![Vec::new(); self.table.fields.len()];
2982        self.table.host_groups = None;
2983        for (index, closed) in closed.into_iter().enumerate() {
2984            let Some(closed) = closed else { continue };
2985            let ClosedDictionary { distinct, frequencies, texts, hosts, encoded, payload } = closed;
2986            self.table.distincts[index] = Some(distinct);
2987            self.table.frequencies[index] = Some(Frequencies::Held(frequencies));
2988            self.table.frequency_texts[index] = texts;
2989            if hosts.is_some() {
2990                self.table.host_groups = hosts;
2991            }
2992            let offset = self.at;
2993            self.put(&encoded.index)?;
2994            self.put(&encoded.ranks)?;
2995            self.put(&encoded.grams)?;
2996            self.table.dictionary_payloads[index] = payload;
2997            let length = encoded
2998                .index
2999                .len()
3000                .checked_add(encoded.ranks.len())
3001                .and_then(|len| len.checked_add(encoded.grams.len()))
3002                .ok_or_else(|| invalid("dictionary page length overflow"))?;
3003            self.table.dictionaries[index] = Some(Page {
3004                offset,
3005                length: u32::try_from(length)
3006                    .map_err(|_| invalid("dictionary page length overflow"))?,
3007                hash: checksum(&encoded.index),
3008            });
3009        }
3010        drop(timing);
3011        let timing = profile.as_deref().map(|profile| profile.span(Stage::Publish));
3012        let placed = self.at - placing;
3013        self.write_stats()?;
3014        let directory = encode_directory(&self.table)?;
3015        if directory.len() > MAX_DIRECTORY {
3016            return Err(invalid("directory exceeds the configured bound"));
3017        }
3018        let offset = self.at;
3019        self.put(&directory)?;
3020        drop(timing);
3021        if let Some(profile) = &profile {
3022            profile.moved(Stage::Dictionary, 0, placed, 0);
3023            profile.moved(Stage::Publish, 0, self.at - before - placed, 0);
3024        }
3025        Ok(Entry {
3026            name: self.table.name.clone(),
3027            fields: self.table.fields.clone(),
3028            rows: self.table.rows,
3029            nonzero: table_nonzero_counts(&self.table),
3030            aggregates: table_aggregate_sums(&self.table),
3031            distincts: self.table.distincts.clone(),
3032            extremes: table_integer_extremes(&self.table),
3033            frequencies: table_complete_numeric_frequencies(&self.table),
3034            directory: Page {
3035                offset,
3036                length: u32::try_from(directory.len())
3037                    .map_err(|_| invalid("directory length overflow"))?,
3038                hash: checksum(&directory),
3039            },
3040        })
3041    }
3042
3043    /// Every global dictionary's page and statistics, by column, as many columns at a time as
3044    /// [`CLOSE_DICTIONARY_BYTES`] allows.
3045    ///
3046    /// The largest column that fits is the one taken next, so the long ones start first and the
3047    /// short ones fill in behind them. A column that does not fit waits for one that is closing to
3048    /// finish, unless nothing is closing, in which case it goes alone.
3049    fn close_dictionaries(&self) -> Result<Vec<Option<ClosedDictionary>>> {
3050        let mut jobs = self
3051            .dictionaries
3052            .iter()
3053            .enumerate()
3054            .filter_map(|(index, dictionary)| {
3055                dictionary
3056                    .as_ref()
3057                    .map(|dictionary| (index, dictionary, dictionary.closing_bytes()))
3058            })
3059            .collect::<Vec<_>>();
3060        jobs.sort_by_key(|&(_, _, bytes)| bytes);
3061        let mut closed = (0..self.dictionaries.len()).map(|_| None).collect::<Vec<_>>();
3062        let workers = close_workers().min(jobs.len());
3063        if workers <= 1 {
3064            for (index, dictionary, _) in jobs {
3065                closed[index] = Some(self.close_dictionary(index, dictionary)?);
3066            }
3067            return Ok(closed);
3068        }
3069        // The columns not taken yet, smallest first, and the bytes the ones closing now hold.
3070        let state = Mutex::new((jobs, 0_usize));
3071        let finished = Condvar::new();
3072        let profile = self.profile.as_deref();
3073        let pieces = std::thread::scope(|scope| {
3074            (0..workers)
3075                .map(|_| {
3076                    scope.spawn(|| {
3077                        let _timing = profile.map(|profile| profile.span(Stage::Dictionary));
3078                        let mut mine = Vec::new();
3079                        loop {
3080                            let mut held = state.lock().map_err(|_| {
3081                                Error::internal("a native dictionary worker panicked")
3082                            })?;
3083                            let (index, dictionary, bytes) = loop {
3084                                let (jobs, busy) = &mut *held;
3085                                if jobs.is_empty() {
3086                                    return Ok(mine);
3087                                }
3088                                let fits = jobs.iter().rposition(|&(_, _, bytes)| {
3089                                    *busy == 0
3090                                        || busy.saturating_add(bytes) <= CLOSE_DICTIONARY_BYTES
3091                                });
3092                                if let Some(at) = fits {
3093                                    let job = jobs.remove(at);
3094                                    *busy += job.2;
3095                                    break job;
3096                                }
3097                                held = finished.wait(held).map_err(|_| {
3098                                    Error::internal("a native dictionary worker panicked")
3099                                })?;
3100                            };
3101                            drop(held);
3102                            // Given back on the way out whether the close worked, failed or
3103                            // panicked, so that a worker waiting for room is never left waiting.
3104                            let _room = Room { state: &state, finished: &finished, bytes };
3105                            mine.push((index, self.close_dictionary(index, dictionary)?));
3106                        }
3107                    })
3108                })
3109                .collect::<Vec<_>>()
3110                .into_iter()
3111                .map(|handle| {
3112                    handle
3113                        .join()
3114                        .map_err(|_| Error::internal("a native dictionary worker panicked"))?
3115                })
3116                .collect::<Result<Vec<_>>>()
3117        })?;
3118        for (index, one) in pieces.into_iter().flatten() {
3119            closed[index] = Some(one);
3120        }
3121        Ok(closed)
3122    }
3123
3124    /// One global dictionary's page and statistics, built from what is already in the file.
3125    ///
3126    /// Nothing is written here, so that [`Self::close`] can run this beside the numeric frequencies
3127    /// and put the pages down afterwards in column order, which is where they always went. The
3128    /// column's values are decoded in here and dropped before it returns, and
3129    /// [`Self::close_dictionaries`] decides how many columns are in here at once.
3130    fn close_dictionary(
3131        &self,
3132        index: usize,
3133        dictionary: &GlobalDictionary,
3134    ) -> Result<ClosedDictionary> {
3135        let (order, flat, bases) = dictionary.ranked_with_values(Some(&self.file))?;
3136        // A code nothing counted is a code no non-null row of this column holds, which is the
3137        // empty string a null was written as and nothing else, because a code is only ever made by
3138        // a row asking for one.
3139        let distinct = dictionary.counts.iter().filter(|count| **count != 0).count() as u64;
3140        let (frequencies, texts) = code_frequency(dictionary, &flat, &bases)?;
3141        let hosts = if self.table.fields[index].name.eq_ignore_ascii_case("Referer") {
3142            host::build(index, dictionary, &flat, &bases)?
3143        } else {
3144            None
3145        };
3146        drop(flat);
3147        drop(bases);
3148        let encoded = encode_global_dictionary(dictionary, &order, &dictionary.placed, true)?;
3149        let payload = dictionary
3150            .placed
3151            .iter()
3152            .try_fold(0_u64, |sum, place| sum.checked_add(place.length))
3153            .ok_or_else(|| invalid("global dictionary payload overflow"))?;
3154        Ok(ClosedDictionary { distinct, frequencies, texts, hosts, encoded, payload })
3155    }
3156
3157    /// Writes the statistics sections for the table being closed, as far as the budget reaches.
3158    ///
3159    /// Called from [`Self::close`] after the last stripe and after the dictionaries, which is the
3160    /// first moment the table's column bytes are final and the last moment before the directory is
3161    /// encoded. Both halves matter: the budget is a share of the column bytes, and a section that
3162    /// went in after the directory would be a section the directory does not name.
3163    ///
3164    /// Nothing here can fail the write. A column whose gather came back blind gets no sections, a
3165    /// column the budget could not reach gets none, and section 3.1 says both of those plan the way
3166    /// they planned before statistics existed. The two errors that are returned are an encode
3167    /// failure and a section count past the bound, and neither is a thing a column can cause.
3168    fn write_stats(&mut self) -> Result<()> {
3169        let gathers = std::mem::take(&mut self.gathers);
3170        let rows = self.table.rows as u64;
3171        let mut payloads = Vec::new();
3172        for (column, gather) in gathers.into_iter().enumerate() {
3173            let Some(gather) = gather else { continue };
3174            // A gather that saw a different number of rows than the table committed is a gather
3175            // that missed some, and a distinct count over some of a column is the one error an
3176            // estimator cannot see coming. This has no way of happening today, since a table is
3177            // written once and every chunk goes through `flush_pending`, and that is exactly why it
3178            // is worth a line: it stays true only while that stays true.
3179            if gather.rows() != rows {
3180                continue;
3181            }
3182            let Some(stats) = gather.finish() else { continue };
3183            let mut summary = Vec::new();
3184            stats.summary.encode(&mut summary)?;
3185            let mut sketches = Vec::new();
3186            stats.sketches.encode(&mut sketches)?;
3187            payloads.push((column, summary, sketches));
3188        }
3189        if payloads.is_empty() {
3190            return Ok(());
3191        }
3192        let costs = payloads
3193            .iter()
3194            .map(|(_, summary, sketches)| summary.len() + sketches.len())
3195            .collect::<Vec<_>>();
3196        let allowance = stats::allowance(stats::column_bytes(&self.table), stats::BUDGET_SHARE);
3197        // Nothing is spent yet. A table this writer is closing is one it wrote from nothing, so the
3198        // only statistics sections it can have are the ones about to go in.
3199        let keep = stats::within(&costs, allowance, 0);
3200        for ((column, summary, sketches), _) in
3201            payloads.iter().zip(&keep).filter(|&(_, &keep)| keep)
3202        {
3203            let id = u64::try_from(*column).map_err(|_| invalid("column index overflow"))?;
3204            for (kind, bytes, header_bytes) in [
3205                // A summary is a header the whole way down: there is nothing behind it a reader
3206                // could decide not to read.
3207                (*section::SUMMARY, summary, summary.len() as u32),
3208                (*section::SKETCHES, sketches, rudb_stats::sketches::HEADER_BYTES),
3209            ] {
3210                let written = write_section(
3211                    &self.file,
3212                    &mut self.at,
3213                    &section::Attachment { kind, id, flags: 0, header_bytes, bytes },
3214                    self.generation,
3215                )?;
3216                self.table.sections.push(written);
3217            }
3218        }
3219        if self.table.sections.len() > MAX_SECTIONS {
3220            return Err(invalid("the table would name more sections than the bound allows"));
3221        }
3222        Ok(())
3223    }
3224
3225    /// Commits every table this writer has written and syncs the file before publishing its header
3226    /// slot.
3227    ///
3228    /// The table handed back is the one the writer was on, which is the last of them. Callers that
3229    /// wrote several already know the others, since they named them.
3230    ///
3231    /// # Errors
3232    ///
3233    /// If directory encoding, writing, or syncing fails.
3234    pub fn finish(mut self) -> Result<Table> {
3235        let entry = self.close()?;
3236        let profile = self.profile.take();
3237        let _timing = profile.as_deref().map(|profile| profile.span(Stage::Publish));
3238        let mut tables = std::mem::take(&mut self.closed);
3239        tables.push(entry);
3240        let catalog = encode_catalog(&tables, &self.views)?;
3241        if catalog.len() > MAX_DIRECTORY {
3242            return Err(invalid("catalog exceeds the configured bound"));
3243        }
3244        let offset = self.at;
3245        self.put(&catalog)?;
3246        if let Some(profile) = &profile {
3247            profile.moved(Stage::Publish, 0, catalog.len() as u64, 0);
3248        }
3249        // Every page and every table directory is on the disk before anything points at them. The
3250        // slot write below is what makes this generation the one a reader picks, so the order of
3251        // these two syncs is the whole of the commit.
3252        synced(&self.file, profile.as_deref())?;
3253        let slot = Slot {
3254            offset,
3255            length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
3256            generation: self.generation,
3257            hash: checksum(&catalog),
3258        };
3259        // The one write that is not an append, and the last one. It goes back over the slot in the
3260        // header, so it names its offset rather than going through `put`, and `at` does not move.
3261        // Which of the two slots it is alternates with the generation, so the one naming the
3262        // generation before this is still intact and still valid until this write lands.
3263        write_at(&self.file, slot_offset(self.generation), &slot.bytes())?;
3264        synced(&self.file, profile.as_deref())?;
3265        Ok(self.table)
3266    }
3267
3268    /// Commits a generation that changes the views and leaves every table exactly where it is.
3269    ///
3270    /// There was no way to do this before views existed, because everything that could change the
3271    /// catalog also wrote a table, so the only way to say something new about a file was to go
3272    /// through a table. A view is the first thing that can change on its own. Without this, adding
3273    /// a view to a database with eight tables in it would rewrite all eight, since the append path
3274    /// needs a table to append and the fallback is the whole file.
3275    ///
3276    /// It is the same commit as [`Writer::finish`] with nothing appended before it. The table
3277    /// entries are carried forward by directory pointer the way an append carries them, the new
3278    /// catalog goes on the end, and the slot write at the end is what publishes it.
3279    ///
3280    /// # Errors
3281    ///
3282    /// If the file has no valid committed directory, is not this build's format, or cannot be
3283    /// written.
3284    pub fn restate(path: impl AsRef<Path>, views: &[ViewEntry]) -> Result<()> {
3285        let path = path.as_ref();
3286        let (_, size, slot, bytes, _) = slot_bytes(path)?;
3287        let (closed, _) = decode_catalog(&bytes, size)?;
3288        let generation = slot
3289            .generation
3290            .checked_add(1)
3291            .ok_or_else(|| invalid("native file generation overflow"))?;
3292        let catalog = encode_catalog(&closed, views)?;
3293        if catalog.len() > MAX_DIRECTORY {
3294            return Err(invalid("catalog exceeds the configured bound"));
3295        }
3296        let file = OpenOptions::new().write(true).read(true).open(path).map_err(io)?;
3297        write_at(&file, size, &catalog)?;
3298        file.sync_all().map_err(io)?;
3299        let slot = Slot {
3300            offset: size,
3301            length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
3302            generation,
3303            hash: checksum(&catalog),
3304        };
3305        write_at(&file, slot_offset(generation), &slot.bytes())?;
3306        file.sync_all().map_err(io)?;
3307        Ok(())
3308    }
3309
3310    /// Adds exact count, sum, distinct, bound, and bounded frequency certificates to an older file without
3311    /// rewriting table pages. The old slot remains readable until the new catalog is synced.
3312    pub fn certify_summaries(path: impl AsRef<Path>) -> Result<()> {
3313        let path = path.as_ref();
3314        let (_, size, slot, bytes, _) = slot_bytes(path)?;
3315        let (mut entries, views) = decode_catalog(&bytes, size)?;
3316        let native = Catalog::open(path)?;
3317        for entry in &mut entries {
3318            let reader = native.table(&entry.name)?;
3319            entry.nonzero = reader_nonzero_counts(&reader)?;
3320            entry.aggregates = reader_aggregate_sums(&reader)?;
3321            entry.distincts = (0..entry.fields.len())
3322                .map(|column| reader.distinct_values(column))
3323                .collect::<Result<Vec<_>>>()?;
3324            entry.extremes = reader_integer_extremes(&reader)?;
3325            entry.frequencies = reader_complete_numeric_frequencies(&reader)?;
3326        }
3327        let generation = slot
3328            .generation
3329            .checked_add(1)
3330            .ok_or_else(|| invalid("native file generation overflow"))?;
3331        let catalog = encode_catalog(&entries, &views)?;
3332        if catalog.len() > MAX_DIRECTORY {
3333            return Err(invalid("catalog exceeds the configured bound"));
3334        }
3335        let file = OpenOptions::new().write(true).read(true).open(path).map_err(io)?;
3336        write_at(&file, size, &catalog)?;
3337        file.sync_all().map_err(io)?;
3338        let slot = Slot {
3339            offset: size,
3340            length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
3341            generation,
3342            hash: checksum(&catalog),
3343        };
3344        write_at(&file, slot_offset(generation), &slot.bytes())?;
3345        file.sync_all().map_err(io)?;
3346        Ok(())
3347    }
3348
3349    /// The earlier name for [`Self::certify_summaries`].
3350    pub fn certify_counts(path: impl AsRef<Path>) -> Result<()> {
3351        Self::certify_summaries(path)
3352    }
3353}
3354
3355/// Appends one run of bytes at `at` and moves it past them, answering where they went.
3356///
3357/// The append half of [`attach`], which cannot use [`Writer::put`] because it is not writing a
3358/// table. Every byte a section costs goes through here, so the offsets in an extent table come
3359/// from one place.
3360fn append(file: &File, at: &mut u64, bytes: &[u8]) -> Result<u64> {
3361    let offset = *at;
3362    write_at(file, offset, bytes)?;
3363    *at =
3364        at.checked_add(bytes.len() as u64).ok_or_else(|| invalid("native file length overflow"))?;
3365    Ok(offset)
3366}
3367
3368/// Writes one attachment's payload as extents and returns the entry that names it.
3369///
3370/// The split is by bytes, and the `first` of each extent is therefore a byte count. A section kind
3371/// whose extents should break on a row boundary instead will want to hand its extents over already
3372/// split; nothing needs that yet, and guessing at the shape of it now would be guessing.
3373fn write_section(
3374    file: &File,
3375    at: &mut u64,
3376    one: &section::Attachment<'_>,
3377    generation: u64,
3378) -> Result<Section> {
3379    // A payload of nothing is the exception, and it is not a special case so much as a different
3380    // reading of the same field: an entry with no bytes has no header to be longer than them, and
3381    // `header_bytes` is what the structure would have cost. See `Section::refused`.
3382    if !one.bytes.is_empty() && one.header_bytes as usize > one.bytes.len() {
3383        return Err(invalid("a section's header is longer than its payload"));
3384    }
3385    let mut extents = Vec::new();
3386    let mut first = 0_u64;
3387    for chunk in one.bytes.chunks(section::MAX_EXTENT as usize) {
3388        let offset = append(file, at, chunk)?;
3389        extents.push(section::Extent {
3390            offset,
3391            length: u32::try_from(chunk.len()).map_err(|_| invalid("extent length overflow"))?,
3392            hash: checksum(chunk),
3393            first,
3394        });
3395        first += chunk.len() as u64;
3396    }
3397    let mut table = Vec::with_capacity(extents.len() * section::EXTENT_BYTES);
3398    section::encode_extents(&extents, &mut table)?;
3399    // A payload of nothing is a section of no extents and no extent table, and its `extent_page`
3400    // is zero rather than the end of the file. Section 3.7 wants that entry to exist: it is how a
3401    // relationship that did not fit the budget is recorded as not built rather than forgotten.
3402    let extent_page = if table.is_empty() { 0 } else { append(file, at, &table)? };
3403    Ok(Section {
3404        kind: one.kind,
3405        id: one.id,
3406        generation,
3407        extents: u32::try_from(extents.len()).map_err(|_| invalid("too many extents"))?,
3408        extent_page,
3409        extent_bytes: u32::try_from(table.len()).map_err(|_| invalid("extent table overflow"))?,
3410        hash: checksum(&table),
3411        flags: one.flags,
3412        header_bytes: one.header_bytes,
3413    })
3414}
3415
3416/// Attaches graph sections to a table already committed in a file, without rewriting a page.
3417///
3418/// This is the second pass spec/graph/03-the-file-format.md section 3.8 asks for. A key map has to
3419/// exist before the link that uses it can be built, and it is built by reading the key column back,
3420/// so the structures of a table cannot be written during the load that wrote the table. They are
3421/// written afterwards, by this, and the file in between the two is a correct file that answers
3422/// every query more slowly.
3423///
3424/// Nothing is overwritten. The payloads, the extent tables, the new directory for this table and
3425/// the new catalog all go on the end of the file past the committed generation, and the last write
3426/// is the header slot, exactly as [`Writer::finish`] does it. So a crash anywhere in here leaves
3427/// the generation before it intact, and leaves unreferenced trailing bytes that the next commit
3428/// writes past.
3429///
3430/// An attachment replaces any section of the same kind and id, and every other section is carried
3431/// through untouched, including one whose kind this build does not know. The table's own generation
3432/// is carried through too, because attaching a section moves no row: see [`Table::generation`].
3433///
3434/// # Errors
3435///
3436/// If the file has no valid committed directory, is an older format than this build writes, holds
3437/// no table of that name, names a section whose payload cannot be written, or would end up naming
3438/// more sections than the format allows.
3439pub fn attach(
3440    path: impl AsRef<Path>,
3441    table: &str,
3442    attachments: &[section::Attachment<'_>],
3443) -> Result<Table> {
3444    let path = path.as_ref();
3445    let (_, size, slot, bytes, _) = slot_bytes(path)?;
3446    let (mut entries, views) = decode_catalog(&bytes, size)?;
3447    let at = entries
3448        .iter()
3449        .position(|entry| entry.name == table)
3450        .ok_or_else(|| invalid(&format!("the file holds no table called {table}")))?;
3451    let file = OpenOptions::new().write(true).read(true).open(path).map_err(io)?;
3452    let mut version = [0; 4];
3453    read_at(&file, 8, &mut version)?;
3454    let version = u32::from_le_bytes(version);
3455    // Readable is not the same as writable. A format 22 file has no section table, and giving its
3456    // directory one without moving the number in its header would leave a file that claims to be
3457    // format 22 and is not, which is worse than refusing. Rewriting it with this build is the
3458    // answer, and the format is at 0.3.x, so nobody has one of these that this project did not
3459    // just make.
3460    if version != FORMAT {
3461        return Err(invalid(&format!(
3462            "the file is format {version} and a graph section needs format {FORMAT}, so it has \
3463             to be written again"
3464        )));
3465    }
3466    let mut directory = vec![0; entries[at].directory.length as usize];
3467    read_at(&file, entries[at].directory.offset, &mut directory)?;
3468    if checksum(&directory) != entries[at].directory.hash {
3469        return Err(invalid(&format!("the directory of table {table} does not checksum")));
3470    }
3471    let mut held = decode_directory(&directory, size)?;
3472    let mut cursor = size;
3473    for one in attachments {
3474        let written = write_section(&file, &mut cursor, one, held.generation)?;
3475        held.sections.retain(|old| !(old.kind == one.kind && old.id == one.id));
3476        held.sections.push(written);
3477    }
3478    if held.sections.len() > MAX_SECTIONS {
3479        return Err(invalid("the table would name more sections than the bound allows"));
3480    }
3481    let encoded = encode_directory(&held)?;
3482    if encoded.len() > MAX_DIRECTORY {
3483        return Err(invalid("directory exceeds the configured bound"));
3484    }
3485    let offset = append(&file, &mut cursor, &encoded)?;
3486    entries[at].directory = Page {
3487        offset,
3488        length: u32::try_from(encoded.len()).map_err(|_| invalid("directory length overflow"))?,
3489        hash: checksum(&encoded),
3490    };
3491    // The views the file already had, written back unchanged. Attaching a section to a table says
3492    // nothing about a view and must not drop one.
3493    let catalog = encode_catalog(&entries, &views)?;
3494    if catalog.len() > MAX_DIRECTORY {
3495        return Err(invalid("catalog exceeds the configured bound"));
3496    }
3497    let offset = append(&file, &mut cursor, &catalog)?;
3498    file.sync_all().map_err(io)?;
3499    let generation =
3500        slot.generation.checked_add(1).ok_or_else(|| invalid("native file generation overflow"))?;
3501    let committed = Slot {
3502        offset,
3503        length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
3504        generation,
3505        hash: checksum(&catalog),
3506    };
3507    write_at(&file, slot_offset(generation), &committed.bytes())?;
3508    file.sync_all().map_err(io)?;
3509    Ok(held)
3510}
3511
3512/// One column's frequency synopsis as values with their row counts, shared by every clone of a
3513/// reader.
3514type Synopsis = Arc<Vec<(Value, u64)>>;
3515
3516/// Reads committed native column pages without holding the table in memory.
3517#[derive(Debug, Clone)]
3518pub struct Reader {
3519    file: Arc<File>,
3520    table: Arc<Table>,
3521    dictionaries: Arc<Vec<OnceLock<Arc<Vector>>>>,
3522    /// Held while a global dictionary is being opened, one per column.
3523    ///
3524    /// The [`OnceLock`] above says whether one has been opened, which is the question a reader that
3525    /// already has it needs answered and is free. It does not say whether one is being opened, and
3526    /// the difference matters because every worker of a scan wants the same dictionary at the same
3527    /// moment. Without this they all miss, all read the page, all verify it and all decode it, and
3528    /// all but one throw the answer away. ClickBench 38 reads the URL dictionary, which is 515,958
3529    /// entries, and was paying for it twice.
3530    loading: Arc<Vec<Mutex<()>>>,
3531    /// Each column's frequency synopsis as values, the first time anything asks for it. See
3532    /// [`Reader::decode_frequencies`].
3533    frequency_values: Arc<Vec<OnceLock<Synopsis>>>,
3534    /// Stored frequency sections are decoded once per open table. A small directory can hold the
3535    /// summary inline, but a larger one otherwise rereads and decodes the same section on every
3536    /// plan and every summary-backed aggregate.
3537    frequency_summaries: Arc<Vec<OnceLock<Arc<FrequencySummary>>>>,
3538    /// How many global dictionaries have been opened. A scan of a dictionary column should open its
3539    /// dictionary once however many workers it has, and the test that says so is the only thing
3540    /// keeping it that way.
3541    opened: Arc<AtomicUsize>,
3542    /// The membership sieves of one stripe of one column, by column and then by stripe, read the
3543    /// first time a probe asks about them. A query filters on one or two columns and never looks at
3544    /// the rest, so reading these at open would be the whole index for the sake of a fraction of it.
3545    sieves: Arc<Vec<Vec<SieveSlot>>>,
3546    /// The per part ranges of one stripe of one column, by column and then by stripe, read the
3547    /// first time something compares that column and kept after that.
3548    part_ranges: Arc<Vec<Vec<RangeSlot>>>,
3549    /// Which stripe and which part of it every part of the table is, by table wide part number.
3550    places: Arc<Vec<Place>>,
3551    cache: Arc<Shelf>,
3552    /// Where the pages above are counted against the database's budget. See [`PagePool`].
3553    pool: PagePool,
3554    /// How many whole stripe pages have been read, which is what the sharing above is judged on. A
3555    /// scan of a column should read each of its stripes once however many workers it has.
3556    pages: Arc<AtomicUsize>,
3557    /// How many index sections have been read. A scan of a column should read each of its stripes
3558    /// once here too, and the test that says so is the only thing keeping it that way.
3559    indexes: Arc<AtomicUsize>,
3560    /// The file's size when it was opened, for [`Reader::layout`].
3561    size: u64,
3562    /// The committed directory's size, for [`Reader::layout`].
3563    directory: u64,
3564    /// What opening the file cost, which is a number rather than a claim.
3565    opening: Opening,
3566}
3567
3568/// What [`Reader::open`] read before it returned.
3569///
3570/// `spec/stats/04-in-memory.md` section 4.2 says opening a table reads the header and the directory
3571/// and nothing else, and once that document's statistics are in the file the tempting change is to
3572/// load a column summary or two on the way past, because they are small and the next query will
3573/// want them. A hundred milliseconds of that is a hundred milliseconds nobody asked for, and an
3574/// embedded database is opened by processes that are about to run one trivial query.
3575///
3576/// So the claim gets a number. Both of these are fixed by the schema and the stripe count and are
3577/// independent of how many rows the file holds, and the test that says so is what stops the
3578/// tempting change from landing quietly.
3579#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
3580pub struct Opening {
3581    /// How many times the file was read. The header, then each directory slot that looked valid
3582    /// enough to check, so three at the most.
3583    pub reads: u32,
3584    /// How many bytes those reads asked for.
3585    pub bytes: u64,
3586}
3587
3588/// What a reader has read, while it was being opened and since.
3589#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
3590pub struct Reads {
3591    /// What opening cost, before any query had been planned.
3592    pub opening: Opening,
3593    /// Whole stripe pages read since.
3594    pub pages: usize,
3595    /// Index sections read since.
3596    pub indexes: usize,
3597    /// Global dictionaries opened since. One per dictionary column that a query touched, however
3598    /// many workers touched it, which is a claim only a test can keep true.
3599    pub dictionaries: usize,
3600}
3601
3602/// Where one table wide part number lands.
3603#[derive(Debug, Clone, Copy)]
3604struct Place {
3605    stripe: u32,
3606    part: u32,
3607    rows: u32,
3608}
3609
3610/// One part's bytes inside one column page.
3611#[derive(Debug, Clone, Copy)]
3612struct PartSpan {
3613    start: usize,
3614    length: usize,
3615    hash: u64,
3616}
3617
3618/// What a reader holds for one stripe of one column.
3619///
3620/// The index is small and is loaded whether the caller wants the whole page or one part of it. The
3621/// page is loaded only by a scan, because a sparse fetch that wants a thousand rows out of sixty
3622/// four thousand would be reading sixty four times what it uses.
3623#[derive(Debug, Clone)]
3624struct CachedColumn {
3625    stripe: usize,
3626    index: Arc<Vec<PartSpan>>,
3627    page: Option<Arc<Vec<u8>>>,
3628}
3629
3630/// One column's stripes a reader holds, and which of them somebody is reading right now.
3631///
3632/// The pages are one slot per stripe of the table rather than a list of the ones being kept, so
3633/// finding a page is an index and not a walk. That matters because the walk happened under the
3634/// lock, once per part per column, and a scan that gives a whole stripe to each of thirty two
3635/// workers keeps enough pages that walking them was the longest thing the lock was held for. The
3636/// slots cost a pointer per stripe per column, which on the ClickBench file is eight kilobytes
3637/// against the forty megabytes of pages they point at. `order` is which of them are filled, oldest
3638/// first, because that is the one thing the slots cannot say by themselves.
3639///
3640/// `loading` is what keeps a scan from reading the same page once per worker. It is a list and not
3641/// a set because it holds at most one stripe per worker on the column and is walked far less often
3642/// than a hash of it would be built.
3643///
3644/// `index` is every index this reader has ever read for the column, one slot per stripe, and it is
3645/// never evicted. An index is a few hundred bytes and a page is a quarter of a megabyte, so the two
3646/// do not belong under the same budget. Riding in the page cache meant a worker that came back to a
3647/// stripe after its page had been evicted read the index again with it, which on the full
3648/// ClickBench file was about thirteen hundred reads out of a hundred and fourteen thousand.
3649#[derive(Debug, Default)]
3650struct Cached {
3651    pages: Vec<Option<Resident>>,
3652    loading: Vec<usize>,
3653    index: Vec<Option<Arc<Vec<PartSpan>>>>,
3654}
3655
3656/// One page a reader holds, and whether anyone has read it since the pool last looked.
3657#[derive(Debug, Clone)]
3658struct Resident {
3659    page: Arc<Vec<u8>>,
3660    used: Arc<AtomicBool>,
3661}
3662
3663/// Every column's pages of one reader, with how many each column holds and the floor under that.
3664#[derive(Debug)]
3665struct Shelf {
3666    columns: Vec<Mutex<Cached>>,
3667    /// How many pages each column holds right now. Counted outside the column locks so that the
3668    /// pool can tell whether a column is at its floor without taking a lock it might be under.
3669    held: Vec<AtomicUsize>,
3670    /// How many stripes of one column are kept whatever the budget says. See
3671    /// [`CACHED_STRIPES_PER_COLUMN`] for what sets it and [`Reader::keep_stripes`] for who raises it.
3672    kept: AtomicUsize,
3673}
3674
3675/// The pages every reader of one database keeps, under one budget in bytes.
3676///
3677/// A reader lives as long as the database does, so the pages it holds are what the next query finds
3678/// already in memory. They used to be four stripes a column, oldest out first, which on TPC-H SF1
3679/// meant every query read every page of lineitem off the file again and paid the system call for
3680/// it. Keeping every page there costs 38 MB and took a third of the system time off the suite.
3681///
3682/// So the question is no longer how many stripes a column keeps but how many bytes the database
3683/// does, and one budget answers it for every reader at once. A table nobody queries gives its pages
3684/// up to one that is being queried, which a count per column cannot do.
3685///
3686/// Pages leave by the clock. Each has a bit a read sets, and when the pool is over budget it walks
3687/// from the oldest: a page with the bit set loses the bit and goes round again, and a page without
3688/// it goes. That keeps what is read over and over and lets a page one scan read once go first.
3689///
3690/// The old count is still a floor. A column never gives up a page while it holds four or fewer,
3691/// because a scan whose workers evict each other's pages reads a quarter of a megabyte for every
3692/// part it takes, and a budget of zero is the cache as it was before the pool existed.
3693#[derive(Debug, Clone, Default)]
3694pub struct PagePool {
3695    ring: Arc<Mutex<Ring>>,
3696    budget: Arc<AtomicUsize>,
3697}
3698
3699#[derive(Debug, Default)]
3700struct Ring {
3701    held: VecDeque<Held>,
3702    bytes: usize,
3703}
3704
3705/// One page in the pool, pointing back at the reader that holds it.
3706///
3707/// Weak, because a reader that has gone, which every reader does at a checkpoint, should take its
3708/// pages with it and not have them kept alive by the pool.
3709#[derive(Debug)]
3710struct Held {
3711    shelf: Weak<Shelf>,
3712    column: usize,
3713    stripe: usize,
3714    bytes: usize,
3715    used: Arc<AtomicBool>,
3716}
3717
3718impl PagePool {
3719    /// A pool that keeps up to `budget` bytes of pages beyond each column's floor.
3720    #[must_use]
3721    pub fn new(budget: usize) -> Self {
3722        let pool = Self::default();
3723        pool.budget.store(budget, Atomic::Relaxed);
3724        pool
3725    }
3726
3727    /// The bytes of pages the pool is counting now.
3728    ///
3729    /// # Panics
3730    ///
3731    /// If the pool's lock is poisoned, which takes a panic while it was held.
3732    #[must_use]
3733    pub fn bytes(&self) -> usize {
3734        self.ring.lock().map_or(0, |ring| ring.bytes)
3735    }
3736
3737    /// Counts a page a reader has just taken in, and lets pages go until the pool is back under its
3738    /// budget or it has looked at every page once.
3739    ///
3740    /// Called with no column lock held. The pages that go are chosen under the pool's lock and
3741    /// dropped under their column's lock afterwards, so no thread ever holds both.
3742    fn admit(&self, held: Held) {
3743        let budget = self.budget.load(Atomic::Relaxed);
3744        let mut gone = Vec::new();
3745        {
3746            let Ok(mut ring) = self.ring.lock() else { return };
3747            ring.bytes += held.bytes;
3748            ring.held.push_back(held);
3749            // One lap and no more. A page read since the last pass loses its bit on this one and
3750            // can only go on a later one, which is the second chance the clock is named for.
3751            let mut looked = 0;
3752            let limit = ring.held.len();
3753            while ring.bytes > budget && looked < limit {
3754                looked += 1;
3755                let Some(entry) = ring.held.pop_front() else { break };
3756                let Some(shelf) = entry.shelf.upgrade() else {
3757                    ring.bytes -= entry.bytes;
3758                    continue;
3759                };
3760                if entry.used.swap(false, Atomic::Relaxed) {
3761                    ring.held.push_back(entry);
3762                    continue;
3763                }
3764                let count = &shelf.held[entry.column];
3765                if count.load(Atomic::Relaxed) <= shelf.kept.load(Atomic::Relaxed).max(1) {
3766                    ring.held.push_back(entry);
3767                    continue;
3768                }
3769                count.fetch_sub(1, Atomic::Relaxed);
3770                ring.bytes -= entry.bytes;
3771                gone.push((shelf, entry));
3772            }
3773            // A reader that has gone leaves its entries behind, and with a budget nobody reaches
3774            // they would pile up one checkpoint after another. The front is where the oldest are.
3775            while ring.held.front().is_some_and(|entry| entry.shelf.strong_count() == 0) {
3776                if let Some(entry) = ring.held.pop_front() {
3777                    ring.bytes -= entry.bytes;
3778                }
3779            }
3780        }
3781        for (shelf, entry) in gone {
3782            let Ok(mut cached) = shelf.columns[entry.column].lock() else { continue };
3783            if let Some(slot) = cached.pages.get_mut(entry.stripe) {
3784                if slot.as_ref().is_some_and(|slot| Arc::ptr_eq(&slot.used, &entry.used)) {
3785                    *slot = None;
3786                }
3787            }
3788        }
3789    }
3790}
3791
3792/// Stripes of one column a reader keeps the bytes of, when nobody has asked for more.
3793///
3794/// This has to hold at least as many stripes as a column has workers in it at once, or the workers
3795/// evict each other's pages and read them again. Four is what a scan that hands parts out in order
3796/// needs, because then every worker is within a few parts of every other and at most a couple of
3797/// stripes are open at a time. A scan that hands a whole stripe to each worker has one stripe open
3798/// per worker for the length of that stripe, and it says so with [`Reader::keep_stripes`] rather
3799/// than paying for sixteen slots on every table that is read one part at a time.
3800///
3801/// It multiplies by the page size, which is a quarter of a megabyte for a four byte column, and by
3802/// the number of columns a query touches.
3803const CACHED_STRIPES_PER_COLUMN: usize = 4;
3804
3805/// The sieves of one stripe of one column, once somebody has asked for them.
3806type SieveSlot = OnceLock<Arc<Vec<Option<Sieve>>>>;
3807
3808type RangeSlot = OnceLock<Arc<Vec<Range>>>;
3809
3810#[derive(Debug)]
3811struct NativeText {
3812    file: Arc<File>,
3813    /// How many values the dictionary holds.
3814    values: usize,
3815    /// Where each value ends inside its payload block, packed at `offset_bits` in runs of
3816    /// [`TEXT_OFFSET_RUN`].
3817    ///
3818    /// Ends rather than starts, because then a block of 1,024 values is 1,024 numbers rather than
3819    /// 1,025: the start of a value is the end of the one before it, and the first value of a block
3820    /// starts at zero by construction. Relative to the block rather than to the payload, because a
3821    /// reader decodes a whole block and slices it, so an offset into the payload is a number it
3822    /// would have to subtract a base from anyway.
3823    offsets: Vec<u8>,
3824    /// Bits one offset is packed at, which is what the largest block of this column spans and is the
3825    /// same for every block of it.
3826    offset_bits: usize,
3827    /// The same ends unpacked, built once enough readers have asked for one at a time.
3828    ///
3829    /// Reading one offset out of the packed form costs about fifty instructions: a division to find
3830    /// the run, a bounds check to slice it, a shift to reach the bit the value starts at and a
3831    /// narrowing on the way out. That is the right price for a reader that wants a handful. It is
3832    /// the wrong price for `STRLEN` over a column, which asks for one per row and nothing else, and
3833    /// where a million of them was a third of ClickBench 28.
3834    ///
3835    /// With the ends unpacked every read is a load, and a vector of lengths is one loop over them.
3836    /// The table is built only once the reads say it will be used, which is what
3837    /// [`Self::ends_worth_unpacking`] decides and [`Self::ends_asked`] counts towards, because a
3838    /// table built for a reader that wanted three values is four bytes a value spent on nothing.
3839    value_ends: OnceLock<Option<Vec<u32>>>,
3840    /// The length of every value, worked out of [`Self::value_ends`] the first time a vector of
3841    /// lengths is asked for.
3842    ///
3843    /// A length out of the ends is two loads, a test for whether the value opens its block and a
3844    /// check that it does not end before it starts, which came to thirteen instructions a row on
3845    /// ClickBench 28. Out of this it is one load. The order is checked once for the whole table
3846    /// while it is built, and a column that fails it gets no table and goes on reading the ends,
3847    /// which is where the error is reported. Four bytes a value, and only for a column something
3848    /// has asked the length of a vector at a time.
3849    value_lens: OnceLock<Option<Vec<u32>>>,
3850    /// How many single offset reads have come in while the table is not built.
3851    ///
3852    /// Relaxed, and read only against a threshold, so two threads racing here means the table is
3853    /// built one read early or one read late. Counting stops the moment the table exists, because
3854    /// [`OnceLock::get`] settles it before this is touched.
3855    ends_asked: AtomicUsize,
3856    /// How many entries the sorted order has, which is the value count.
3857    ranks: usize,
3858    /// Where the sorted order starts in the file. It is read a block at a time and only when
3859    /// something searches it, so a query that never compares this column against a literal never
3860    /// touches it at all.
3861    rank_at: u64,
3862    /// Where each block of the sorted order ends, as a byte offset from `rank_at`. A block is packed
3863    /// at whatever width its own heads need, so unlike the entries it replaced its length is not
3864    /// arithmetic on the block number.
3865    rank_ends: Vec<u64>,
3866    rank_hashes: Vec<u64>,
3867    rank_blocks: Vec<OnceLock<Result<Vec<u8>>>>,
3868    /// Bits one code is packed at, which is what the value count needs and is the same for every
3869    /// block of the column.
3870    code_bits: usize,
3871    /// The sorted order turned round, built the first time a reader asks for it.
3872    ///
3873    /// Four bytes per value against the four the offsets already hold, so a column that has this is
3874    /// carrying half again what it carried before rather than something of a new order. It is built
3875    /// only when something asks, which is a grouped min or max over this column and nothing else,
3876    /// and that reader was going to read the payload of this column once per row otherwise.
3877    code_ranks: OnceLock<Option<Vec<u32>>>,
3878    /// Where each block of the payload starts in the file, and how many stored bytes it is.
3879    ///
3880    /// Absolute rather than an offset from a base the blocks share, because a block is written the
3881    /// moment it fills and what comes after it in the file is whatever the load wrote next. A file
3882    /// old enough to have them back to back is read into these same two lists by adding the base to
3883    /// the ends it carries, so nothing below here knows which kind of file it came from.
3884    starts: Vec<u64>,
3885    lengths: Vec<u64>,
3886    hashes: Vec<u64>,
3887    /// Conservative four-byte substring signatures, read only by a compatible LIKE filter.
3888    grams: Option<NativeGrams>,
3889    /// The payload, read and decoded a block at a time and kept after that.
3890    blocks: Vec<OnceLock<Result<Vec<u8>>>>,
3891    /// How many decoded payload bytes this column keeps before a sweep stops keeping what it reads.
3892    /// [`TEXT_KEEP_BUDGET`] everywhere but in the test of the ceiling.
3893    keep_budget: usize,
3894    /// Roughly how many decoded payload bytes are being kept, which is what [`TEXT_KEEP_BUDGET`]
3895    /// is measured against.
3896    ///
3897    /// Roughly, because two threads that keep the same block at the same time both add its length
3898    /// while [`OnceLock`] keeps one of the two. That makes the count read high and the budget bind
3899    /// a little early, which is the harmless direction, and it costs one relaxed add a block rather
3900    /// than a lock on the path every scan of a string column goes through.
3901    payload_kept: AtomicUsize,
3902    /// Which payload blocks a sweep has decoded before, one flag a block.
3903    ///
3904    /// A sweep keeps a block the second time it decodes it and not the first. A process that runs
3905    /// one statement, which is how a benchmark or a script uses the engine, sweeps each block once
3906    /// and so keeps nothing: on ten million rows a `URL LIKE` held 396 MB with every block kept and
3907    /// 97 MB with none, for the same processor time. A session that asks again pays the decode one
3908    /// more time and reads kept blocks from then on, under the same [`TEXT_KEEP_BUDGET`].
3909    swept: Vec<AtomicBool>,
3910    /// The boundaries this dictionary has already been searched for, by the value searched for.
3911    ///
3912    /// A search is the expensive thing this type does. It settles a probe on the stored head where
3913    /// it can and reads a value where it cannot, and reading a value decodes the payload block it
3914    /// sits in, so one search can cost several blocks. The thing that makes remembering worth it is
3915    /// that the same search comes back: a top N asks once a chunk whether anything left can beat its
3916    /// worst candidate, and the worst candidate settles long before the chunks run out.
3917    ///
3918    /// Shared across the instances of a scan rather than kept per instance, because each of them has
3919    /// its own worst candidate and all of them are searching the same dictionary. One lock per chunk
3920    /// is nothing next to a probe of a file.
3921    ///
3922    /// Bounded by [`TEXT_SEARCH_MEMO`] and emptied rather than evicted when it is full. What fills
3923    /// it is a top N improving its bound, which happens a few dozen times and then stops, so the
3924    /// bound is there for the filter that searches for a different literal every chunk rather than
3925    /// for anything this is meant to help.
3926    searched: Mutex<HashMap<Vec<u8>, (usize, bool)>>,
3927}
3928
3929#[derive(Debug)]
3930struct NativeGrams {
3931    start: u64,
3932    length: usize,
3933    hash: u64,
3934    loaded: OnceLock<Result<Vec<u8>>>,
3935}
3936
3937/// How many searched for values a column's dictionary remembers the boundary of.
3938///
3939/// See [`NativeText::searched`]. Small because the case it is for repeats one value, not because a
3940/// larger one would be wrong.
3941const TEXT_SEARCH_MEMO: usize = 64;
3942
3943/// How many values of a dictionary go in one block of the payload.
3944///
3945/// The block is the unit the string cascade encodes, the unit a checksum covers, and the unit a
3946/// reader has to decode to get at a single value, so it is the one number the payload format turns
3947/// on. Blocking by values rather than by bytes is what keeps a value out of two blocks at once: the
3948/// block holding a code is `code / TEXT_PAYLOAD_VALUES` and nothing has to be stitched.
3949///
3950/// A probe on the five ClickBench columns that have a dictionary worth the name, written up on
3951/// #347, measured the ratio and the decode speed at 128, 256, 512, 1,024 and 4,096 values. Both get
3952/// better all the way up, because front coding and the LZ matcher have more to look back at and
3953/// because the per chunk setup is spread over more values. What stops it is the point read: a query
3954/// that wants ten values has to decode ten blocks, so the block is what a lookup costs. At 1,024
3955/// values a block is between 67 KB and 394 KB decoded across those five columns, and the ratios are
3956/// 2.3 to 4.5. Going up to 4,096 buys two to six percent more and makes a block as much as 1.5 MB.
3957/// Going down to 512 gives up five to nine percent.
3958const TEXT_PAYLOAD_VALUES: usize = 1024;
3959
3960/// Two KiB per payload block makes a four-byte substring a useful negative test without keeping a
3961/// large lookup table. The load and file-size costs must pass the same end-to-end gate as queries.
3962const TEXT_GRAM_BYTES: usize = 2048;
3963
3964/// A fast mixing step for exactly four bytes, shared by load and query.
3965fn gram_bits(bytes: &[u8]) -> [usize; 2] {
3966    let original = u32::from_le_bytes(bytes.try_into().expect("a four-byte gram"));
3967    let mut first = original ^ (original >> 16);
3968    first = first.wrapping_mul(0x7feb_352d);
3969    first ^= first >> 15;
3970    let mut second = original ^ (original >> 17);
3971    second = second.wrapping_mul(0x846c_a68b);
3972    second ^= second >> 16;
3973    let mask = TEXT_GRAM_BYTES * 8 - 1;
3974    [(first as usize) & mask, (second as usize) & mask]
3975}
3976
3977/// How many decoded payload bytes one dictionary keeps before a sweep stops keeping what it reads.
3978///
3979/// A sweep of the whole dictionary decodes every block whatever it does, and the only question is
3980/// whether it hangs on to them. Keeping all of them is 4.2 GB on ClickBench `URL` at a hundred
3981/// million rows, which is what #997 was right to stop. Keeping none of them means the next query
3982/// asking the same thing decodes all of it again, and on the same column at a million rows that
3983/// took a `LIKE` from 2.7 ms to 16.2 ms, because the decode used to be paid once by a session and
3984/// is now paid by every statement in it. Neither end is the answer. A bound is.
3985///
3986/// So a sweep keeps what it decodes until the column is holding this much and decodes without
3987/// keeping after that. At a million rows the five ClickBench string columns decode to between 8 MB
3988/// and 85 MB, so they sit inside it and a repeated `LIKE` reads a decoded block rather than a
3989/// stored one. At a hundred million rows `URL` fills it and the rest of that column is read and
3990/// dropped, which is the old cost on the part that does not fit and none of the old footprint.
3991///
3992/// Two hundred and fifty six megabytes a column is a number and not a policy, and the policy is
3993/// what should replace it: this wants to be a buffer pool over the whole database, sized against
3994/// the memory limit the session was given, with the blocks of every column competing for it and the
3995/// least useful one evicted. That is F2 work. What is here is the part of it that can be written
3996/// without an eviction order, which is a ceiling.
3997const TEXT_KEEP_BUDGET: usize = 256 * 1024 * 1024;
3998
3999/// The length of every value out of where each one ends inside its payload block, or `None` for
4000/// ends that go backwards somewhere inside a block.
4001///
4002/// A value that opens a block starts at zero and every other one starts where the value before it
4003/// ends, so a block is a run of differences.
4004fn lengths_of(ends: &[u32]) -> Option<Vec<u32>> {
4005    let mut lens = Vec::with_capacity(ends.len());
4006    for block in ends.chunks(TEXT_PAYLOAD_VALUES) {
4007        let mut start = 0;
4008        for &end in block {
4009            lens.push(end.checked_sub(start)?);
4010            start = end;
4011        }
4012    }
4013    Some(lens)
4014}
4015
4016/// How many offsets go in one packed run.
4017///
4018/// A payload block holds 1,024 values and `bitpack::pack_tail` takes fewer than 1,024 at a time,
4019/// since a whole unit of that many belongs in the transposed layout instead. So the offsets of a
4020/// block go in two runs. Five hundred and twelve values at any width is a whole number of bytes, so
4021/// a run starts where a multiply says it does and nothing is padded.
4022const TEXT_OFFSET_RUN: usize = 512;
4023
4024/// Bytes at the front of a global dictionary index: the value count, the values a payload block
4025/// holds, the block count and the bits an offset is packed at.
4026const DICTIONARY_HEADER: usize = 16;
4027
4028/// Set beside the offset width in the fourth word of a global dictionary index, meaning each
4029/// payload block says where in the file it starts and how long it is, rather than sitting directly
4030/// behind the block before it.
4031///
4032/// In that word rather than in a word of its own because the width is at most 32 and lives in a
4033/// `u32`, so the top of it has never been anything. A build old enough not to know the flag reads
4034/// the file's format before it reads any of this and refuses it there, and if it somehow did get
4035/// here it would find an offset width of two billion and say so.
4036///
4037/// The point of the flag is that a block written the moment it fills does not know what will be
4038/// written after it, so the payload of a column cannot be one run of bytes unless the whole column
4039/// is held until the file is closed. That is the memory the load cannot afford. What it costs is
4040/// eight bytes a block, against the block being a thousand values.
4041const DICTIONARY_SCATTERED: u32 = 1 << 31;
4042/// The dictionary index carries one four-byte substring signature per payload block.
4043const DICTIONARY_GRAMS: u32 = 1 << 30;
4044
4045/// How many entries of a dictionary's sorted order sit in one block that is read and checked as a
4046/// unit.
4047///
4048/// Five hundred and twelve entries is between two and three kilobytes on the ClickBench string
4049/// columns, which is well under a page. A binary search over half a million entries makes nineteen
4050/// probes, and the first ten land in ten different blocks while the last nine land in the one block
4051/// that holds the answer, so the whole search reads about thirty kilobytes of a megabyte of order. A
4052/// smaller block would save a little on the early probes, cost a checksum and an end list four times
4053/// as long, and give the heads less to share a base with. A larger one would read more than it uses
4054/// on every probe.
4055const TEXT_RANK_BLOCK: usize = 512;
4056
4057/// Bytes at the front of a rank block, which is the base of its heads and the width they are packed
4058/// at.
4059///
4060/// An entry used to be twelve bytes flat, eight for the head and four for the code, and on the five
4061/// ClickBench columns that have a dictionary worth the name that was 744 MB of a 12.2 GB file. Both
4062/// halves of it are nearly empty. The heads are the first eight bytes of the values in sorted order,
4063/// so a block of five hundred and twelve of them spans a tiny slice of the column, and on a column of
4064/// URLs they are all `http://w` and the block holds one distinct head. The codes are positions in a
4065/// dictionary of eighteen million, which is twenty five bits and not thirty two.
4066///
4067/// So a block now writes the smallest head in it, the bits the largest is above that, and the heads
4068/// and the codes packed at the width each needs. A block where every head agrees costs nine bytes
4069/// and the codes.
4070const RANK_BLOCK_HEADER: usize = size_of::<u64>() + 1;
4071
4072impl NativeText {
4073    /// One block of the payload, read and decoded the first time anything asks for a value in it.
4074    ///
4075    /// The bytes handed back are the values of the block laid end to end, which is what the offsets
4076    /// describe, so a caller slices it with the offsets it already has. Where the block sits in the
4077    /// file is the only thing the caller cannot work out for itself, because the stored form is
4078    /// shorter than the decoded one and by a different amount in every block.
4079    fn payload_block(&self, block: usize) -> Result<Option<&[u8]>> {
4080        let Some(slot) = self.blocks.get(block) else { return Ok(None) };
4081        let bytes = slot.get_or_init(|| self.decode_block(block)).as_ref().map_err(Clone::clone)?;
4082        Ok(Some(bytes.as_slice()))
4083    }
4084
4085    /// Reads and decodes one block of the payload, without deciding who keeps it.
4086    ///
4087    /// [`Self::payload_block`] keeps it forever, which is what a point read wants and what a walk
4088    /// of the whole dictionary must not do. Both call this and they differ in nothing else.
4089    fn decode_block(&self, block: usize) -> Result<Vec<u8>> {
4090        let len = self.lengths[block];
4091        let mut stored = vec![
4092            0;
4093            usize::try_from(len).map_err(|_| invalid(
4094                "global dictionary block does not fit in memory"
4095            ))?
4096        ];
4097        read_at(&self.file, self.starts[block], &mut stored)?;
4098        if checksum(&stored) != self.hashes[block] {
4099            return Err(invalid("global dictionary payload checksum differs"));
4100        }
4101        let first = block * TEXT_PAYLOAD_VALUES;
4102        let last = (first + TEXT_PAYLOAD_VALUES).min(self.values);
4103        let want = self.end_within(last - 1)? as usize;
4104        let values = string::decode_flat(&stored)?;
4105        if values.len() != last - first {
4106            return Err(invalid("global dictionary block holds the wrong value count"));
4107        }
4108        let bytes = values.into_bytes();
4109        if bytes.len() != want {
4110            return Err(invalid("global dictionary block decodes to the wrong length"));
4111        }
4112        Ok(bytes)
4113    }
4114
4115    /// How many single offset reads make [`Self::value_ends`] worth building.
4116    ///
4117    /// As many reads as the dictionary has values. Building the table costs about thirty
4118    /// instructions a value once the fresh pages it lands in are counted, and a read out of it saves
4119    /// about thirty five, so it repays itself after roughly one read per value. The reads so far are
4120    /// the only guess there is at the reads to come, and waiting until they match the size of the
4121    /// dictionary is betting that a column read that much will be read that much again.
4122    ///
4123    /// A sixteenth was the first answer, from counting the unpacking alone at three instructions a
4124    /// value. ClickBench 38 showed what that missed: it reads about twenty thousand titles a
4125    /// statement out of a dictionary of three hundred and fifty thousand, crossed a sixteenth in its
4126    /// second statement and was two percent slower for a table it did not read enough to repay. A
4127    /// scan asking for the length of every row crosses it part way through its first statement on
4128    /// ClickBench, where a string column has about two rows for every value, and a filter that keeps
4129    /// a few thousand rows never does. The floor is there
4130    /// because a short dictionary would otherwise build a table for a handful of reads.
4131    fn ends_worth_unpacking(&self) -> usize {
4132        self.values.max(TEXT_PAYLOAD_VALUES)
4133    }
4134
4135    /// The unpacked ends, if they are built or if this read is the one that makes them worth it.
4136    fn value_ends(&self) -> Option<&[u32]> {
4137        if let Some(built) = self.value_ends.get() {
4138            return built.as_deref();
4139        }
4140        if self.ends_asked.fetch_add(1, Atomic::Relaxed) < self.ends_worth_unpacking() {
4141            return None;
4142        }
4143        self.value_ends.get_or_init(|| self.unpack_ends()).as_deref()
4144    }
4145
4146    /// Every end of the column, a run at a time.
4147    ///
4148    /// `None` rather than an error on anything wrong, because this is a cache in front of a reader
4149    /// that answers the same question. A column whose offsets are short or whose ends do not fit in
4150    /// four bytes gets no table and the same error it would have got, from the read that wanted it.
4151    fn unpack_ends(&self) -> Option<Vec<u32>> {
4152        let mut ends = vec![0u32; self.values];
4153        for (run, into) in ends.chunks_mut(TEXT_OFFSET_RUN).enumerate() {
4154            let bytes = self.offsets.get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)?;
4155            bitpack::unpack_tail_into(bytes, self.offset_bits, into, |bits| {
4156                u32::try_from(bits).unwrap_or(u32::MAX)
4157            })
4158            .ok()?;
4159        }
4160        // An end that did not fit was stored as the sentinel, and a real one cannot reach it because
4161        // a payload block is far smaller than four gigabytes. So the column keeps the packed reader.
4162        if ends.contains(&u32::MAX) { None } else { Some(ends) }
4163    }
4164
4165    /// Where the value at `index` ends inside its payload block.
4166    fn end_within(&self, index: usize) -> Result<u32> {
4167        if let Some(ends) = self.value_ends() {
4168            return ends
4169                .get(index)
4170                .copied()
4171                .ok_or_else(|| invalid("global dictionary offsets are short"));
4172        }
4173        let run = index / TEXT_OFFSET_RUN;
4174        let bytes = self
4175            .offsets
4176            .get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)
4177            .ok_or_else(|| invalid("global dictionary offsets are short"))?;
4178        let end = bitpack::tail_at(bytes, self.offset_bits, index % TEXT_OFFSET_RUN)
4179            .map_err(|_| invalid("global dictionary offsets are short"))?;
4180        u32::try_from(end).map_err(|_| invalid("global dictionary offset is past the payload"))
4181    }
4182
4183    /// Where every value in `first..last` ends inside its payload block, in one pass over the runs.
4184    ///
4185    /// [`Self::end_within`] answers for one value and pays for it twice over: it shifts a window to
4186    /// the bit the value starts at, and the copy that fills that window is a length the compiler does
4187    /// not know, so it is a call to `memcpy` rather than a load. A sweep asked for two of those per
4188    /// value, one for the end and one for the start that is the end before it, and on the ClickBench
4189    /// `URL` dictionary of eighteen million that was most of the half second a `LIKE` over it took.
4190    ///
4191    /// [`bitpack::unpack_tail_into`] walks the run instead, which makes the window a fixed width and
4192    /// so an unaligned load, and reads the bit position off a counter. A run is five hundred and
4193    /// twelve values and a block is two of them, so a block of a thousand and twenty four values
4194    /// costs two calls here and nothing per value.
4195    ///
4196    /// The answer is written straight into the result. A run that is wanted from its first value,
4197    /// which is every run but the one the sweep starts in, unpacks into its own window of the result
4198    /// and is never copied. Only a run joined part way through needs the scratch buffer, and there is
4199    /// at most one of those per sweep, so the buffer is allocated the first time one turns up.
4200    fn ends_within(&self, first: usize, last: usize) -> Result<Vec<u64>> {
4201        let mut ends = vec![0u64; last.saturating_sub(first)];
4202        let mut scratch = Vec::new();
4203        let mut at = first;
4204        while at < last {
4205            let run = at / TEXT_OFFSET_RUN;
4206            let stop = ((run + 1) * TEXT_OFFSET_RUN).min(last);
4207            let held = self.values.saturating_sub(run * TEXT_OFFSET_RUN).min(TEXT_OFFSET_RUN);
4208            let bytes = self
4209                .offsets
4210                .get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)
4211                .ok_or_else(|| invalid("global dictionary offsets are short"))?;
4212            let from = at % TEXT_OFFSET_RUN;
4213            let upto = stop - run * TEXT_OFFSET_RUN;
4214            if upto > held || bytes.len() < bitpack::tail_len(held, self.offset_bits) {
4215                return Err(invalid("global dictionary offsets are short"));
4216            }
4217            let into = &mut ends[at - first..stop - first];
4218            if from == 0 {
4219                bitpack::unpack_tail_into(bytes, self.offset_bits, into, |bits| bits)
4220                    .map_err(|_| invalid("global dictionary offsets are short"))?;
4221            } else {
4222                scratch.resize(held, 0);
4223                bitpack::unpack_tail_into(bytes, self.offset_bits, &mut scratch, |bits| bits)
4224                    .map_err(|_| invalid("global dictionary offsets are short"))?;
4225                into.copy_from_slice(&scratch[from..upto]);
4226            }
4227            at = stop;
4228        }
4229        Ok(ends)
4230    }
4231
4232    /// Where the value at `index` starts inside its payload block, which is where the value before
4233    /// it ended unless it is the first of the block.
4234    fn start_within(&self, index: usize) -> Result<u32> {
4235        if index % TEXT_PAYLOAD_VALUES == 0 { Ok(0) } else { self.end_within(index - 1) }
4236    }
4237
4238    /// Where the value at `index` starts and ends inside its payload block.
4239    ///
4240    /// The two offsets sit next to each other in the same run unless the value opens one, and a run
4241    /// of seventeen bit offsets, which is what a block of a thousand strings needs, puts a pair of
4242    /// them inside one eight byte load. So the common case reads the packed bytes once rather than
4243    /// twice and does the bounds arithmetic once. This is asked once per string a text column hands
4244    /// out, and on ClickBench 27 the two reads together were a quarter of the query.
4245    fn span_within(&self, index: usize) -> Result<(u32, u32)> {
4246        if let Some(ends) = self.value_ends() {
4247            let end =
4248                *ends.get(index).ok_or_else(|| invalid("global dictionary offsets are short"))?;
4249            // The value before it in the same block, and zero where there is no value before it.
4250            // `index` is inside the table, so the one under it is too.
4251            let start = if index % TEXT_PAYLOAD_VALUES == 0 { 0 } else { ends[index - 1] };
4252            if start > end {
4253                return Err(invalid("global dictionary value ends before it starts"));
4254            }
4255            return Ok((start, end));
4256        }
4257        let within = index % TEXT_OFFSET_RUN;
4258        let (start, end) = if within == 0 {
4259            (self.start_within(index)?, self.end_within(index)?)
4260        } else {
4261            let run = index / TEXT_OFFSET_RUN;
4262            let bytes = self
4263                .offsets
4264                .get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)
4265                .ok_or_else(|| invalid("global dictionary offsets are short"))?;
4266            let (start, end) = bitpack::tail_pair(bytes, self.offset_bits, within)
4267                .map_err(|_| invalid("global dictionary offsets are short"))?;
4268            let ends = u32::try_from(end)
4269                .map_err(|_| invalid("global dictionary offset is past the payload"))?;
4270            let starts = u32::try_from(start)
4271                .map_err(|_| invalid("global dictionary offset is past the payload"))?;
4272            (starts, ends)
4273        };
4274        if start > end {
4275            return Err(invalid("global dictionary value ends before it starts"));
4276        }
4277        Ok((start, end))
4278    }
4279
4280    /// The block of the sorted order that holds `rank`, and where in it that rank sits.
4281    ///
4282    /// The block is read from the file and checked against the hash the index carries for it the
4283    /// first time anything asks, and kept after that, the same way a payload block is. A search
4284    /// makes about as many probes as the order has bits, so the whole search reads a handful of
4285    /// these and never the rest.
4286    fn rank_parts(&self, rank: usize) -> Result<(&[u8], usize)> {
4287        let slot = self
4288            .rank_blocks
4289            .get(rank / TEXT_RANK_BLOCK)
4290            .ok_or_else(|| invalid("global dictionary rank is past the order"))?;
4291        let block = slot
4292            .get_or_init(|| {
4293                let which = rank / TEXT_RANK_BLOCK;
4294                let start = if which == 0 { 0 } else { self.rank_ends[which - 1] };
4295                let end = self.rank_ends[which];
4296                let mut bytes = vec![0; (end - start) as usize];
4297                read_at(&self.file, self.rank_at + start, &mut bytes)?;
4298                if checksum(&bytes)
4299                    != *self
4300                        .rank_hashes
4301                        .get(rank / TEXT_RANK_BLOCK)
4302                        .ok_or_else(|| invalid("global dictionary rank block has no checksum"))?
4303                {
4304                    return Err(invalid("global dictionary rank checksum differs"));
4305                }
4306                Ok(bytes)
4307            })
4308            .as_ref()
4309            .map_err(Clone::clone)?;
4310        Ok((block.as_slice(), rank % TEXT_RANK_BLOCK))
4311    }
4312
4313    /// The first eight bytes of the value at `rank`, as the integer a comparison reads.
4314    fn head_at(&self, rank: usize) -> Result<u64> {
4315        let (block, within) = self.rank_parts(rank)?;
4316        let (base, width, packed) = rank_heads(block)?;
4317        let above = bitpack::tail_at(packed, width, within)
4318            .map_err(|_| invalid("global dictionary rank block is short of heads"))?;
4319        Ok(base.wrapping_add(above))
4320    }
4321
4322    /// The packed codes of one rank block, which follow the heads on the next byte boundary.
4323    fn rank_codes<'block>(&self, block: &'block [u8], count: usize) -> Result<&'block [u8]> {
4324        let (_, width, packed) = rank_heads(block)?;
4325        packed
4326            .get(bitpack::tail_len(count, width)..)
4327            .ok_or_else(|| invalid("global dictionary rank block is short of codes"))
4328    }
4329
4330    /// How many entries the block holding `rank` has, which is a full block except at the end.
4331    fn rank_block_len(&self, rank: usize) -> usize {
4332        let first = rank / TEXT_RANK_BLOCK * TEXT_RANK_BLOCK;
4333        TEXT_RANK_BLOCK.min(self.ranks - first)
4334    }
4335}
4336
4337/// The base, the width and the packed bytes of one rank block's heads.
4338fn rank_heads(block: &[u8]) -> Result<(u64, usize, &[u8])> {
4339    let header = block
4340        .get(..RANK_BLOCK_HEADER)
4341        .ok_or_else(|| invalid("global dictionary rank block is short"))?;
4342    let base = u64::from_le_bytes(header[..8].try_into().expect("eight bytes"));
4343    let width = header[8] as usize;
4344    if width > 64 {
4345        return Err(invalid("global dictionary rank block packs heads past a word"));
4346    }
4347    Ok((base, width, &block[RANK_BLOCK_HEADER..]))
4348}
4349
4350/// Bits one offset of a dictionary takes, which is what its widest payload block spans.
4351///
4352/// One width for the whole column rather than one a block. A block is 1,024 values of the same
4353/// column, so the blocks of a column are within a factor of two of each other on every ClickBench
4354/// string column, and a width a block would save a fraction of a bit and cost a byte a block plus
4355/// the arithmetic that finds where a block starts.
4356fn offset_width(ends: &[u32]) -> usize {
4357    // The ends are already relative to the block the value is in, so the last end of a block is that
4358    // block's total and the largest end anywhere is the widest block. There is no subtraction left
4359    // to do and no need to walk the blocks to find where one starts.
4360    let span = ends.iter().copied().max().unwrap_or(0);
4361    (u32::BITS - span.leading_zeros()) as usize
4362}
4363
4364/// How many bytes `values` offsets take at `bits`, which is what the reader has to know before it
4365/// has read any of them.
4366fn offset_bytes(values: usize, bits: usize) -> usize {
4367    let full = values / TEXT_OFFSET_RUN;
4368    let rest = values % TEXT_OFFSET_RUN;
4369    full * TEXT_OFFSET_RUN / 8 * bits + bitpack::tail_len(rest, bits)
4370}
4371
4372/// The end of every value within its payload block, packed a run at a time.
4373/// A run never straddles a block, because [`TEXT_OFFSET_RUN`] divides [`TEXT_PAYLOAD_VALUES`], which
4374/// is what lets this be a walk of the ends rather than arithmetic against a per block base.
4375fn encode_offsets(ends: &[u32], bits: usize, out: &mut Vec<u8>) -> Result<()> {
4376    let mut run = Vec::with_capacity(TEXT_OFFSET_RUN);
4377    for chunk in ends.chunks(TEXT_OFFSET_RUN) {
4378        run.clear();
4379        run.extend(chunk.iter().map(|&end| u64::from(end)));
4380        bitpack::pack_tail(&run, bits, out)
4381            .map_err(|_| invalid("global dictionary offsets do not pack"))?;
4382    }
4383    Ok(())
4384}
4385
4386/// How many bits a code of a dictionary of `values` entries takes.
4387fn code_width(values: usize) -> usize {
4388    match u64::try_from(values).unwrap_or(u64::MAX) {
4389        0 | 1 => 0,
4390        last => (u64::BITS - (last - 1).leading_zeros()) as usize,
4391    }
4392}
4393
4394impl TextSource for NativeText {
4395    fn len(&self) -> usize {
4396        self.values
4397    }
4398
4399    fn might_contain(&self, first: usize, literal: &[u8]) -> Result<bool> {
4400        let Some(grams) = &self.grams else { return Ok(true) };
4401        if literal.len() < 4 || first >= self.values {
4402            return Ok(true);
4403        }
4404        let bytes = grams
4405            .loaded
4406            .get_or_init(|| {
4407                let mut bytes = vec![0; grams.length];
4408                read_at(&self.file, grams.start, &mut bytes)?;
4409                if checksum(&bytes) != grams.hash {
4410                    return Err(invalid("global dictionary substring signatures checksum differs"));
4411                }
4412                Ok(bytes)
4413            })
4414            .as_ref()
4415            .map_err(Clone::clone)?;
4416        let block = first / TEXT_PAYLOAD_VALUES;
4417        let Some(bits) = bytes.get(block * TEXT_GRAM_BYTES..(block + 1) * TEXT_GRAM_BYTES) else {
4418            return Ok(true);
4419        };
4420        Ok(literal.windows(4).all(|gram| {
4421            gram_bits(gram).into_iter().all(|bit| bits[bit / 8] & (1 << (bit % 8)) != 0)
4422        }))
4423    }
4424
4425    fn bytes_at(&self, index: usize) -> Result<Option<&[u8]>> {
4426        if index >= self.values {
4427            return Ok(None);
4428        }
4429        let (start, end) = self.span_within(index)?;
4430        if start == end {
4431            return Ok(Some(&[]));
4432        }
4433        // A block holds a fixed number of values rather than a fixed number of bytes, so the value
4434        // is in one block and the offsets already say where in it.
4435        let block = index / TEXT_PAYLOAD_VALUES;
4436        let Some(bytes) = self.payload_block(block)? else { return Ok(None) };
4437        Ok(bytes.get(start as usize..end as usize))
4438    }
4439
4440    fn bytes_len_at(&self, index: usize) -> Result<Option<usize>> {
4441        if index >= self.values {
4442            return Ok(None);
4443        }
4444        let (start, end) = self.span_within(index)?;
4445        Ok(Some((end - start) as usize))
4446    }
4447
4448    /// Every length out of the unpacked ends in one loop, which is the point of having them.
4449    ///
4450    /// The whole run of positions counts towards [`Self::ends_worth_unpacking`] at once, because a
4451    /// caller asking for a vector of lengths has said how many it wants, and a vector of them is
4452    /// usually enough on its own. Until the table is worth building this is the row at a time read,
4453    /// the same as the default.
4454    fn bytes_lens_at(&self, indices: &[u32], into: &mut [i64]) -> Result<()> {
4455        self.ends_asked.fetch_add(indices.len(), Atomic::Relaxed);
4456        let Some(ends) = self.value_ends() else {
4457            for (slot, &index) in into.iter_mut().zip(indices) {
4458                *slot = self
4459                    .bytes_len_at(index as usize)?
4460                    .map_or(0, |len| i64::try_from(len).unwrap_or(i64::MAX));
4461            }
4462            return Ok(());
4463        };
4464        if let Some(lens) = self.value_lens.get_or_init(|| lengths_of(ends)) {
4465            for (slot, &index) in into.iter_mut().zip(indices) {
4466                // Past the end is no value and so no length, which is what a row at a time read
4467                // says.
4468                *slot = lens.get(index as usize).map_or(0, |&len| i64::from(len));
4469            }
4470            return Ok(());
4471        }
4472        for (slot, &index) in into.iter_mut().zip(indices) {
4473            let index = index as usize;
4474            // Past the end is no value and so no length, which is what a row at a time read says.
4475            let Some(&end) = ends.get(index) else {
4476                *slot = 0;
4477                continue;
4478            };
4479            let start = if index % TEXT_PAYLOAD_VALUES == 0 { 0 } else { ends[index - 1] };
4480            if start > end {
4481                return Err(invalid("global dictionary value ends before it starts"));
4482            }
4483            *slot = i64::from(end - start);
4484        }
4485        Ok(())
4486    }
4487
4488    /// The rest of the block holding `first`, decoded into a buffer that may die with the call.
4489    ///
4490    /// A block is the unit this format decodes, so a walk that wants every value is going to decode
4491    /// every block whatever it does. The question is whether it keeps them, and both answers are
4492    /// wrong on their own. [`Self::payload_block`] keeps every block it is asked for, so a reader
4493    /// that walked the whole dictionary through `bytes_at` ended up holding the whole dictionary
4494    /// decoded, 4.2 GB on ClickBench `URL`. Keeping none of them makes the next statement asking
4495    /// the same question decode all of it again, which on the same column at a million rows is a
4496    /// `LIKE` going from 2.7 ms to 16.2 ms.
4497    ///
4498    /// So a sweep keeps what it decodes for the second time while the column is under
4499    /// [`TEXT_KEEP_BUDGET`] and drops it after that. A block already in hand is used where it is there and costs nothing either way.
4500    fn sweep(
4501        &self,
4502        first: usize,
4503        limit: usize,
4504        body: &mut dyn FnMut(usize, &[u8]) -> Result<()>,
4505    ) -> Result<usize> {
4506        let limit = limit.min(self.values);
4507        if first >= limit {
4508            return Ok(first);
4509        }
4510        let block = first / TEXT_PAYLOAD_VALUES;
4511        let last = ((block + 1) * TEXT_PAYLOAD_VALUES).min(limit);
4512        let decoded;
4513        let kept = self.blocks.get(block).and_then(OnceLock::get);
4514        let again = kept.is_none()
4515            && self.swept.get(block).is_some_and(|swept| swept.swap(true, Atomic::Relaxed));
4516        let bytes: &[u8] = match kept {
4517            Some(Ok(kept)) => kept,
4518            _ if again && self.payload_kept.load(Atomic::Relaxed) < self.keep_budget => {
4519                let kept = self
4520                    .payload_block(block)?
4521                    .ok_or_else(|| invalid("global dictionary block is past the payload"))?;
4522                self.payload_kept.fetch_add(kept.len(), Atomic::Relaxed);
4523                kept
4524            }
4525            _ => {
4526                decoded = self.decode_block(block)?;
4527                &decoded
4528            }
4529        };
4530        let ends = self.ends_within(first, last)?;
4531        if ends.len() != last - first {
4532            return Err(invalid("global dictionary offsets are short"));
4533        }
4534        let mut start = u64::from(self.start_within(first)?);
4535        // row at a time: the caller is handed one value after another, and what it does with one is
4536        // its own business, so there is no shape here for anything but a walk.
4537        for (index, &end) in (first..last).zip(&ends) {
4538            let value = usize::try_from(start)
4539                .ok()
4540                .zip(usize::try_from(end).ok())
4541                .and_then(|(from, to)| bytes.get(from..to))
4542                .ok_or_else(|| invalid("global dictionary value is past its block"))?;
4543            body(index, value)?;
4544            start = end;
4545        }
4546        Ok(last)
4547    }
4548
4549    /// Each block the indices land in, decoded once and dropped, or read where it is already kept.
4550    ///
4551    /// Never kept, unlike [`Self::sweep`] under its budget, because a scattered read is a one off:
4552    /// a synopsis turned into values is turned once and remembered by the reader as values, a few
4553    /// kilobytes, where the blocks it went through are megabytes nobody asks for again.
4554    fn visit(
4555        &self,
4556        indices: &[usize],
4557        body: &mut dyn FnMut(usize, &[u8]) -> Result<()>,
4558    ) -> Result<()> {
4559        let mut at = 0;
4560        while at < indices.len() {
4561            let block = indices[at] / TEXT_PAYLOAD_VALUES;
4562            let upto =
4563                at + indices[at..].partition_point(|&index| index / TEXT_PAYLOAD_VALUES == block);
4564            let wanted = &indices[at..upto];
4565            if wanted.iter().any(|&index| index >= self.values) {
4566                return Err(invalid("a visited value is past the global dictionary"));
4567            }
4568            let decoded;
4569            let bytes: &[u8] = match self.blocks.get(block).and_then(OnceLock::get) {
4570                Some(Ok(kept)) => kept,
4571                _ => {
4572                    decoded = self.decode_block(block)?;
4573                    &decoded
4574                }
4575            };
4576            for (offset, &index) in wanted.iter().enumerate() {
4577                let (start, end) = self.span_within(index)?;
4578                let value = bytes
4579                    .get(start as usize..end as usize)
4580                    .ok_or_else(|| invalid("global dictionary value is past its block"))?;
4581                body(at + offset, value)?;
4582            }
4583            at = upto;
4584        }
4585        Ok(())
4586    }
4587
4588    fn ranks(&self) -> Option<usize> {
4589        (self.ranks > 0).then_some(self.ranks)
4590    }
4591
4592    /// The boundary for `wanted`, out of [`Self::searched`] where it is there and put there where
4593    /// it is not.
4594    ///
4595    /// The lock is held over the search rather than dropped and taken again, so that two threads
4596    /// asking for the same value at the same time do the work once between them. That is the shape
4597    /// the scan actually arrives in: sixteen instances of a top N, all reading the same column, all
4598    /// improving their bound over the same early chunks.
4599    fn below(&self, ranks: usize, wanted: &[u8]) -> Result<(usize, bool)> {
4600        let mut memo = self.searched.lock().map_err(|_| invalid("a poisoned dictionary search"))?;
4601        if let Some(&answer) = memo.get(wanted) {
4602            return Ok(answer);
4603        }
4604        let answer = search_below(self, ranks, wanted)?;
4605        if memo.len() >= TEXT_SEARCH_MEMO {
4606            memo.clear();
4607        }
4608        memo.insert(wanted.to_vec(), answer);
4609        Ok(answer)
4610    }
4611
4612    fn compare_rank(&self, rank: usize, wanted: &[u8]) -> Result<Ordering> {
4613        // The head settles the probe unless the two values start with the same eight bytes, and
4614        // only then is a value read. On a column of URLs that is the difference between a search
4615        // that touches one block of the payload and a search that touches nineteen of them.
4616        let settled = self.head_at(rank)?.cmp(&head(wanted));
4617        if settled != Ordering::Equal {
4618            return Ok(settled);
4619        }
4620        let code = self.code_at_rank(rank)?;
4621        let bytes = self
4622            .bytes_at(code as usize)?
4623            .ok_or_else(|| invalid("global dictionary order names a code it does not have"))?;
4624        Ok(bytes.cmp(wanted))
4625    }
4626
4627    fn code_at_rank(&self, rank: usize) -> Result<u32> {
4628        let (block, within) = self.rank_parts(rank)?;
4629        let codes = self.rank_codes(block, self.rank_block_len(rank))?;
4630        let code = bitpack::tail_at(codes, self.code_bits, within)
4631            .map_err(|_| invalid("global dictionary rank block is short of codes"))?;
4632        let code = u32::try_from(code)
4633            .map_err(|_| invalid("global dictionary order names a code it does not have"))?;
4634        if code as usize >= self.len() {
4635            return Err(invalid("global dictionary order names a code it does not have"));
4636        }
4637        Ok(code)
4638    }
4639
4640    fn code_ranks(&self) -> Option<&[u32]> {
4641        // The order is a permutation of the positions, so inverting it needs every position to be
4642        // named exactly once. Anything else and the slice would have holes, and a caller indexing
4643        // it by a code would read a rank that belongs to nothing.
4644        if self.ranks == 0 || self.ranks != self.len() {
4645            return None;
4646        }
4647        self.code_ranks
4648            .get_or_init(|| {
4649                let mut ranks = vec![u32::MAX; self.ranks];
4650                // A block at a time rather than a rank at a time, because reading it per rank pays
4651                // for the bounds check, the division and the lock on every one of them.
4652                for first in (0..self.ranks).step_by(TEXT_RANK_BLOCK) {
4653                    let (block, _) = self.rank_parts(first).ok()?;
4654                    let count = self.rank_block_len(first);
4655                    let codes = self.rank_codes(block, count).ok()?;
4656                    for (within, code) in bitpack::unpack_tail(codes, self.code_bits, count)
4657                        .ok()?
4658                        .into_iter()
4659                        .enumerate()
4660                    {
4661                        let code = usize::try_from(code).ok()?;
4662                        *ranks.get_mut(code)? = u32::try_from(first + within).ok()?;
4663                    }
4664                }
4665                if ranks.contains(&u32::MAX) {
4666                    return None;
4667                }
4668                Some(ranks)
4669            })
4670            .as_deref()
4671    }
4672
4673    fn footprint(&self) -> usize {
4674        self.offsets.capacity()
4675            + self
4676                .value_ends
4677                .get()
4678                .and_then(Option::as_ref)
4679                .map_or(0, |ends| ends.capacity() * size_of::<u32>())
4680            + self
4681                .value_lens
4682                .get()
4683                .and_then(Option::as_ref)
4684                .map_or(0, |lens| lens.capacity() * size_of::<u32>())
4685            + self
4686                .code_ranks
4687                .get()
4688                .and_then(Option::as_ref)
4689                .map_or(0, |ranks| ranks.capacity() * size_of::<u32>())
4690            + self.rank_hashes.capacity() * size_of::<u64>()
4691            + self.rank_ends.capacity() * size_of::<u64>()
4692            + self.rank_blocks.capacity() * size_of::<OnceLock<Result<Vec<u8>>>>()
4693            + self
4694                .rank_blocks
4695                .iter()
4696                .filter_map(OnceLock::get)
4697                .filter_map(|result| result.as_ref().ok())
4698                .map(Vec::capacity)
4699                .sum::<usize>()
4700            + self.blocks.capacity() * size_of::<OnceLock<Result<Vec<u8>>>>()
4701            + self.hashes.capacity() * size_of::<u64>()
4702            + self.starts.capacity() * size_of::<u64>()
4703            + self.lengths.capacity() * size_of::<u64>()
4704            + self
4705                .grams
4706                .as_ref()
4707                .and_then(|grams| grams.loaded.get())
4708                .and_then(|result| result.as_ref().ok())
4709                .map_or(0, Vec::capacity)
4710            + self
4711                .blocks
4712                .iter()
4713                .filter_map(OnceLock::get)
4714                .filter_map(|result| result.as_ref().ok())
4715                .map(Vec::capacity)
4716                .sum::<usize>()
4717    }
4718}
4719
4720/// Every table wide part number in order, with the stripe it belongs to.
4721fn places(table: &Table) -> Result<Vec<Place>> {
4722    let mut places = Vec::with_capacity(table.stripes.len().saturating_mul(STRIPE_PARTS));
4723    for (at, stripe) in table.stripes.iter().enumerate() {
4724        let index = u32::try_from(at).map_err(|_| invalid("too many stripes"))?;
4725        for (part, &rows) in stripe.parts.iter().enumerate() {
4726            places.push(Place {
4727                stripe: index,
4728                part: u32::try_from(part).map_err(|_| invalid("too many parts in a stripe"))?,
4729                rows,
4730            });
4731        }
4732    }
4733    Ok(places)
4734}
4735
4736/// Reads one column's section of a stripe's index page.
4737///
4738/// The section carries its own checksum, so a reader that wants one column out of a hundred and
4739/// five preads a few hundred bytes and still knows that what it got is what was written.
4740fn read_index(file: &File, stripe: &Stripe, column: usize) -> Result<Vec<PartSpan>> {
4741    let parts = stripe.parts.len();
4742    let section = index_section(parts)?;
4743    let at = column.checked_mul(section).ok_or_else(|| invalid("index page offset overflow"))?;
4744    let end = at.checked_add(section).ok_or_else(|| invalid("index page offset overflow"))?;
4745    if end > stripe.index.length as usize {
4746        return Err(invalid("index page is shorter than its columns"));
4747    }
4748    let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
4749    let mut bytes = vec![0; section];
4750    let offset = stripe
4751        .index
4752        .offset
4753        .checked_add(at as u64)
4754        .ok_or_else(|| invalid("index page offset overflow"))?;
4755    read_at(file, offset, &mut bytes)?;
4756    let entries = section - size_of::<u64>();
4757    let stored = u64::from_le_bytes(bytes[entries..].try_into().expect("eight bytes"));
4758    if checksum(&bytes[..entries]) != stored {
4759        // With where it was read from, because the two ways this fires look identical from the
4760        // message alone: a file somebody damaged, and a file we wrote to the wrong offset.
4761        return Err(invalid(&format!(
4762            "index page section checksum differs, column {column} of {parts} parts at {offset}, \
4763             wanted {stored:016x} and got {:016x}",
4764            checksum(&bytes[..entries]),
4765        )));
4766    }
4767    let mut spans = Vec::with_capacity(parts);
4768    let mut start = 0_usize;
4769    for part in 0..parts {
4770        let at = part * INDEX_ENTRY;
4771        let length = u32::from_le_bytes(bytes[at..at + 4].try_into().expect("four bytes")) as usize;
4772        let hash = u64::from_le_bytes(bytes[at + 4..at + 12].try_into().expect("eight bytes"));
4773        spans.push(PartSpan { start, length, hash });
4774        start = start.checked_add(length).ok_or_else(|| invalid("column page length overflow"))?;
4775    }
4776    if start != page.length as usize {
4777        return Err(invalid("column page length differs from its index"));
4778    }
4779    Ok(spans)
4780}
4781
4782/// One part's bytes out of a whole column page.
4783fn part_bytes(page: &[u8], span: PartSpan) -> Result<&[u8]> {
4784    let end = span.start.checked_add(span.length).ok_or_else(|| invalid("part range overflow"))?;
4785    page.get(span.start..end).ok_or_else(|| invalid("part exceeds its column page"))
4786}
4787
4788/// Puts one stripe of one column in the cache, and hands back the page for the pool to count when
4789/// it is a page the column did not already hold.
4790///
4791/// The index goes in its own slot and stays. Only the page is under the budget, and the pool is
4792/// what enforces it, once the caller has let go of the column's lock.
4793fn remember(cached: &mut Cached, held: &CachedColumn) -> Option<(usize, Arc<AtomicBool>)> {
4794    if let Some(slot) = cached.index.get_mut(held.stripe) {
4795        if slot.is_none() {
4796            *slot = Some(Arc::clone(&held.index));
4797        }
4798    }
4799    let page = held.page.clone()?;
4800    let slot = cached.pages.get_mut(held.stripe)?;
4801    if slot.is_some() {
4802        return None;
4803    }
4804    let bytes = page.len();
4805    // Set, so that the page a worker has just paid to read is not the one the pass it pays for
4806    // lets go of before the worker has read a part out of it.
4807    let used = Arc::new(AtomicBool::new(true));
4808    *slot = Some(Resident { page, used: Arc::clone(&used) });
4809    Some((bytes, used))
4810}
4811
4812/// Every table a native file holds, without the directory of any of them.
4813///
4814/// This is what opening a database reads. It is the small level of the directory, so the cost is
4815/// proportional to how many tables there are rather than to how much data they hold, and a session
4816/// that touches two tables of eight decodes two table directories.
4817///
4818/// The file handle is shared with every reader this hands out. Eight tables in one file is one open
4819/// file descriptor, not eight, which is the other thing one file buys over a file per table.
4820#[derive(Debug, Clone)]
4821pub struct Catalog {
4822    file: Arc<File>,
4823    size: u64,
4824    entries: Arc<Vec<Entry>>,
4825    /// The views the file holds, whole, since a view has no second level to read later.
4826    views: Arc<Vec<ViewEntry>>,
4827    opening: Opening,
4828    /// Where every reader this hands out counts its pages.
4829    pool: PagePool,
4830}
4831
4832/// Signed integer sums and non-null counts for selected columns, plus total table rows.
4833#[derive(Debug, Clone, PartialEq, Eq)]
4834pub struct CertifiedSums {
4835    pub columns: Vec<(i128, u64)>,
4836    pub rows: u64,
4837}
4838
4839/// Exact ends of an integer or date column, including a certified all-null column.
4840#[derive(Debug, Clone, Copy, PartialEq, Eq)]
4841pub enum IntegerExtremes {
4842    Null,
4843    Values { low: i128, high: i128 },
4844}
4845
4846/// A complete numeric value-to-row-count synopsis; `None` represents SQL NULL.
4847pub type NumericFrequencies = Vec<(Option<i128>, u64)>;
4848
4849impl Catalog {
4850    /// Reads the highest valid catalog slot and nothing under it.
4851    ///
4852    /// The readers it hands out keep pages in a pool of their own with no budget, so each column
4853    /// holds its floor of four stripes and no more. A database opens with [`Catalog::open_in`].
4854    ///
4855    /// # Errors
4856    ///
4857    /// If the file has no valid committed catalog or a catalog pointer is out of bounds.
4858    pub fn open(path: impl AsRef<Path>) -> Result<Self> {
4859        Self::open_in(path, &PagePool::default())
4860    }
4861
4862    /// The same, with every reader it hands out keeping its pages in `pool`.
4863    ///
4864    /// # Errors
4865    ///
4866    /// If the file has no valid committed catalog or a catalog pointer is out of bounds.
4867    pub fn open_in(path: impl AsRef<Path>, pool: &PagePool) -> Result<Self> {
4868        let (file, size, _, bytes, opening) = slot_bytes(path)?;
4869        let (entries, views) = decode_catalog(&bytes, size)?;
4870        Ok(Self {
4871            file: Arc::new(file),
4872            size,
4873            entries: Arc::new(entries),
4874            views: Arc::new(views),
4875            opening,
4876            pool: pool.clone(),
4877        })
4878    }
4879
4880    /// The tables in the file, in the order they were written.
4881    pub fn names(&self) -> impl ExactSizeIterator<Item = &str> {
4882        self.entries.iter().map(|entry| entry.name.as_str())
4883    }
4884
4885    /// The same tables with how many rows each of them holds.
4886    ///
4887    /// The names alone answer which tables the file has, which is what a checkpoint needs to know.
4888    /// A load asks a second question: whether a table already in the file is really in the way of
4889    /// the one it wants to write. A table with no rows is not, because it has no pages the next
4890    /// generation would have to carry, so the count has to come out of the catalog beside the name.
4891    pub fn rows(&self) -> impl ExactSizeIterator<Item = (&str, usize)> {
4892        self.entries.iter().map(|entry| (entry.name.as_str(), entry.rows))
4893    }
4894
4895    /// The views in the file, in the order they were written.
4896    ///
4897    /// Whole, unlike [`Catalog::names`], which hands back names and makes the caller ask for a table
4898    /// by one. A view is a few strings and a column list and it was all read at open, so there is
4899    /// nothing left to go and fetch and no reason to make the caller ask twice.
4900    pub fn views(&self) -> impl ExactSizeIterator<Item = &ViewEntry> {
4901        self.views.iter()
4902    }
4903
4904    /// How many tables the file holds.
4905    #[must_use]
4906    pub fn len(&self) -> usize {
4907        self.entries.len()
4908    }
4909
4910    /// Whether the file holds no table at all, which is what [`Writer::empty`] writes and what a
4911    /// database somebody dropped the last table out of comes back as.
4912    #[must_use]
4913    pub fn is_empty(&self) -> bool {
4914        self.entries.is_empty()
4915    }
4916
4917    /// Opens one table by name, decoding its directory now.
4918    ///
4919    /// # Errors
4920    ///
4921    /// If there is no table by that name, or its directory is torn or points outside the file.
4922    pub fn table(&self, name: &str) -> Result<Reader> {
4923        let entry = self
4924            .entries
4925            .iter()
4926            .find(|entry| entry.name == name)
4927            .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
4928        // Checked and then decoded a window at a time, so that the directory's own bytes are never
4929        // all in memory beside the table they decode into. It is read twice, and the second read
4930        // comes out of the page cache the first one filled.
4931        let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
4932        if file_checksum(&self.file, offset, length)? != entry.directory.hash {
4933            return Err(invalid(&format!("the directory of table {name} does not checksum")));
4934        }
4935        let mut opening = self.opening;
4936        opening.reads += 1;
4937        opening.bytes += u64::from(entry.directory.length);
4938        Reader::build(
4939            Arc::clone(&self.file),
4940            self.size,
4941            read_directory(Cursor::over(&self.file, offset, length), self.size, Some(offset))?,
4942            u64::from(entry.directory.length),
4943            opening,
4944            self.pool.clone(),
4945        )
4946    }
4947
4948    /// Counts non-null, nonzero values from a validated native directory without building a
4949    /// reader for every stripe. Returns `None` when the bounded frequency synopsis cannot prove
4950    /// the count, so callers can use the ordinary query path.
4951    pub fn nonzero_count(&self, name: &str, column: usize) -> Result<Option<u64>> {
4952        let entry = self
4953            .entries
4954            .iter()
4955            .find(|entry| entry.name == name)
4956            .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
4957        let Some(field) = entry.fields.get(column) else {
4958            return Err(invalid("frequency column index out of range"));
4959        };
4960        if !matches!(
4961            field.ty,
4962            LogicalType::TinyInt
4963                | LogicalType::SmallInt
4964                | LogicalType::Integer
4965                | LogicalType::BigInt
4966                | LogicalType::UTinyInt
4967                | LogicalType::USmallInt
4968                | LogicalType::UInteger
4969                | LogicalType::UBigInt
4970        ) {
4971            return Ok(None);
4972        }
4973        let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
4974        if file_checksum(&self.file, offset, length)? != entry.directory.hash {
4975            return Err(invalid(&format!("the directory of table {name} does not checksum")));
4976        }
4977        if let Some(count) = entry.nonzero.get(column).copied().flatten() {
4978            return Ok(Some(count));
4979        }
4980        quick_nonzero(
4981            Cursor::over(&self.file, offset, length),
4982            &entry.name,
4983            &entry.fields,
4984            entry.rows,
4985            column,
4986        )
4987    }
4988
4989    /// Exact signed-integer sums and non-null counts from the small catalog. The table directory
4990    /// checksum is still checked once before any certificate can answer a query.
4991    pub fn aggregate_sums(&self, name: &str, columns: &[usize]) -> Result<Option<CertifiedSums>> {
4992        let entry = self
4993            .entries
4994            .iter()
4995            .find(|entry| entry.name == name)
4996            .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
4997        let mut sums = Vec::with_capacity(columns.len());
4998        for &column in columns {
4999            let Some(field) = entry.fields.get(column) else {
5000                return Err(invalid("aggregate column index out of range"));
5001            };
5002            if !signed_integer(&field.ty) {
5003                return Ok(None);
5004            }
5005            let Some(sum) = entry.aggregates[column] else {
5006                return Ok(None);
5007            };
5008            sums.push(sum);
5009        }
5010        let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
5011        if file_checksum(&self.file, offset, length)? != entry.directory.hash {
5012            return Err(invalid(&format!("the directory of table {name} does not checksum")));
5013        }
5014        Ok(Some(CertifiedSums { columns: sums, rows: entry.rows as u64 }))
5015    }
5016
5017    /// Exact non-null distinct count from the small catalog, after checking the table directory.
5018    pub fn distinct_count(&self, name: &str, column: usize) -> Result<Option<u64>> {
5019        let entry = self
5020            .entries
5021            .iter()
5022            .find(|entry| entry.name == name)
5023            .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
5024        let Some(count) = entry.distincts.get(column).copied() else {
5025            return Err(invalid("distinct column index out of range"));
5026        };
5027        let Some(count) = count else { return Ok(None) };
5028        let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
5029        if file_checksum(&self.file, offset, length)? != entry.directory.hash {
5030            return Err(invalid(&format!("the directory of table {name} does not checksum")));
5031        }
5032        Ok(Some(count))
5033    }
5034
5035    /// Exact integer or date ends from the small catalog after checking the table directory.
5036    pub fn integer_extremes(&self, name: &str, column: usize) -> Result<Option<IntegerExtremes>> {
5037        let entry = self
5038            .entries
5039            .iter()
5040            .find(|entry| entry.name == name)
5041            .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
5042        let Some(extremes) = entry.extremes.get(column).copied() else {
5043            return Err(invalid("extremes column index out of range"));
5044        };
5045        let Some(extremes) = extremes else { return Ok(None) };
5046        let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
5047        if file_checksum(&self.file, offset, length)? != entry.directory.hash {
5048            return Err(invalid(&format!("the directory of table {name} does not checksum")));
5049        }
5050        Ok(Some(match extremes {
5051            None => IntegerExtremes::Null,
5052            Some((low, high)) => IntegerExtremes::Values { low, high },
5053        }))
5054    }
5055
5056    /// Complete numeric frequencies from the small catalog, after checking the table directory.
5057    pub fn exact_numeric_frequencies(
5058        &self,
5059        name: &str,
5060        column: usize,
5061    ) -> Result<Option<NumericFrequencies>> {
5062        let entry = self
5063            .entries
5064            .iter()
5065            .find(|entry| entry.name == name)
5066            .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
5067        let Some(frequencies) = entry.frequencies.get(column).cloned() else {
5068            return Err(invalid("numeric frequency column index out of range"));
5069        };
5070        let Some(frequencies) = frequencies else { return Ok(None) };
5071        let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
5072        if file_checksum(&self.file, offset, length)? != entry.directory.hash {
5073            return Err(invalid(&format!("the directory of table {name} does not checksum")));
5074        }
5075        Ok(Some(frequencies))
5076    }
5077
5078    /// The schema copied into the small file catalog, available without opening the table directory.
5079    pub fn table_fields(&self, name: &str) -> Option<&[Field]> {
5080        self.entries.iter().find(|entry| entry.name == name).map(|entry| entry.fields.as_slice())
5081    }
5082}
5083
5084/// Where the slot naming `generation` goes, which is the one the generation before it did not use.
5085///
5086/// Generation 1 takes the slot at 16, so a file written once is byte for byte the file this wrote
5087/// before there was a second generation to write.
5088fn slot_offset(generation: u64) -> u64 {
5089    16 + (generation - 1) % 2 * SLOT_BYTES as u64
5090}
5091
5092/// The header and the bytes the highest valid slot points at.
5093///
5094/// Both levels of the directory are reached this way, so the magic check, the version check and the
5095/// choice between the two slots live here rather than being written out twice.
5096fn slot_bytes(path: impl AsRef<Path>) -> Result<(File, u64, Slot, Vec<u8>, Opening)> {
5097    let mut file = File::open(path).map_err(io)?;
5098    let size = file.metadata().map_err(io)?.len();
5099    if size < HEADER {
5100        return Err(invalid("file is shorter than its header"));
5101    }
5102    let mut header = [0; HEADER as usize];
5103    file.read_exact(&mut header).map_err(io)?;
5104    let mut opening = Opening { reads: 1, bytes: HEADER };
5105    let version = u32::from_le_bytes([header[8], header[9], header[10], header[11]]);
5106    // The two halves are worth telling apart. A wrong magic is a file that was never ours and
5107    // the answer is to look at the path. A wrong version is our own file from another build,
5108    // and the number this build wants is the only thing that tells the reader whether to
5109    // rebuild the file or to go back to the binary that wrote it.
5110    if &header[..8] != MAGIC {
5111        return Err(invalid("the header does not begin with a rudb native magic"));
5112    }
5113    if !READABLE.contains(&version) {
5114        return Err(invalid(&format!(
5115            "the file is format {version} and this build reads format {FORMAT}, so it has to \
5116                 be written again"
5117        )));
5118    }
5119    let mut selected = None;
5120    for start in [16, 16 + SLOT_BYTES] {
5121        let slot = Slot::read(&header[start..start + SLOT_BYTES]);
5122        if slot.generation == 0 || slot.length == 0 || slot.length as usize > MAX_DIRECTORY {
5123            continue;
5124        }
5125        let Some(end) = slot.offset.checked_add(u64::from(slot.length)) else { continue };
5126        if slot.offset < HEADER || end > size {
5127            continue;
5128        }
5129        let mut bytes = vec![0; slot.length as usize];
5130        file.seek(SeekFrom::Start(slot.offset)).map_err(io)?;
5131        file.read_exact(&mut bytes).map_err(io)?;
5132        opening.reads += 1;
5133        opening.bytes += u64::from(slot.length);
5134        if checksum(&bytes) == slot.hash
5135            && selected
5136                .as_ref()
5137                .is_none_or(|(old, _): &(Slot, Vec<u8>)| old.generation < slot.generation)
5138        {
5139            selected = Some((slot, bytes));
5140        }
5141    }
5142    let (slot, bytes) = selected.ok_or_else(|| invalid("no committed directory slot is valid"))?;
5143    Ok((file, size, slot, bytes, opening))
5144}
5145
5146impl Reader {
5147    /// Opens a file that holds exactly one table.
5148    ///
5149    /// # Errors
5150    ///
5151    /// If the file has no valid committed directory, a directory pointer is out of bounds, or the
5152    /// file holds more than one table, which is a file that has to be opened by name.
5153    pub fn open(path: impl AsRef<Path>) -> Result<Self> {
5154        let catalog = Catalog::open(path)?;
5155        let mut names = catalog.names();
5156        let name = names.next().ok_or_else(|| invalid("the file holds no table"))?.to_string();
5157        if names.next().is_some() {
5158            return Err(invalid(
5159                "the file holds more than one table, so it has to be opened by name",
5160            ));
5161        }
5162        catalog.table(&name)
5163    }
5164
5165    /// Builds a reader over one decoded table directory.
5166    fn build(
5167        file: Arc<File>,
5168        size: u64,
5169        table: Table,
5170        directory: u64,
5171        opening: Opening,
5172        pool: PagePool,
5173    ) -> Result<Self> {
5174        let places = places(&table)?;
5175        let dictionaries = (0..table.fields.len()).map(|_| OnceLock::new()).collect();
5176        let table_fields = table.fields.len();
5177        let stripes = table.stripes.len();
5178        let columns = (0..table.fields.len())
5179            .map(|_| {
5180                Mutex::new(Cached {
5181                    pages: (0..stripes).map(|_| None).collect(),
5182                    index: (0..stripes).map(|_| None).collect(),
5183                    ..Cached::default()
5184                })
5185            })
5186            .collect::<Vec<_>>();
5187        let cache = Shelf {
5188            columns,
5189            held: (0..table_fields).map(|_| AtomicUsize::new(0)).collect(),
5190            kept: AtomicUsize::new(CACHED_STRIPES_PER_COLUMN),
5191        };
5192        let sieves: Vec<Vec<SieveSlot>> = (0..table.fields.len())
5193            .map(|_| table.stripes.iter().map(|_| OnceLock::new()).collect())
5194            .collect();
5195        let part_ranges: Vec<Vec<RangeSlot>> = (0..table.fields.len())
5196            .map(|_| table.stripes.iter().map(|_| OnceLock::new()).collect())
5197            .collect();
5198        Ok(Self {
5199            file,
5200            table: Arc::new(table),
5201            dictionaries: Arc::new(dictionaries),
5202            loading: Arc::new((0..table_fields).map(|_| Mutex::new(())).collect()),
5203            frequency_values: Arc::new((0..table_fields).map(|_| OnceLock::new()).collect()),
5204            frequency_summaries: Arc::new((0..table_fields).map(|_| OnceLock::new()).collect()),
5205            opened: Arc::new(AtomicUsize::new(0)),
5206            sieves: Arc::new(sieves),
5207            part_ranges: Arc::new(part_ranges),
5208            places: Arc::new(places),
5209            cache: Arc::new(cache),
5210            pool,
5211            pages: Arc::new(AtomicUsize::new(0)),
5212            indexes: Arc::new(AtomicUsize::new(0)),
5213            size,
5214            directory,
5215            opening,
5216        })
5217    }
5218
5219    /// What this reader has read so far, and what opening it cost.
5220    ///
5221    /// Public because the claim of `spec/stats/04-in-memory.md` section 4.2 is about this number
5222    /// and a claim nobody can check is a comment. A caller that wants to know whether opening a
5223    /// file touched the data asks here, and gets an answer that does not depend on what the page
5224    /// cache happened to hold.
5225    #[must_use]
5226    pub fn reads(&self) -> Reads {
5227        Reads {
5228            opening: self.opening,
5229            pages: self.pages.load(Atomic::Relaxed),
5230            indexes: self.indexes.load(Atomic::Relaxed),
5231            dictionaries: self.opened.load(Atomic::Relaxed),
5232        }
5233    }
5234
5235    /// Where the file's bytes went, from the directory alone.
5236    ///
5237    /// No page is read, so this costs the same on a 45 GB table as on an empty one. See [`Layout`]
5238    /// for what is charged where and for why the three things that are not columns stay separate.
5239    #[must_use]
5240    pub fn layout(&self) -> Layout {
5241        let table = &self.table;
5242        let stripes = table.stripes.as_slice();
5243        let columns = table
5244            .fields
5245            .iter()
5246            .enumerate()
5247            .map(|(at, field)| ColumnLayout {
5248                name: field.name.clone(),
5249                kind: field.ty.to_string(),
5250                pages: sum(stripes.iter().map(|stripe| span_bytes(&stripe.pages, at))),
5251                memberships: sum(stripes.iter().map(|stripe| stripe.memberships.bytes(at))),
5252                sieves: sum(stripes.iter().map(|stripe| stripe.sieves.bytes(at))),
5253                part_ranges: sum(stripes.iter().map(|stripe| stripe.part_ranges.bytes(at))),
5254                dictionary: dictionary_bytes(table, at),
5255            })
5256            .collect();
5257        Layout {
5258            file: self.size,
5259            rows: table.rows,
5260            stripes: stripes.len(),
5261            parts: self.places.len(),
5262            columns,
5263            indexes: sum(stripes.iter().map(|stripe| u64::from(stripe.index.length))),
5264            directory: self.directory,
5265            header: HEADER,
5266        }
5267    }
5268
5269    /// What every part of one column is stored as, which is what `pragma_storage_info` reports.
5270    ///
5271    /// Unlike [`Self::layout`] this reads the data, because the encoder's choice is in the page and
5272    /// nowhere else. The directory says how many bytes a column took and says nothing about what
5273    /// shape they are in, and the shape is the question worth asking: the same rows in a different
5274    /// order come back bit packed on one file and plain on another, and that is the difference a
5275    /// clustered load makes to a scan.
5276    ///
5277    /// One read per stripe rather than one per part. A part is a few kilobytes out of a page that
5278    /// is a quarter of a megabyte, so asking part by part would read the same page sixty four
5279    /// times. Nothing is put in the page cache, because a caller asking what a file looks like is
5280    /// not about to scan it and evicting the pages a real query wants would be a poor trade.
5281    ///
5282    /// # Errors
5283    ///
5284    /// If the column is outside the schema, or a page, index section or checksum is invalid.
5285    pub fn stored(&self, column: usize) -> Result<Vec<StoredPart>> {
5286        let field = self
5287            .table
5288            .fields
5289            .get(column)
5290            .ok_or_else(|| invalid("stored column index out of range"))?;
5291        let mut stored = Vec::with_capacity(self.places.len());
5292        let mut row = 0;
5293        for (at, stripe) in self.table.stripes.iter().enumerate() {
5294            let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
5295            let index = read_index(&self.file, stripe, column)?;
5296            let mut bytes = vec![0; page.length as usize];
5297            read_at(&self.file, page.offset, &mut bytes)?;
5298            let ranges = self.stripe_part_ranges(at, column);
5299            for (part, &rows) in stripe.parts.iter().enumerate() {
5300                let span = *index.get(part).ok_or_else(|| invalid("part index out of range"))?;
5301                let held = part_bytes(&bytes, span)?;
5302                let range = ranges.and_then(|held| held.get(part));
5303                stored.push(StoredPart {
5304                    stripe: at,
5305                    part,
5306                    row,
5307                    rows: rows as usize,
5308                    encoding: page_encoding(&field.ty, rows as usize, held),
5309                    bytes: span.length as u64,
5310                    page: page.offset,
5311                    offset: span.start as u64,
5312                    low: range
5313                        .and_then(|range| range.low.clone())
5314                        .and_then(|bound| bound.into_value(&field.ty)),
5315                    high: range
5316                        .and_then(|range| range.high.clone())
5317                        .and_then(|bound| bound.into_value(&field.ty)),
5318                    nulls: range.map(|range| range.nulls),
5319                });
5320                row += rows as usize;
5321            }
5322        }
5323        Ok(stored)
5324    }
5325
5326    /// How many parts the table has, which is how many chunks a scan of it reads.
5327    #[must_use]
5328    pub fn parts(&self) -> usize {
5329        self.places.len()
5330    }
5331
5332    /// The parts of each stripe, in table wide part numbers.
5333    ///
5334    /// A scan that wants one worker to own the page it reads hands work out in these runs. The
5335    /// stripes are contiguous in part numbering and all but the last hold sixty four parts, but a
5336    /// stripe can be flushed early when rows arrive out of order, so the runs are read off the
5337    /// directory rather than worked out from a constant.
5338    #[must_use]
5339    pub fn stripe_parts(&self) -> Vec<std::ops::Range<usize>> {
5340        let mut runs = Vec::with_capacity(self.table.stripes.len());
5341        let mut start = 0;
5342        for stripe in &self.table.stripes {
5343            let end = start + stripe.parts.len();
5344            runs.push(start..end);
5345            start = end;
5346        }
5347        runs
5348    }
5349
5350    /// How many rows one stripe holds, in the numbering [`Self::stripe_parts`] hands back.
5351    ///
5352    /// Off the directory, which is already in memory, rather than by the caller asking for each
5353    /// part in turn through the catalog. Nothing past the end holds any rows.
5354    #[must_use]
5355    pub fn stripe_rows(&self, stripe: usize) -> usize {
5356        self.table.stripes.get(stripe).map_or(0, |held| held.rows)
5357    }
5358
5359    /// Asks the page cache to keep `stripes` stripes of every column instead of the default.
5360    ///
5361    /// This only ever raises the number. A scan that gives each worker a whole stripe has one page
5362    /// per column per worker open at once, and a cache smaller than that is worse than no cache at
5363    /// all: every worker's page is evicted by the others before it has finished its stripe, so it
5364    /// reads a quarter of a megabyte for every part it takes out of it.
5365    pub fn keep_stripes(&self, stripes: usize) {
5366        self.cache.kept.fetch_max(stripes, Atomic::Relaxed);
5367    }
5368
5369    /// Rows in one part, or zero when the part number is past the table.
5370    #[must_use]
5371    pub fn part_rows(&self, at: usize) -> usize {
5372        self.places.get(at).map_or(0, |place| place.rows as usize)
5373    }
5374
5375    /// The committed table directory.
5376    #[must_use]
5377    pub fn table(&self) -> &Table {
5378        &self.table
5379    }
5380
5381    /// Exact leading frequencies when the stored synopsis proves a count-descending prefix.
5382    ///
5383    /// The returned list can be longer than `top`. Keeping the stored tail lets a later TopN apply
5384    /// additional ordering keys without losing a value tied with the requested boundary.
5385    ///
5386    /// # Errors
5387    ///
5388    /// If the column is outside the schema or a stored value does not fit its declared type.
5389    pub fn top_frequencies(&self, column: usize, top: usize) -> Result<Option<Vec<(Value, u64)>>> {
5390        let field = self
5391            .table
5392            .fields
5393            .get(column)
5394            .ok_or_else(|| invalid("frequency column index out of range"))?;
5395        let Some(summary) = self.frequency_summary(column)? else {
5396            return Ok(None);
5397        };
5398        if top == 0 || summary.entries.len() < top {
5399            return Ok(None);
5400        }
5401        let boundary = summary.entries[top - 1].count;
5402        if boundary <= summary.omitted_max {
5403            return Ok(None);
5404        }
5405        self.decode_frequencies(column, &field.ty, &summary.entries).map(Some)
5406    }
5407
5408    /// Exact leading counts for a numeric key paired with a stable-dictionary string key.
5409    ///
5410    /// The stored prefix is returned only when its requested boundary strictly beats the bound on
5411    /// every pair omitted at load time. The returned tail may be longer than `top`, as with
5412    /// [`Self::top_frequencies`], so downstream ordering can settle ties without reading rows.
5413    ///
5414    /// # Errors
5415    ///
5416    /// If either column is outside the schema or persisted pair metadata is inconsistent with the
5417    /// frequency synopsis or dictionary it names.
5418    pub fn top_pair_frequencies(
5419        &self,
5420        first: usize,
5421        second: usize,
5422        top: usize,
5423    ) -> Result<Option<PairFrequencyCounts>> {
5424        if first >= self.table.fields.len() || second >= self.table.fields.len() {
5425            return Err(invalid("pair frequency column index out of range"));
5426        }
5427        let Some(summary) =
5428            self.table.pair_frequencies.iter().find(|summary| {
5429                summary.first as usize == first && summary.second as usize == second
5430            })
5431        else {
5432            return Ok(None);
5433        };
5434        if top == 0 || summary.entries.len() < top {
5435            return Ok(None);
5436        }
5437        let boundary = summary.entries[top - 1].count;
5438        if boundary <= summary.omitted_max {
5439            return Ok(None);
5440        }
5441        let first_summary = self
5442            .frequency_summary(first)?
5443            .ok_or_else(|| invalid("pair frequency first column has no synopsis"))?;
5444        let anchors = self
5445            .decode_frequencies(first, &self.table.fields[first].ty, &first_summary.entries)?
5446            .into_iter()
5447            .map(|(value, _)| value)
5448            .collect::<Vec<_>>();
5449        let dictionary = self
5450            .dictionary(second)?
5451            .ok_or_else(|| invalid("pair frequency second column has no dictionary"))?;
5452        let mut codes = summary.entries.iter().filter_map(|entry| entry.second).collect::<Vec<_>>();
5453        codes.sort_unstable();
5454        codes.dedup();
5455        let texts = dictionary
5456            .try_values_visited(&codes.iter().map(|&code| code as usize).collect::<Vec<_>>())?;
5457        let mut out = Vec::with_capacity(summary.entries.len());
5458        for entry in &summary.entries {
5459            if entry.count < boundary {
5460                break;
5461            }
5462            let first = anchors
5463                .get(entry.first_entry as usize)
5464                .cloned()
5465                .ok_or_else(|| invalid("pair frequency anchor is outside its values"))?;
5466            let second = match entry.second {
5467                None => Value::Null,
5468                Some(code) => {
5469                    let at = codes
5470                        .binary_search(&code)
5471                        .map_err(|_| invalid("pair frequency code was not among the codes read"))?;
5472                    texts[at].clone()
5473                }
5474            };
5475            out.push((vec![first, second], entry.count));
5476        }
5477        Ok(Some(out))
5478    }
5479
5480    /// Every value of one column with the number of rows holding it, when the synopsis is complete.
5481    ///
5482    /// The heavy hitter pass keeps a bounded set of candidates and decrements them all when it runs
5483    /// out of room, so what it usually ends with is the leading values and a bound on everything it
5484    /// dropped. `omitted_max` of zero says that never happened: no candidate was ever decremented and
5485    /// the entries did not overflow the stored budget, so the list is every distinct value of the
5486    /// column with an exact count, and a null counts as a value of its own rather than being skipped.
5487    ///
5488    /// That makes a whole class of question answerable without reading a row. How many rows hold a
5489    /// value, how many do not, and what a `GROUP BY` of that column with a count over it produces are
5490    /// all in here. It is only ever true of a column with few enough distinct values, which is the
5491    /// case worth having, because that is exactly the column a grouping or an equality filter would
5492    /// otherwise walk every row to answer.
5493    ///
5494    /// `None` when the column has no synopsis, or has one that dropped anything.
5495    ///
5496    /// # Errors
5497    ///
5498    /// If the column is outside the schema or a stored value does not fit its declared type.
5499    pub fn exact_frequencies(&self, column: usize) -> Result<Option<Vec<(Value, u64)>>> {
5500        let Some(prefix) = self.frequency_prefix(column)? else {
5501            return Ok(None);
5502        };
5503        Ok((prefix.omitted_max == 0).then_some(prefix.entries))
5504    }
5505
5506    /// Every value the synopsis lists with the number of rows holding it, and a bound on the rest.
5507    ///
5508    /// The counts are exact whether or not the list is complete. The heavy hitter pass keeps a
5509    /// bounded candidate set and then recounts only the candidates that survived it, so a value that
5510    /// made it into the list carries the number of rows that really hold it rather than whatever the
5511    /// pass had left over. What the pass loses is values, not counts.
5512    ///
5513    /// `omitted_max` is how many rows the most common value left out can hold, and zero says nothing
5514    /// was left out at all, which is what [`exact_frequencies`] asks for. Above zero the list is the
5515    /// leading values of the column and everything else is somewhere between no rows and that bound.
5516    ///
5517    /// That prefix is worth reading on its own. A column with a value in half its rows and a long
5518    /// tail behind it has no complete synopsis and never will, and it is the column where dividing
5519    /// the rows by the distinct count is furthest from the truth.
5520    ///
5521    /// `None` when the column has no synopsis.
5522    ///
5523    /// # Errors
5524    ///
5525    /// If the column is outside the schema or a stored value does not fit its declared type.
5526    ///
5527    /// [`exact_frequencies`]: Self::exact_frequencies
5528    pub fn frequency_prefix(&self, column: usize) -> Result<Option<FrequencyPrefix>> {
5529        let field = self
5530            .table
5531            .fields
5532            .get(column)
5533            .ok_or_else(|| invalid("frequency column index out of range"))?;
5534        let Some(summary) = self.frequency_summary(column)? else {
5535            return Ok(None);
5536        };
5537        let entries = self.decode_frequencies(column, &field.ty, &summary.entries)?;
5538        Ok(Some(FrequencyPrefix { entries, omitted_max: summary.omitted_max }))
5539    }
5540
5541    /// One column's synopsis, read back from the file when the directory left it there.
5542    fn frequency_summary(&self, column: usize) -> Result<Option<Cow<'_, FrequencySummary>>> {
5543        Ok(match self.table.frequencies.get(column) {
5544            None | Some(None) => None,
5545            Some(Some(Frequencies::Held(summary))) => Some(Cow::Borrowed(summary)),
5546            Some(Some(Frequencies::Stored { span, values })) => {
5547                let slot = self
5548                    .frequency_summaries
5549                    .get(column)
5550                    .ok_or_else(|| invalid("frequency column index out of range"))?;
5551                if let Some(summary) = slot.get() {
5552                    return Ok(Some(Cow::Borrowed(summary.as_ref())));
5553                }
5554                let field = self
5555                    .table
5556                    .fields
5557                    .get(column)
5558                    .ok_or_else(|| invalid("frequency column index out of range"))?;
5559                let mut bytes = vec![0; span.length as usize];
5560                read_at(&self.file, span.offset, &mut bytes)?;
5561                let summary =
5562                    decode_summary(&mut Cursor::new(&bytes), field, self.table.rows, *values)?;
5563                let summary = summary.ok_or_else(|| invalid("a stored synopsis is missing"))?;
5564                let _ = slot.set(Arc::new(summary));
5565                Some(Cow::Borrowed(slot.get().expect("the decoded summary was stored").as_ref()))
5566            }
5567        })
5568    }
5569
5570    /// Turns stored frequency entries into values of the column's own type.
5571    ///
5572    /// Remembered per column, because the planner asks once for every estimate that touches the
5573    /// column and the executor asks again, and the answer is a few hundred values. The codes of a
5574    /// string column are read through [`Vector::try_values_visited`], which does not keep the blocks
5575    /// it decodes, so what a query answered out of the synopsis holds is those values and not the
5576    /// hundred or so dictionary blocks they are scattered over.
5577    fn decode_frequencies(
5578        &self,
5579        column: usize,
5580        ty: &LogicalType,
5581        entries: &[FrequencyEntry],
5582    ) -> Result<Vec<(Value, u64)>> {
5583        if let Some(values) = self.frequency_values.get(column).and_then(OnceLock::get) {
5584            return Ok(values.as_ref().clone());
5585        }
5586        let values = self.decode_frequencies_once(column, ty, entries)?;
5587        if let Some(slot) = self.frequency_values.get(column) {
5588            let _ = slot.set(Arc::new(values.clone()));
5589        }
5590        Ok(values)
5591    }
5592
5593    fn decode_frequencies_once(
5594        &self,
5595        column: usize,
5596        ty: &LogicalType,
5597        entries: &[FrequencyEntry],
5598    ) -> Result<Vec<(Value, u64)>> {
5599        let stored_texts = self.table.frequency_texts.get(column).filter(|texts| !texts.is_empty());
5600        if stored_texts.is_some_and(|texts| texts.len() != entries.len()) {
5601            return Err(invalid("frequency text count differs from its synopsis"));
5602        }
5603        let dictionary = if *ty == LogicalType::Varchar && stored_texts.is_none() {
5604            self.dictionary(column)?
5605        } else {
5606            None
5607        };
5608        let mut codes = entries
5609            .iter()
5610            .filter_map(|entry| match entry.value {
5611                FrequencyValue::Code(code) => Some(code as usize),
5612                _ => None,
5613            })
5614            .collect::<Vec<_>>();
5615        codes.sort_unstable();
5616        codes.dedup();
5617        let texts = match &dictionary {
5618            Some(dictionary) if !codes.is_empty() => dictionary.try_values_visited(&codes)?,
5619            _ => Vec::new(),
5620        };
5621        let mut out = Vec::with_capacity(entries.len());
5622        for (entry_at, entry) in entries.iter().enumerate() {
5623            let value = match entry.value {
5624                FrequencyValue::Null => {
5625                    if stored_texts.and_then(|texts| texts[entry_at].as_ref()).is_some() {
5626                        return Err(invalid("a null frequency entry has text"));
5627                    }
5628                    Value::Null
5629                }
5630                FrequencyValue::Integer(value) => match *ty {
5631                    LogicalType::TinyInt => Value::TinyInt(
5632                        i8::try_from(value)
5633                            .map_err(|_| invalid("frequency TINYINT is out of range"))?,
5634                    ),
5635                    LogicalType::UTinyInt => Value::UTinyInt(
5636                        u8::try_from(value)
5637                            .map_err(|_| invalid("frequency UTINYINT is out of range"))?,
5638                    ),
5639                    LogicalType::USmallInt => Value::USmallInt(
5640                        u16::try_from(value)
5641                            .map_err(|_| invalid("frequency USMALLINT is out of range"))?,
5642                    ),
5643                    LogicalType::UInteger => Value::UInteger(
5644                        u32::try_from(value)
5645                            .map_err(|_| invalid("frequency UINTEGER is out of range"))?,
5646                    ),
5647                    LogicalType::UBigInt => Value::UBigInt(
5648                        u64::try_from(value)
5649                            .map_err(|_| invalid("frequency UBIGINT is out of range"))?,
5650                    ),
5651                    LogicalType::SmallInt => Value::SmallInt(
5652                        i16::try_from(value)
5653                            .map_err(|_| invalid("frequency SMALLINT is out of range"))?,
5654                    ),
5655                    LogicalType::Integer => Value::Integer(
5656                        i32::try_from(value)
5657                            .map_err(|_| invalid("frequency INTEGER is out of range"))?,
5658                    ),
5659                    LogicalType::BigInt => Value::BigInt(
5660                        i64::try_from(value)
5661                            .map_err(|_| invalid("frequency BIGINT is out of range"))?,
5662                    ),
5663                    LogicalType::Date => Value::Date(
5664                        i32::try_from(value)
5665                            .map_err(|_| invalid("frequency DATE is out of range"))?,
5666                    ),
5667                    LogicalType::Timestamp => Value::Timestamp(
5668                        i64::try_from(value)
5669                            .map_err(|_| invalid("frequency TIMESTAMP is out of range"))?,
5670                    ),
5671                    _ => return Err(invalid("integer frequency belongs to another type")),
5672                },
5673                FrequencyValue::Code(code) => {
5674                    if let Some(text) = stored_texts.and_then(|texts| texts[entry_at].as_ref()) {
5675                        Value::Varchar(
5676                            String::from_utf8(text.clone())
5677                                .map_err(|_| invalid("frequency text is not UTF-8"))?,
5678                        )
5679                    } else {
5680                        if dictionary.is_none() {
5681                            return Err(invalid("frequency code has no dictionary or stored text"));
5682                        }
5683                        let at = codes
5684                            .binary_search(&(code as usize))
5685                            .map_err(|_| invalid("frequency code was not among the codes read"))?;
5686                        texts[at].clone()
5687                    }
5688                }
5689            };
5690            out.push((value, entry.count));
5691        }
5692        Ok(out)
5693    }
5694
5695    /// Sparse rows belonging to the bounded numeric frequency candidate set.
5696    ///
5697    /// The list is omitted when collecting it would exceed the fixed storage budget. A composite
5698    /// aggregate may accept a result over these rows only when its requested boundary is strictly
5699    /// greater than `omitted_max`.
5700    ///
5701    /// # Errors
5702    ///
5703    /// If the column is outside the schema.
5704    pub fn frequency_occurrences(&self, column: usize) -> Result<Option<FrequencyOccurrences>> {
5705        let field = self
5706            .table
5707            .fields
5708            .get(column)
5709            .ok_or_else(|| invalid("frequency column index out of range"))?;
5710        let Some(summary) = self.frequency_summary(column)? else {
5711            return Ok(None);
5712        };
5713        if summary.ordinals.is_empty() {
5714            return Ok(None);
5715        }
5716        let (anchors, anchor_indices) = if summary.ordinal_entries.len() == summary.ordinals.len() {
5717            let entries = self.decode_frequencies(column, &field.ty, &summary.entries)?;
5718            (entries.into_iter().map(|(value, _)| value).collect(), summary.ordinal_entries.clone())
5719        } else {
5720            (Vec::new(), Vec::new())
5721        };
5722        Ok(Some(FrequencyOccurrences {
5723            omitted_max: summary.omitted_max,
5724            ordinals: summary.ordinals.clone(),
5725            anchors,
5726            anchor_indices,
5727        }))
5728    }
5729
5730    /// How many distinct values one column holds, counting a null as no value.
5731    ///
5732    /// A string column of this format is written against one dictionary that covers the whole table.
5733    /// A code is handed out the first time a value is seen and nothing ever removes one, so the
5734    /// number of codes is the number of distinct values exactly rather than an estimate. That makes
5735    /// `COUNT(DISTINCT column)` over a whole table a question the directory already knows the answer
5736    /// to, and the alternative is a hash table with a row per distinct value built from a pass over
5737    /// every row.
5738    ///
5739    /// A null in the column used to make this `None` and no longer does. A null row is written as
5740    /// the code for the empty string, so a nullable column's dictionary can hold an empty string
5741    /// that no row of it actually has, and the dictionary on its own does not say which case it is.
5742    /// The writer does know, because it counts the non-null rows that use each code on its way to
5743    /// the frequency summary, so it records how many codes any row holds and the directory carries
5744    /// that number. This reads it rather than the size of the dictionary, which also means the
5745    /// dictionary page is not opened to answer.
5746    ///
5747    /// An integer column has no dictionary, and its count comes from the set the writer keeps on its
5748    /// numeric frequency pass instead, which is exact up to a cap. `None` for a column past that cap
5749    /// and for every column that is neither, where a sketch would answer approximately and SQL asked
5750    /// for the exact number.
5751    ///
5752    /// # Errors
5753    ///
5754    /// If the column is outside the schema.
5755    pub fn distinct_values(&self, column: usize) -> Result<Option<u64>> {
5756        self.table
5757            .distincts
5758            .get(column)
5759            .copied()
5760            .ok_or_else(|| invalid("distinct column index out of range"))
5761    }
5762
5763    /// How many rows of one column are null, added up over the stripes.
5764    ///
5765    /// Every stripe records this exactly when it is written, because a null count is not a bound
5766    /// that is allowed to be wide the way a minimum and a maximum are: a filter that reads one too
5767    /// many is slow and a `COUNT` that reads one too many is wrong. Adding up a few hundred numbers
5768    /// already in memory is what makes `COUNT(column)` over a whole table free.
5769    ///
5770    /// # Errors
5771    ///
5772    /// If the column is outside the schema.
5773    pub fn null_count(&self, column: usize) -> Result<u64> {
5774        if column >= self.table.fields.len() {
5775            return Err(invalid("null count column index out of range"));
5776        }
5777        let mut nulls = 0_u64;
5778        for stripe in &self.table.stripes {
5779            let range = stripe
5780                .zone
5781                .column(column)
5782                .ok_or_else(|| invalid("stripe zone is narrower than the schema"))?;
5783            nulls = nulls
5784                .checked_add(range.nulls as u64)
5785                .ok_or_else(|| invalid("null count overflow"))?;
5786        }
5787        Ok(nulls)
5788    }
5789
5790    /// The smallest and the largest value of one string column, from the order beside its values.
5791    ///
5792    /// The dictionary holds exactly the values the column holds, so the first and the last of them
5793    /// in sorted order are the column's minimum and maximum. Two reads of a rank block settle what
5794    /// otherwise walks a million rows.
5795    ///
5796    /// `None` when the column is not a string, when the file was written before version 9 and so has
5797    /// no order, when the column has no values at all, or when it has a null in it, which is the
5798    /// placeholder again: the empty string a null is written as would sort ahead of every real
5799    /// value and be reported as the minimum.
5800    ///
5801    /// # Errors
5802    ///
5803    /// If the column is outside the schema, or a rank names a code the dictionary does not have.
5804    pub fn text_extremes(&self, column: usize) -> Result<Option<(Value, Value)>> {
5805        if self.null_count(column)? > 0 {
5806            return Ok(None);
5807        }
5808        let Some(dictionary) = self.dictionary(column)? else { return Ok(None) };
5809        let Some(ranks) = dictionary.ranks() else { return Ok(None) };
5810        if ranks == 0 {
5811            return Ok(None);
5812        }
5813        let low = text_at_rank(&dictionary, 0)?;
5814        let high = text_at_rank(&dictionary, ranks - 1)?;
5815        Ok(Some((low, high)))
5816    }
5817
5818    /// The smallest and the largest value of one column, when every stripe wrote exact ends.
5819    ///
5820    /// A stripe's ends are allowed to be wider than the truth, because a bound that rules out a
5821    /// chunk that could not match is still correct when it rules out nothing. That is what makes
5822    /// them cheap to write for a bit packed or a dictionary column, and it is also what stops them
5823    /// answering a `MIN`. So each stripe says which of the two it wrote, and this answers only when
5824    /// all of them walked their rows.
5825    ///
5826    /// `None` for a column with no ends, for an empty table, and for a column any stripe of which
5827    /// guessed. Nulls need no special case, because the ends skip them the same way `MIN` does.
5828    ///
5829    /// One case is given up on that did not have to be. A stripe merges the ends of its sixty four
5830    /// parts, and a part with no ends at all erases the merged ones, because a part whose rows are
5831    /// not covered by the stripe's ends is a stripe that would skip rows it should keep. A part of
5832    /// nothing but nulls has no rows to cover and so did not need to erase anything, but the merge
5833    /// cannot tell that part from a part whose layout it could not read. So a column with a chunk
5834    /// of nothing but nulls in the middle of it goes and reads the rows. That is slow and right,
5835    /// and the fix is a row count per part rather than anything here.
5836    ///
5837    /// # Errors
5838    ///
5839    /// If the column is outside the schema.
5840    pub fn exact_extremes(&self, column: usize) -> Result<Option<(Bound, Bound)>> {
5841        if column >= self.table.fields.len() {
5842            return Err(invalid("extremes column index out of range"));
5843        }
5844        let mut low: Option<Bound> = None;
5845        let mut high: Option<Bound> = None;
5846        for stripe in &self.table.stripes {
5847            let range = stripe
5848                .zone
5849                .column(column)
5850                .ok_or_else(|| invalid("stripe zone is narrower than the schema"))?;
5851            if !range.exact {
5852                return Ok(None);
5853            }
5854            // A stripe of nothing but nulls has no ends and says nothing about the column's, which
5855            // is why this skips it rather than giving up on the whole column. A stripe that has
5856            // rows and still has no end is a layout whose values this cannot see, and skipping that
5857            // one would answer with an end taken from the other stripes, so it gives up instead.
5858            let (Some(small), Some(large)) = (range.low.as_ref(), range.high.as_ref()) else {
5859                if stripe.rows > range.nulls {
5860                    return Ok(None);
5861                }
5862                continue;
5863            };
5864            low = Some(low.map_or_else(|| small.clone(), |held| held.smaller(small.clone())));
5865            high = Some(high.map_or_else(|| large.clone(), |held| held.larger(large.clone())));
5866        }
5867        Ok(low.zip(high))
5868    }
5869
5870    /// The sum of one integer column and how many rows went into it, when every stripe wrote one.
5871    ///
5872    /// The count beside the sum is the non-null rows, because that is what a `SUM` adds up and what
5873    /// an `AVG` divides by, and a caller that had to work it out from the row count and the null
5874    /// count would be doing the same walk twice.
5875    ///
5876    /// `None` for anything that is not an integer column, for a file written by something that did
5877    /// not record it, and when adding the stripes together would overflow.
5878    ///
5879    /// # Errors
5880    ///
5881    /// If the column is outside the schema.
5882    pub fn exact_sum(&self, column: usize) -> Result<Option<(i128, u64)>> {
5883        if column >= self.table.fields.len() {
5884            return Err(invalid("sum column index out of range"));
5885        }
5886        let mut total = 0_i128;
5887        let mut rows = 0_u64;
5888        for stripe in &self.table.stripes {
5889            let range = stripe
5890                .zone
5891                .column(column)
5892                .ok_or_else(|| invalid("stripe zone is narrower than the schema"))?;
5893            let Some(part) = range.sum else { return Ok(None) };
5894            let Some(sum) = total.checked_add(part) else { return Ok(None) };
5895            total = sum;
5896            rows = rows.saturating_add(stripe.rows as u64 - range.nulls as u64);
5897        }
5898        Ok(Some((total, rows)))
5899    }
5900
5901    /// Certified host groups over a string column, when the caller's inclusive row-count bound
5902    /// excludes every host the synopsis omitted.
5903    pub fn host_groups(
5904        &self,
5905        column: usize,
5906        minimum_count: u64,
5907    ) -> Result<Option<Vec<host::HostEntry>>> {
5908        if column >= self.table.fields.len() {
5909            return Err(invalid("host group column index out of range"));
5910        }
5911        let Some(summary) = &self.table.host_groups else { return Ok(None) };
5912        if summary.column != column || minimum_count <= summary.omitted_max {
5913            return Ok(None);
5914        }
5915        Ok(Some(summary.entries.clone()))
5916    }
5917
5918    /// The global dictionary of a column, opened once however many workers ask for it at once.
5919    ///
5920    /// The unlocked look is first because it is the answer every time after the first and it costs a
5921    /// load. Everybody who misses it queues on [`Self::loading`] and looks again on the way in, so
5922    /// the one who arrived first does the reading and the rest take what it left. Waiting is the
5923    /// cheaper thing to do: the work behind the lock is a page read, a checksum and the decode of a
5924    /// dictionary that can hold half a million entries, and the alternative is every worker of the
5925    /// scan doing all of it and all but one dropping the result on the floor.
5926    fn dictionary(&self, column: usize) -> Result<Option<Arc<Vector>>> {
5927        let Some(page) = self.table.dictionaries[column] else { return Ok(None) };
5928        if let Some(dictionary) = self.dictionaries[column].get() {
5929            return Ok(Some(Arc::clone(dictionary)));
5930        }
5931        let _queued = self.loading[column].lock().map_err(|_| invalid("a poisoned dictionary"))?;
5932        if let Some(dictionary) = self.dictionaries[column].get() {
5933            return Ok(Some(Arc::clone(dictionary)));
5934        }
5935        self.opened.fetch_add(1, Atomic::Relaxed);
5936        let dictionary = Arc::new(open_global_dictionary(
5937            Arc::clone(&self.file),
5938            page,
5939            &self.table.fields[column].ty,
5940            TEXT_KEEP_BUDGET,
5941        )?);
5942        let _ = self.dictionaries[column].set(Arc::clone(&dictionary));
5943        Ok(Some(dictionary))
5944    }
5945
5946    /// Reads one section's extent table and checks it against the entry that names it.
5947    ///
5948    /// # Errors
5949    ///
5950    /// If the entry points outside the file, the table does not checksum, or it does not decode as
5951    /// a run of extents in element order.
5952    pub fn extents(&self, of: &Section) -> Result<Vec<section::Extent>> {
5953        if of.extent_bytes == 0 {
5954            return Ok(Vec::new());
5955        }
5956        let mut bytes = vec![0; of.extent_bytes as usize];
5957        read_at(&self.file, of.extent_page, &mut bytes)?;
5958        if checksum(&bytes) != of.hash {
5959            return Err(invalid("a section's extent table does not checksum"));
5960        }
5961        let extents = section::decode_extents(&bytes)?;
5962        if extents.len() != of.extents as usize {
5963            return Err(invalid("a section's extent table is not the length the entry says"));
5964        }
5965        Ok(extents)
5966    }
5967
5968    /// Reads and verifies one extent of a section.
5969    ///
5970    /// This is what section 3.2's second rule is for. A reduction that needs one extent of a two
5971    /// gigabyte forward link reads and checksums that extent and nothing else, which is the whole
5972    /// difference between a structure that works at SF100 and issue #745.
5973    ///
5974    /// # Errors
5975    ///
5976    /// If the extent points outside the file, or its bytes do not checksum.
5977    pub fn extent(&self, of: &section::Extent) -> Result<Vec<u8>> {
5978        let end = of
5979            .offset
5980            .checked_add(u64::from(of.length))
5981            .ok_or_else(|| invalid("an extent overflows the file"))?;
5982        if of.offset < HEADER || end > self.size {
5983            return Err(invalid("an extent is outside the file"));
5984        }
5985        let mut bytes = vec![0; of.length as usize];
5986        read_at(&self.file, of.offset, &mut bytes)?;
5987        if checksum(&bytes) != of.hash {
5988            return Err(invalid("an extent does not checksum"));
5989        }
5990        Ok(bytes)
5991    }
5992
5993    /// Reads a whole section's payload, every extent of it, in order.
5994    ///
5995    /// For a structure that is resident anyway, which a key map is. Anything large enough that the
5996    /// split matters should be walking [`Reader::extents`] and taking the one it needs.
5997    ///
5998    /// # Errors
5999    ///
6000    /// If the extent table or any extent fails its check.
6001    pub fn payload(&self, of: &Section) -> Result<Vec<u8>> {
6002        let extents = self.extents(of)?;
6003        let mut bytes =
6004            Vec::with_capacity(sum(extents.iter().map(|one| u64::from(one.length))) as usize);
6005        for one in &extents {
6006            if one.first != bytes.len() as u64 {
6007                return Err(invalid("a section's extents do not join up"));
6008            }
6009            bytes.extend_from_slice(&self.extent(one)?);
6010        }
6011        // The same exception `write_section` makes: a budget record has no bytes, so its
6012        // `header_bytes` is a size rather than a header and there is nothing for it to run past.
6013        if !bytes.is_empty() && of.header_bytes as usize > bytes.len() {
6014            return Err(invalid("a section's header is longer than its payload"));
6015        }
6016        Ok(bytes)
6017    }
6018
6019    /// Reads only the named columns from one part.
6020    ///
6021    /// The whole stripe page each column lives in is read and kept, because a scan asks for the
6022    /// parts of a stripe one after another and this is what turns sixty four reads into one.
6023    ///
6024    /// # Errors
6025    ///
6026    /// If a part, column, page, or checksum is invalid.
6027    pub fn read(&self, part: usize, columns: &[usize]) -> Result<Chunk> {
6028        self.read_impl(part, columns, true)
6029    }
6030
6031    /// Reads named columns from one part without keeping the stripe page it came out of.
6032    ///
6033    /// This is for sparse row fetches after a selective TopN or filter, which reach a few parts of
6034    /// a stripe rather than all of them. A caller that will read most of a stripe should use
6035    /// [`Self::read`] instead, because this reads and discards the page index every time.
6036    ///
6037    /// # Errors
6038    ///
6039    /// If a part, column, page, or checksum is invalid.
6040    pub fn read_sparse(&self, part: usize, columns: &[usize]) -> Result<Chunk> {
6041        self.read_impl(part, columns, false)
6042    }
6043
6044    /// Whether an exact global-code membership index proves that the stripe holding a part cannot
6045    /// contain any of the sorted candidate codes.
6046    ///
6047    /// # Errors
6048    ///
6049    /// If the part, column, index page, checksum, or delta stream is invalid.
6050    pub fn skips_codes(&self, part: usize, column: usize, candidates: &[u32]) -> Result<bool> {
6051        if candidates.is_empty() {
6052            return Ok(true);
6053        }
6054        if candidates.windows(2).any(|pair| pair[0] >= pair[1]) {
6055            return Err(Error::internal("native code candidates are not sorted and unique"));
6056        }
6057        let stripe = self.stripe_of(part)?;
6058        let Some(page) = stripe.memberships.get(column) else {
6059            return Ok(false);
6060        };
6061        let mut bytes = vec![0; page.length as usize];
6062        read_at(&self.file, page.offset, &mut bytes)?;
6063        if checksum(&bytes) != page.hash {
6064            return Err(invalid("membership page checksum differs"));
6065        }
6066        let codes = decode_membership(&bytes)?;
6067        let mut left = 0;
6068        let mut right = 0;
6069        while left < codes.len() && right < candidates.len() {
6070            match codes[left].cmp(&candidates[right]) {
6071                Ordering::Less => left += 1,
6072                Ordering::Greater => right += 1,
6073                Ordering::Equal => return Ok(false),
6074            }
6075        }
6076        Ok(true)
6077    }
6078
6079    fn stripe_of(&self, part: usize) -> Result<&Stripe> {
6080        let place = self.places.get(part).ok_or_else(|| invalid("part index out of range"))?;
6081        self.table
6082            .stripes
6083            .get(place.stripe as usize)
6084            .ok_or_else(|| invalid("stripe index out of range"))
6085    }
6086
6087    /// The page index of one column of one stripe, and its page when the caller wants all of it.
6088    ///
6089    /// A scan hands parts out in order, so every worker on a column crosses into a new stripe within
6090    /// a few parts of the others and they all want the same page at the same moment. This used to
6091    /// let all of them read it, which cost the scan as many copies of every page as it had workers.
6092    /// On the full ClickBench file a `MIN(EventDate), MAX(EventDate)` moved 3.2 GB off the disk to
6093    /// look at 400 MB of column.
6094    ///
6095    /// A worker that finds the page it wants already being read neither waits for it nor reads it
6096    /// again. It comes back with the index alone, which sends [`Reader::read_impl`] down the path
6097    /// that reads the one part it came for, a few kilobytes against a quarter of a megabyte, and it
6098    /// picks the page up from the cache on its next part. Waiting would be the other way to avoid
6099    /// the duplicate read and it is worse: the pages that matter are the wide string ones, they take
6100    /// milliseconds to copy even warm, and every other worker would be stopped for all of it.
6101    ///
6102    /// The file is never read under the lock.
6103    fn held(&self, at: usize, stripe: &Stripe, column: usize, whole: bool) -> Result<CachedColumn> {
6104        let cache =
6105            self.cache.columns.get(column).ok_or_else(|| invalid("column index out of range"))?;
6106        let mut cached = cache.lock().map_err(|_| invalid("column page cache is poisoned"))?;
6107        let known = cached.index.get(at).and_then(Clone::clone);
6108        let page = cached.pages.get(at).and_then(Option::as_ref).map(|slot| {
6109            slot.used.store(true, Atomic::Relaxed);
6110            Arc::clone(&slot.page)
6111        });
6112        if let Some(index) = known.clone() {
6113            if !whole || page.is_some() {
6114                return Ok(CachedColumn { stripe: at, index, page });
6115            }
6116        }
6117        if cached.loading.contains(&at) {
6118            drop(cached);
6119            // The index is almost always already here, because somebody read this stripe to get
6120            // into the loading list in the first place, so this branch usually costs no read at
6121            // all and the one part read in `read_impl` is all the losing worker pays for.
6122            if let Some(index) = known {
6123                return Ok(CachedColumn { stripe: at, index, page: None });
6124            }
6125            let held = self.page_of(stripe, column, at, false, None)?;
6126            let mut cached = cache.lock().map_err(|_| invalid("column page cache is poisoned"))?;
6127            remember(&mut cached, &held);
6128            return Ok(held);
6129        }
6130        cached.loading.push(at);
6131        drop(cached);
6132
6133        let read = self.page_of(stripe, column, at, whole, known);
6134
6135        // The stripe leaves the loading list and its page enters the cache under one lock. Doing
6136        // them separately would leave a moment where another worker sees neither and reads the
6137        // page a second time, which is the whole thing this is here to stop.
6138        let mut cached = cache.lock().map_err(|_| invalid("column page cache is poisoned"))?;
6139        if let Some(position) = cached.loading.iter().position(|loading| *loading == at) {
6140            cached.loading.remove(position);
6141        }
6142        let held = read?;
6143        let taken = remember(&mut cached, &held);
6144        drop(cached);
6145        if let Some((bytes, used)) = taken {
6146            self.cache.held[column].fetch_add(1, Atomic::Relaxed);
6147            self.pool.admit(Held {
6148                shelf: Arc::downgrade(&self.cache),
6149                column,
6150                stripe: at,
6151                bytes,
6152                used,
6153            });
6154        }
6155        Ok(held)
6156    }
6157
6158    /// Reads one stripe's index for a column, and its page when the caller wants all of it.
6159    ///
6160    /// `known` is the index when the reader has already read it, which after the first worker
6161    /// through a stripe it always has, because [`remember`] keeps every index for the life of the
6162    /// reader. Without that a scan reads the index again on every part that misses the page cache.
6163    fn page_of(
6164        &self,
6165        stripe: &Stripe,
6166        column: usize,
6167        at: usize,
6168        whole: bool,
6169        known: Option<Arc<Vec<PartSpan>>>,
6170    ) -> Result<CachedColumn> {
6171        let index = match known {
6172            Some(index) => index,
6173            None => {
6174                self.indexes.fetch_add(1, Atomic::Relaxed);
6175                Arc::new(read_index(&self.file, stripe, column)?)
6176            }
6177        };
6178        let page = if whole {
6179            self.pages.fetch_add(1, Atomic::Relaxed);
6180            let span = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
6181            let mut bytes = vec![0; span.length as usize];
6182            read_at(&self.file, span.offset, &mut bytes)?;
6183            Some(Arc::new(bytes))
6184        } else {
6185            None
6186        };
6187        Ok(CachedColumn { stripe: at, index, page })
6188    }
6189
6190    fn read_impl(&self, at: usize, columns: &[usize], whole: bool) -> Result<Chunk> {
6191        let place = *self.places.get(at).ok_or_else(|| invalid("part index out of range"))?;
6192        let index = place.stripe as usize;
6193        let stripe =
6194            self.table.stripes.get(index).ok_or_else(|| invalid("stripe index out of range"))?;
6195        let rows = place.rows as usize;
6196        let mut picked = Vec::with_capacity(columns.len());
6197        for &column in columns {
6198            let field = self
6199                .table
6200                .fields
6201                .get(column)
6202                .ok_or_else(|| invalid("column index out of range"))?;
6203            let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
6204            let held = self.held(index, stripe, column, whole)?;
6205            let span = *held
6206                .index
6207                .get(place.part as usize)
6208                .ok_or_else(|| invalid("part index out of range"))?;
6209            let owned;
6210            let bytes = match &held.page {
6211                Some(held) => part_bytes(held, span)?,
6212                None => {
6213                    let offset = page
6214                        .offset
6215                        .checked_add(span.start as u64)
6216                        .ok_or_else(|| invalid("part range overflow"))?;
6217                    let mut bytes = vec![0; span.length];
6218                    read_at(&self.file, offset, &mut bytes)?;
6219                    owned = bytes;
6220                    &owned
6221                }
6222            };
6223            if checksum(bytes) != span.hash {
6224                return Err(invalid(&format!(
6225                    "column page checksum differs, column {column} part {} at {}+{} of {} bytes, \
6226                     wanted {:016x} and got {:016x}",
6227                    place.part,
6228                    page.offset,
6229                    span.start,
6230                    span.length,
6231                    span.hash,
6232                    checksum(bytes),
6233                )));
6234            }
6235            let dictionary = self.dictionary(column)?;
6236            // Held as a page, because a column that came out of a file is handed out more than
6237            // once. A group by clones its key columns out of the chunk so the keys outlive it, a
6238            // projection of a bare column name does the same, and a cut of a flat run copies unless
6239            // the run is a page. One `Arc` per column per part buys all of those, and it moves the
6240            // run into the `Arc` without touching a value.
6241            picked.push(decode(&field.ty, rows, bytes, dictionary)?.into_pages());
6242        }
6243        Chunk::with_rows(picked, rows)
6244    }
6245
6246    /// Whether persisted statistics prove that a part cannot match the predicates.
6247    ///
6248    /// Three of them, asked cheapest first.
6249    ///
6250    /// The stripe's bounds are in memory already, so they are free, and they are also the coarsest:
6251    /// every part of a stripe gets the same answer and a scan that skips one part that way skips all
6252    /// sixty four. Then the part's own bounds, which are a read of one page per column per stripe
6253    /// and are sixty four times finer. Then the sieves, which are per part and answer equality, the
6254    /// test bounds are worst at: a column of identifiers has every stripe and nearly every part
6255    /// covering the whole of its type, so bounds keep them all and the sieve keeps the ones that
6256    /// really hold the value.
6257    ///
6258    /// The middle one is what an ordered comparison on a column the rows are not sorted by needs. On
6259    /// ClickBench 24 the stripe bounds leave eight stripes of sixteen alive, which is half the file,
6260    /// and the part bounds leave thirty parts of nine hundred and seventy four.
6261    #[must_use]
6262    pub fn skips(&self, part: usize, probes: &[Probe]) -> bool {
6263        let Some(place) = self.places.get(part).copied() else { return false };
6264        let Some(stripe) = self.table.stripes.get(place.stripe as usize) else { return false };
6265        if stripe.zone.skips(probes) {
6266            return true;
6267        }
6268        probes.iter().any(|probe| self.outside(place, probe) || self.sifted(place, probe))
6269    }
6270
6271    /// Whether the bounds of one part rule out one probe.
6272    ///
6273    /// The part's own two ends, which are narrower than the stripe's and cost a page read the first
6274    /// time this is asked about a column. A column with no page here answers `false`, which is the
6275    /// answer a caller got before there were any.
6276    fn outside(&self, place: Place, probe: &Probe) -> bool {
6277        match self.stripe_part_ranges(place.stripe as usize, probe.column) {
6278            Some(ranges) => ranges
6279                .get(place.part as usize)
6280                .is_some_and(|range| range.excludes(probe.op, &probe.value)),
6281            None => false,
6282        }
6283    }
6284
6285    /// The per part ranges of one stripe of one column, read once and kept.
6286    ///
6287    /// `None` when the column has no page in that stripe and when the page is damaged, on the same
6288    /// reasoning as the sieves: this is an index over data that is still there, so a caller that
6289    /// cannot read one reads the rows and gets the right answer slowly.
6290    fn stripe_part_ranges(&self, stripe: usize, column: usize) -> Option<&[Range]> {
6291        let slot = self.part_ranges.get(column)?.get(stripe)?;
6292        if let Some(held) = slot.get() {
6293            return Some(held);
6294        }
6295        let page = self.table.stripes.get(stripe)?.part_ranges.get(column)?;
6296        let mut bytes = vec![0; page.length as usize];
6297        read_at(&self.file, page.offset, &mut bytes).ok()?;
6298        if checksum(&bytes) != page.hash {
6299            return None;
6300        }
6301        let ranges = Arc::new(decode_part_ranges(&bytes).ok()?);
6302        let _ = slot.set(ranges);
6303        slot.get().map(|held| held.as_slice())
6304    }
6305
6306    /// Whether persisted statistics prove that every row of a part matches the predicates.
6307    ///
6308    /// Only the bounds. The sieves say nothing here, because a sieve that holds a value is a sieve
6309    /// that may be holding somebody else's hash, so it can rule a part out and can never wave one
6310    /// through.
6311    ///
6312    /// The stripe first and the part after it, the same two steps and in the same order as
6313    /// [`Self::skips`]. The stripe's bounds are in memory already and its null count covers sixty
6314    /// four parts rather than one, so a stripe that answers is an answer for nothing, and the part's
6315    /// own bounds are only read for the probes it could not settle. Both directions are safe: a
6316    /// stretch where everything passes contains no narrower stretch where something fails, and a
6317    /// stripe with no nulls has no nulls in any of its parts.
6318    ///
6319    /// A string end a part recorded is cut down to its first few bytes, so a part's stretch can be
6320    /// wider than its rows really are as well. That is the same safe direction for the same reason,
6321    /// and it is why this asks the two ends rather than anything `exact` says.
6322    #[must_use]
6323    pub fn certain(&self, part: usize, probes: &[Probe]) -> bool {
6324        let Some(place) = self.places.get(part).copied() else { return false };
6325        let Some(stripe) = self.table.stripes.get(place.stripe as usize) else { return false };
6326        if stripe.zone.certain(probes) {
6327            return true;
6328        }
6329        probes
6330            .iter()
6331            .all(|probe| stripe.zone.certain(slice::from_ref(probe)) || self.inside(place, probe))
6332    }
6333
6334    /// Whether one part's own two ends prove that every row of it passes `probe`.
6335    ///
6336    /// The mirror of [`Self::outside`], reading the same page. `false` for a part whose stripe wrote
6337    /// no range page, which is a stripe of one part, because there the stripe's own bounds are the
6338    /// part's and the caller has already asked them.
6339    fn inside(&self, place: Place, probe: &Probe) -> bool {
6340        match self.stripe_part_ranges(place.stripe as usize, probe.column) {
6341            Some(ranges) => ranges
6342                .get(place.part as usize)
6343                .is_some_and(|range| range.certain(probe.op, &probe.value)),
6344            None => false,
6345        }
6346    }
6347
6348    /// Whether the bounds of one stripe prove that none of its parts can match the predicates.
6349    ///
6350    /// The cheap half of [`Self::skips`], asked about a whole stripe at once. The bounds live in the
6351    /// directory and are already in memory, so this answers without touching the file, and that is
6352    /// the reason it is worth having on its own: a caller that wants to know roughly where the work
6353    /// is before it starts any workers can ask this about sixteen stripes for nothing, where asking
6354    /// [`Self::skips`] about nine hundred parts would read and decode a sieve page per stripe first.
6355    ///
6356    /// It keeps stripes that [`Self::skips`] would rule out part by part, which is the right way for
6357    /// it to be wrong: the parts are still checked when they are read.
6358    #[must_use]
6359    pub fn stripe_skips(&self, stripe: usize, probes: &[Probe]) -> bool {
6360        self.table.stripes.get(stripe).is_some_and(|held| held.zone.skips(probes))
6361    }
6362
6363    /// Whether the sieve of one part rules out one probe.
6364    ///
6365    /// Only equality. An ordered comparison is what the bounds are for and a sieve says nothing
6366    /// about it, and a read that cannot answer keeps the part, which is the answer a caller with no
6367    /// sieve gets anyway.
6368    fn sifted(&self, place: Place, probe: &Probe) -> bool {
6369        if probe.op != Op::Equal {
6370            return false;
6371        }
6372        match self.stripe_sieves(place.stripe as usize, probe.column) {
6373            Some(sieves) => sieves
6374                .get(place.part as usize)
6375                .and_then(Option::as_ref)
6376                .is_some_and(|sieve| sieve.excludes(&probe.value)),
6377            None => false,
6378        }
6379    }
6380
6381    /// The sieves of one stripe of one column, read once and kept.
6382    ///
6383    /// `None` when the column has no sieves in that stripe, when the page is damaged, and when the
6384    /// bytes are not a page this version can read. A sieve is an index over data that is still there
6385    /// and a caller that cannot read one reads the rows, so this is the one place in the file where
6386    /// a bad checksum is a slow query rather than an error.
6387    fn stripe_sieves(&self, stripe: usize, column: usize) -> Option<&[Option<Sieve>]> {
6388        let slot = self.sieves.get(column)?.get(stripe)?;
6389        if let Some(held) = slot.get() {
6390            return Some(held);
6391        }
6392        let page = self.table.stripes.get(stripe)?.sieves.get(column)?;
6393        let mut bytes = vec![0; page.length as usize];
6394        read_at(&self.file, page.offset, &mut bytes).ok()?;
6395        if checksum(&bytes) != page.hash {
6396            return None;
6397        }
6398        let sieves = Arc::new(decode_sieves(&bytes).ok()?);
6399        let _ = slot.set(sieves);
6400        slot.get().map(|held| held.as_slice())
6401    }
6402}
6403
6404/// The value sitting at one position of a dictionary's sorted order.
6405fn text_at_rank(dictionary: &Vector, rank: usize) -> Result<Value> {
6406    let code = dictionary.code_at_rank(rank)? as usize;
6407    let text = dictionary
6408        .try_text_at(code)?
6409        .ok_or_else(|| invalid("global dictionary order names a code it does not have"))?;
6410    Ok(Value::Varchar(text.into()))
6411}
6412
6413/// Writes one span of a file at an offset, without depending on where the cursor is.
6414///
6415/// The writer owns an offset of its own and passes it in here, so that nothing it writes depends on
6416/// a cursor that a read is entitled to move. Both of these can come back short and both loop.
6417#[cfg(unix)]
6418fn write_at(file: &File, mut offset: u64, mut bytes: &[u8]) -> Result<()> {
6419    use std::os::unix::fs::FileExt;
6420    while !bytes.is_empty() {
6421        let written = file.write_at(bytes, offset).map_err(io)?;
6422        if written == 0 {
6423            return Err(invalid("a write to the native file wrote nothing"));
6424        }
6425        offset += written as u64;
6426        bytes = &bytes[written..];
6427    }
6428    Ok(())
6429}
6430
6431/// The same write, on the call Windows spells differently.
6432#[cfg(windows)]
6433fn write_at(file: &File, mut offset: u64, mut bytes: &[u8]) -> Result<()> {
6434    use std::os::windows::fs::FileExt;
6435    while !bytes.is_empty() {
6436        let written = file.seek_write(bytes, offset).map_err(io)?;
6437        if written == 0 {
6438            return Err(invalid("a write to the native file wrote nothing"));
6439        }
6440        offset += written as u64;
6441        bytes = &bytes[written..];
6442    }
6443    Ok(())
6444}
6445
6446/// Somewhere that is neither, where the cursor is all there is.
6447#[cfg(not(any(unix, windows)))]
6448fn write_at(file: &File, offset: u64, bytes: &[u8]) -> Result<()> {
6449    use std::io::Write;
6450    let mut file = file.try_clone().map_err(io)?;
6451    file.seek(SeekFrom::Start(offset)).map_err(io)?;
6452    file.write_all(bytes).map_err(io)
6453}
6454
6455/// Reads one span of a file at an offset, without moving a cursor anybody else can see.
6456///
6457/// Every reader of a table shares one [`File`] behind an [`Arc`], and a grouped aggregate reads its
6458/// pages from several threads at once, so this has to be positional. Seeking and then reading is
6459/// two calls with a gap in the middle, and in that gap another thread's seek lands and the read
6460/// comes back with somebody else's bytes.
6461///
6462/// Both of these can come back short, so both loop. A read of zero bytes before the span is filled
6463/// means the file stops earlier than the directory said it does.
6464#[cfg(unix)]
6465fn read_at(file: &File, mut offset: u64, mut bytes: &mut [u8]) -> Result<()> {
6466    use std::os::unix::fs::FileExt;
6467    while !bytes.is_empty() {
6468        let read = file.read_at(bytes, offset).map_err(io)?;
6469        if read == 0 {
6470            return Err(invalid("column page ends before its declared length"));
6471        }
6472        offset += read as u64;
6473        bytes = &mut bytes[read..];
6474    }
6475    Ok(())
6476}
6477
6478/// The same read, on the call Windows spells differently.
6479///
6480/// `seek_read` is one `ReadFile` carrying the offset with it, so two of them cannot interleave the
6481/// way a seek and a read can. It does leave the shared cursor somewhere afterwards, which is why
6482/// nothing in this file may read that cursor.
6483#[cfg(windows)]
6484fn read_at(file: &File, mut offset: u64, mut bytes: &mut [u8]) -> Result<()> {
6485    use std::os::windows::fs::FileExt;
6486    while !bytes.is_empty() {
6487        let read = file.seek_read(bytes, offset).map_err(io)?;
6488        if read == 0 {
6489            return Err(invalid("column page ends before its declared length"));
6490        }
6491        offset += read as u64;
6492        bytes = &mut bytes[read..];
6493    }
6494    Ok(())
6495}
6496
6497/// Somewhere that is neither, where the cursor is all there is.
6498///
6499/// This one does race, and there is no way to write it so it does not. Nothing we build for runs
6500/// here, so it exists to keep the crate compiling rather than to be correct under threads.
6501#[cfg(not(any(unix, windows)))]
6502fn read_at(file: &File, offset: u64, bytes: &mut [u8]) -> Result<()> {
6503    let mut file = file.try_clone().map_err(io)?;
6504    file.seek(SeekFrom::Start(offset)).map_err(io)?;
6505    file.read_exact(bytes).map_err(io)
6506}
6507
6508/// What a column type is called in the directory.
6509///
6510/// A tag is a number in a file somebody else wrote, so a tag that has been used is used forever and
6511/// the only thing that may happen to this list is that it grows. 1 to 13 are the tags the format
6512/// had when it could store thirteen types, and 14 to 27 are the rest, in the order they were added
6513/// rather than in an order that means anything.
6514fn type_tag(ty: &LogicalType) -> Result<u8> {
6515    match ty {
6516        LogicalType::SmallInt => Ok(1),
6517        LogicalType::Integer => Ok(2),
6518        LogicalType::BigInt => Ok(3),
6519        LogicalType::Varchar => Ok(4),
6520        LogicalType::Date => Ok(5),
6521        LogicalType::Timestamp => Ok(6),
6522        LogicalType::Boolean => Ok(7),
6523        LogicalType::TinyInt => Ok(8),
6524        LogicalType::UTinyInt => Ok(9),
6525        LogicalType::USmallInt => Ok(10),
6526        LogicalType::UInteger => Ok(11),
6527        LogicalType::UBigInt => Ok(12),
6528        LogicalType::Decimal { .. } => Ok(13),
6529        LogicalType::Float => Ok(14),
6530        LogicalType::Double => Ok(15),
6531        LogicalType::HugeInt => Ok(16),
6532        LogicalType::UHugeInt => Ok(17),
6533        LogicalType::Time => Ok(18),
6534        LogicalType::TimeTz => Ok(19),
6535        LogicalType::TimestampTz => Ok(20),
6536        LogicalType::Interval => Ok(21),
6537        LogicalType::Uuid => Ok(22),
6538        LogicalType::Blob => Ok(23),
6539        LogicalType::Bit => Ok(24),
6540        LogicalType::TimestampS => Ok(25),
6541        LogicalType::TimestampMs => Ok(26),
6542        LogicalType::TimestampNs => Ok(27),
6543        _ => Err(Error::not_implemented(format!("native storage for {ty}"))),
6544    }
6545}
6546
6547/// The tag of a column type, and the parameters of the ones that have any.
6548///
6549/// Only `DECIMAL` has parameters today. Width and scale go after the tag rather than into it
6550/// because they are what says how wide a value is on disk, and a reader that guessed would read the
6551/// wrong number of bytes per row rather than the wrong number of digits.
6552fn put_type(out: &mut Vec<u8>, ty: &LogicalType) -> Result<()> {
6553    out.push(type_tag(ty)?);
6554    if let LogicalType::Decimal { width, scale } = ty {
6555        out.push(*width);
6556        out.push(*scale);
6557    }
6558    Ok(())
6559}
6560
6561/// The other half of [`put_type`], reading the parameters the tag says are there.
6562fn read_type(cur: &mut Cursor<'_>) -> Result<LogicalType> {
6563    let tag = cur.u8()?;
6564    if tag == 13 {
6565        let width = cur.u8()?;
6566        let scale = cur.u8()?;
6567        return LogicalType::decimal(width, scale)
6568            .map_err(|_| invalid("decimal column width and scale are not a decimal"));
6569    }
6570    tag_type(tag)
6571}
6572
6573fn tag_type(tag: u8) -> Result<LogicalType> {
6574    match tag {
6575        1 => Ok(LogicalType::SmallInt),
6576        2 => Ok(LogicalType::Integer),
6577        3 => Ok(LogicalType::BigInt),
6578        4 => Ok(LogicalType::Varchar),
6579        5 => Ok(LogicalType::Date),
6580        6 => Ok(LogicalType::Timestamp),
6581        7 => Ok(LogicalType::Boolean),
6582        8 => Ok(LogicalType::TinyInt),
6583        9 => Ok(LogicalType::UTinyInt),
6584        10 => Ok(LogicalType::USmallInt),
6585        11 => Ok(LogicalType::UInteger),
6586        12 => Ok(LogicalType::UBigInt),
6587        14 => Ok(LogicalType::Float),
6588        15 => Ok(LogicalType::Double),
6589        16 => Ok(LogicalType::HugeInt),
6590        17 => Ok(LogicalType::UHugeInt),
6591        18 => Ok(LogicalType::Time),
6592        19 => Ok(LogicalType::TimeTz),
6593        20 => Ok(LogicalType::TimestampTz),
6594        21 => Ok(LogicalType::Interval),
6595        22 => Ok(LogicalType::Uuid),
6596        23 => Ok(LogicalType::Blob),
6597        24 => Ok(LogicalType::Bit),
6598        25 => Ok(LogicalType::TimestampS),
6599        26 => Ok(LogicalType::TimestampMs),
6600        27 => Ok(LogicalType::TimestampNs),
6601        _ => Err(invalid("column type tag is unknown")),
6602    }
6603}
6604
6605fn put_u16(out: &mut Vec<u8>, value: u16) {
6606    out.extend_from_slice(&value.to_le_bytes());
6607}
6608fn put_u32(out: &mut Vec<u8>, value: u32) {
6609    out.extend_from_slice(&value.to_le_bytes());
6610}
6611fn put_u64(out: &mut Vec<u8>, value: u64) {
6612    out.extend_from_slice(&value.to_le_bytes());
6613}
6614fn put_var_u64(out: &mut Vec<u8>, mut value: u64) {
6615    while value >= 0x80 {
6616        out.push((value as u8 & 0x7f) | 0x80);
6617        value >>= 7;
6618    }
6619    out.push(value as u8);
6620}
6621
6622fn frequency_order(left: FrequencyValue, right: FrequencyValue) -> Ordering {
6623    match (left, right) {
6624        (FrequencyValue::Null, FrequencyValue::Null) => Ordering::Equal,
6625        (FrequencyValue::Null, _) => Ordering::Less,
6626        (_, FrequencyValue::Null) => Ordering::Greater,
6627        (FrequencyValue::Integer(left), FrequencyValue::Integer(right)) => left.cmp(&right),
6628        (FrequencyValue::Code(left), FrequencyValue::Code(right)) => left.cmp(&right),
6629        (FrequencyValue::Integer(_), FrequencyValue::Code(_)) => Ordering::Less,
6630        (FrequencyValue::Code(_), FrequencyValue::Integer(_)) => Ordering::Greater,
6631    }
6632}
6633
6634/// Leaves the [`FREQUENCY_ENTRIES`] commonest entries in order and says what the next one counted.
6635///
6636/// There is one entry a distinct value, so on `URL` this is handed two and a quarter million of
6637/// them and keeps five hundred and twelve. Sorting all of them to throw almost all of them away is
6638/// the whole of what counting a dictionary column used to cost, 2.13 seconds of it on `URL` at eight
6639/// million rows against 11.93 for compressing the same column's values.
6640///
6641/// Partitioning answers both questions instead. It puts the five hundred and thirteenth entry where
6642/// it belongs and everything commoner in front of it, which is the entries to keep and the count to
6643/// report as the largest one omitted, and then only the part that survives is sorted. The order that
6644/// comes out is the order the sort gave, because the tie break makes the comparison total: two
6645/// entries never hold the same value.
6646fn keep_most_frequent(entries: &mut Vec<FrequencyEntry>) -> u64 {
6647    let order = |left: &FrequencyEntry, right: &FrequencyEntry| {
6648        right.count.cmp(&left.count).then_with(|| frequency_order(left.value, right.value))
6649    };
6650    let omitted_max = if entries.len() > FREQUENCY_ENTRIES {
6651        let (_, next, _) = entries.select_nth_unstable_by(FREQUENCY_ENTRIES, order);
6652        let omitted_max = next.count;
6653        entries.truncate(FREQUENCY_ENTRIES);
6654        omitted_max
6655    } else {
6656        0
6657    };
6658    entries.sort_unstable_by(order);
6659    omitted_max
6660}
6661
6662fn code_frequency(
6663    dictionary: &GlobalDictionary,
6664    flat: &[u8],
6665    bases: &[u64],
6666) -> Result<(FrequencySummary, Vec<Option<Vec<u8>>>)> {
6667    let mut entries = dictionary
6668        .counts
6669        .iter()
6670        .enumerate()
6671        .filter(|(_, count)| **count != 0)
6672        .map(|(code, &count)| FrequencyEntry { value: FrequencyValue::Code(code as u32), count })
6673        .collect::<Vec<_>>();
6674    if dictionary.nulls != 0 {
6675        entries.push(FrequencyEntry { value: FrequencyValue::Null, count: dictionary.nulls });
6676    }
6677    let omitted_max = keep_most_frequent(&mut entries);
6678    let mut spans = Vec::with_capacity(entries.len());
6679    let mut text_bytes = 0_usize;
6680    for entry in &entries {
6681        let span = match entry.value {
6682            FrequencyValue::Code(code) => {
6683                let span = GlobalDictionary::value_span(&dictionary.ends, bases, code as usize);
6684                let bytes = flat
6685                    .get(span.0..span.1)
6686                    .ok_or_else(|| invalid("a frequency code is outside its dictionary"))?;
6687                text_bytes = text_bytes.saturating_add(bytes.len());
6688                Some(span)
6689            }
6690            FrequencyValue::Null | FrequencyValue::Integer(_) => None,
6691        };
6692        spans.push(span);
6693    }
6694    let texts = if text_bytes > FREQUENCY_TEXT_BUDGET {
6695        Vec::new()
6696    } else {
6697        spans.into_iter().map(|span| span.map(|(from, to)| flat[from..to].to_vec())).collect()
6698    };
6699    Ok((
6700        FrequencySummary {
6701            entries,
6702            omitted_max,
6703            ordinals: Vec::new(),
6704            ordinal_entries: Vec::new(),
6705        },
6706        texts,
6707    ))
6708}
6709
6710fn encode_directory(table: &Table) -> Result<Vec<u8>> {
6711    let mut out = DIRECTORY.to_vec();
6712    let name = table.name.as_bytes();
6713    put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("table name too long"))?);
6714    out.extend_from_slice(name);
6715    put_u16(&mut out, u16::try_from(table.fields.len()).map_err(|_| invalid("too many columns"))?);
6716    for field in &table.fields {
6717        let name = field.name.as_bytes();
6718        put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("column name too long"))?);
6719        out.extend_from_slice(name);
6720        put_type(&mut out, &field.ty)?;
6721        out.push(u8::from(field.not_null));
6722    }
6723    for dictionary in &table.dictionaries {
6724        match dictionary {
6725            None => out.push(0),
6726            Some(page) => {
6727                out.push(1);
6728                put_u64(&mut out, page.offset);
6729                put_u32(&mut out, page.length);
6730                put_u64(&mut out, page.hash);
6731            }
6732        }
6733    }
6734    for distinct in &table.distincts {
6735        match distinct {
6736            None => out.push(0),
6737            Some(count) => {
6738                out.push(1);
6739                put_u64(&mut out, *count);
6740            }
6741        }
6742    }
6743    put_u64(&mut out, u64::try_from(table.rows).map_err(|_| invalid("row count overflow"))?);
6744    put_u32(&mut out, u32::try_from(table.stripes.len()).map_err(|_| invalid("too many stripes"))?);
6745    for stripe in &table.stripes {
6746        put_u32(
6747            &mut out,
6748            u32::try_from(stripe.parts.len()).map_err(|_| invalid("too many parts in a stripe"))?,
6749        );
6750        for &rows in &stripe.parts {
6751            put_u32(&mut out, rows);
6752        }
6753        put_u64(&mut out, stripe.index.offset);
6754        put_u32(&mut out, stripe.index.length);
6755        for page in &stripe.pages {
6756            put_u64(&mut out, page.offset);
6757            put_u32(&mut out, page.length);
6758        }
6759        // A membership index says which of a dictionary's codes a part holds, so a column the writer
6760        // decided against giving a dictionary has nothing for it to be about and writes none. Every
6761        // file written before that decision existed has a dictionary on every varchar column, so
6762        // this reads those files byte for byte the way it always did.
6763        for ((field, dictionary), membership) in
6764            table.fields.iter().zip(&table.dictionaries).zip(stripe.memberships.slots())
6765        {
6766            if field.ty != LogicalType::Varchar || dictionary.is_none() {
6767                continue;
6768            }
6769            let page =
6770                membership.ok_or_else(|| invalid("string page has no code membership index"))?;
6771            put_u64(&mut out, page.offset);
6772            put_u32(&mut out, page.length);
6773            put_u64(&mut out, page.hash);
6774        }
6775        for sieve in stripe.sieves.slots() {
6776            match sieve {
6777                None => out.push(0),
6778                Some(page) => {
6779                    out.push(1);
6780                    put_u64(&mut out, page.offset);
6781                    put_u32(&mut out, page.length);
6782                    put_u64(&mut out, page.hash);
6783                }
6784            }
6785        }
6786        for held in stripe.part_ranges.slots() {
6787            match held {
6788                None => out.push(0),
6789                Some(page) => {
6790                    out.push(1);
6791                    put_u64(&mut out, page.offset);
6792                    put_u32(&mut out, page.length);
6793                    put_u64(&mut out, page.hash);
6794                }
6795            }
6796        }
6797        for range in stripe.zone.columns() {
6798            put_bound(&mut out, range.low.as_ref())?;
6799            put_bound(&mut out, range.high.as_ref())?;
6800            put_u32(
6801                &mut out,
6802                u32::try_from(range.nulls).map_err(|_| invalid("null count overflow"))?,
6803            );
6804            out.push(u8::from(range.exact));
6805            match range.sum {
6806                None => out.push(0),
6807                Some(total) => {
6808                    out.push(1);
6809                    out.extend_from_slice(&total.to_le_bytes());
6810                }
6811            }
6812        }
6813    }
6814    out.extend_from_slice(FREQUENCIES);
6815    put_u16(
6816        &mut out,
6817        u16::try_from(table.frequencies.len())
6818            .map_err(|_| invalid("too many frequency columns"))?,
6819    );
6820    for summary in &table.frequencies {
6821        let summary = match summary {
6822            None => {
6823                out.push(0);
6824                continue;
6825            }
6826            Some(Frequencies::Held(summary)) => summary,
6827            // Only a reader leaves a synopsis in the file, and nothing writes a reader's table back.
6828            Some(Frequencies::Stored { .. }) => {
6829                return Err(invalid("a synopsis left in the file cannot be written back"));
6830            }
6831        };
6832        out.push(1);
6833        put_u64(&mut out, summary.omitted_max);
6834        put_u32(
6835            &mut out,
6836            u32::try_from(summary.entries.len())
6837                .map_err(|_| invalid("too many frequency entries"))?,
6838        );
6839        for entry in &summary.entries {
6840            match entry.value {
6841                FrequencyValue::Null => out.push(0),
6842                FrequencyValue::Integer(value) => {
6843                    out.push(1);
6844                    out.extend_from_slice(&value.to_le_bytes());
6845                }
6846                FrequencyValue::Code(value) => {
6847                    out.push(2);
6848                    put_u32(&mut out, value);
6849                }
6850            }
6851            put_u64(&mut out, entry.count);
6852        }
6853        put_u32(
6854            &mut out,
6855            u32::try_from(summary.ordinals.len())
6856                .map_err(|_| invalid("too many frequency ordinals"))?,
6857        );
6858        let mut previous = 0_u64;
6859        for (at, &ordinal) in summary.ordinals.iter().enumerate() {
6860            let delta = if at == 0 {
6861                ordinal
6862            } else {
6863                ordinal
6864                    .checked_sub(previous)
6865                    .ok_or_else(|| invalid("frequency ordinals are not ordered"))?
6866            };
6867            if at != 0 && delta == 0 {
6868                return Err(invalid("frequency ordinals are not unique"));
6869            }
6870            put_var_u64(&mut out, delta);
6871            previous = ordinal;
6872        }
6873        if summary.ordinal_entries.len() != summary.ordinals.len() {
6874            return Err(invalid("frequency ordinal values have a different length"));
6875        }
6876        for &entry in &summary.ordinal_entries {
6877            if entry as usize >= summary.entries.len() {
6878                return Err(invalid("frequency ordinal value is outside its entries"));
6879            }
6880            put_u16(&mut out, entry);
6881        }
6882    }
6883    if !table.pair_frequencies.is_empty() {
6884        out.extend_from_slice(PAIR_FREQUENCIES);
6885        put_u16(
6886            &mut out,
6887            u16::try_from(table.pair_frequencies.len())
6888                .map_err(|_| invalid("too many pair frequency summaries"))?,
6889        );
6890        for summary in &table.pair_frequencies {
6891            put_u16(&mut out, summary.first);
6892            put_u16(&mut out, summary.second);
6893            put_u64(&mut out, summary.omitted_max);
6894            put_u16(
6895                &mut out,
6896                u16::try_from(summary.entries.len())
6897                    .map_err(|_| invalid("too many pair frequency entries"))?,
6898            );
6899            for entry in &summary.entries {
6900                put_u16(&mut out, entry.first_entry);
6901                match entry.second {
6902                    None => out.push(0),
6903                    Some(code) => {
6904                        out.push(1);
6905                        put_u32(&mut out, code);
6906                    }
6907                }
6908                put_u64(&mut out, entry.count);
6909            }
6910        }
6911    }
6912    let text_columns = table.frequency_texts.iter().filter(|texts| !texts.is_empty()).count();
6913    if text_columns != 0 {
6914        out.extend_from_slice(FREQUENCY_TEXTS);
6915        put_u16(
6916            &mut out,
6917            u16::try_from(text_columns)
6918                .map_err(|_| invalid("too many string frequency columns"))?,
6919        );
6920        for (column, texts) in table.frequency_texts.iter().enumerate() {
6921            if texts.is_empty() {
6922                continue;
6923            }
6924            put_u16(
6925                &mut out,
6926                u16::try_from(column).map_err(|_| invalid("frequency text column overflows"))?,
6927            );
6928            put_u16(
6929                &mut out,
6930                u16::try_from(texts.len())
6931                    .map_err(|_| invalid("too many frequency text entries"))?,
6932            );
6933            for text in texts {
6934                match text {
6935                    None => out.push(0),
6936                    Some(text) => {
6937                        out.push(1);
6938                        put_u32(
6939                            &mut out,
6940                            u32::try_from(text.len())
6941                                .map_err(|_| invalid("frequency text is too long"))?,
6942                        );
6943                        out.extend_from_slice(text);
6944                    }
6945                }
6946            }
6947        }
6948    }
6949    if let Some(summary) = &table.host_groups {
6950        out.extend_from_slice(HOST_GROUPS);
6951        put_u16(
6952            &mut out,
6953            u16::try_from(summary.column).map_err(|_| invalid("host column overflows"))?,
6954        );
6955        put_u64(&mut out, summary.omitted_max);
6956        put_u16(
6957            &mut out,
6958            u16::try_from(summary.entries.len()).map_err(|_| invalid("too many host groups"))?,
6959        );
6960        for entry in &summary.entries {
6961            put_u32(
6962                &mut out,
6963                u32::try_from(entry.host.len()).map_err(|_| invalid("host name is too long"))?,
6964            );
6965            out.extend_from_slice(entry.host.as_bytes());
6966            put_u64(&mut out, entry.count);
6967            out.extend_from_slice(&entry.bytes_sum.to_le_bytes());
6968            put_u32(
6969                &mut out,
6970                u32::try_from(entry.minimum.len())
6971                    .map_err(|_| invalid("host minimum is too long"))?,
6972            );
6973            out.extend_from_slice(entry.minimum.as_bytes());
6974        }
6975    }
6976    // Written only when there is a declaration, so that the common file is the same bytes it was
6977    // and the section is not a byte of zero on every table in the world that never asked for one.
6978    if let Some(clustering) = &table.clustering {
6979        out.extend_from_slice(CLUSTERING);
6980        out.push(clustering.width().tag());
6981        put_u16(
6982            &mut out,
6983            u16::try_from(clustering.columns().len())
6984                .map_err(|_| invalid("too many clustering columns"))?,
6985        );
6986        for &column in clustering.columns() {
6987            put_u16(
6988                &mut out,
6989                u16::try_from(column).map_err(|_| invalid("clustering column index overflow"))?,
6990            );
6991        }
6992    }
6993    // The section table, last, behind its own magic, for the same reason the frequency block is
6994    // behind its own: a reader that stops before it gets a table with no sections, and a table with
6995    // no sections is a correct table. The one difference from the blocks before it is that this one
6996    // is written even when it is empty, so that a file written by this build always says which
6997    // sections it has rather than leaving a reader to infer it from where the bytes ran out.
6998    out.extend_from_slice(SECTIONS);
6999    put_u64(&mut out, table.generation);
7000    put_u16(
7001        &mut out,
7002        u16::try_from(table.sections.len()).map_err(|_| invalid("too many sections"))?,
7003    );
7004    for held in &table.sections {
7005        held.encode(&mut out)?;
7006    }
7007    if table.dictionary_payloads.iter().any(|&bytes| bytes != 0) {
7008        out.extend_from_slice(DICTIONARY_PAYLOADS);
7009        put_u16(
7010            &mut out,
7011            u16::try_from(table.fields.len()).map_err(|_| invalid("too many columns"))?,
7012        );
7013        for at in 0..table.fields.len() {
7014            put_u64(&mut out, table.dictionary_payloads.get(at).copied().unwrap_or(0));
7015        }
7016    }
7017    Ok(out)
7018}
7019
7020/// The small level of the directory, naming every table in the file.
7021///
7022/// This is what a footer slot points at. Each entry carries its own checksum over its table
7023/// directory, so a table whose directory is torn is found when that table is first touched rather
7024/// than being trusted because the catalog around it checksummed.
7025///
7026/// The views go after the tables and are whole here, since a view is text and a column list and has
7027/// no pages for a second level to point at.
7028fn table_nonzero_counts(table: &Table) -> Vec<Option<u64>> {
7029    table
7030        .fields
7031        .iter()
7032        .enumerate()
7033        .map(|(column, field)| {
7034            if !matches!(
7035                field.ty,
7036                LogicalType::TinyInt
7037                    | LogicalType::SmallInt
7038                    | LogicalType::Integer
7039                    | LogicalType::BigInt
7040                    | LogicalType::UTinyInt
7041                    | LogicalType::USmallInt
7042                    | LogicalType::UInteger
7043                    | LogicalType::UBigInt
7044            ) {
7045                return None;
7046            }
7047            let Some(Frequencies::Held(summary)) = &table.frequencies[column] else {
7048                return None;
7049            };
7050            let zero = summary
7051                .entries
7052                .iter()
7053                .find(|entry| entry.value == FrequencyValue::Integer(0))
7054                .map(|entry| entry.count)
7055                .or_else(|| (summary.omitted_max == 0).then_some(0))?;
7056            let nulls = table.stripes.iter().try_fold(0_u64, |count, stripe| {
7057                count.checked_add(stripe.zone.column(column)?.nulls as u64)
7058            })?;
7059            (table.rows as u64).checked_sub(nulls)?.checked_sub(zero)
7060        })
7061        .collect()
7062}
7063
7064fn signed_integer(ty: &LogicalType) -> bool {
7065    matches!(
7066        ty,
7067        LogicalType::TinyInt | LogicalType::SmallInt | LogicalType::Integer | LogicalType::BigInt
7068    )
7069}
7070
7071fn integer_or_date(ty: &LogicalType) -> bool {
7072    matches!(
7073        ty,
7074        LogicalType::TinyInt
7075            | LogicalType::SmallInt
7076            | LogicalType::Integer
7077            | LogicalType::BigInt
7078            | LogicalType::UTinyInt
7079            | LogicalType::USmallInt
7080            | LogicalType::UInteger
7081            | LogicalType::UBigInt
7082            | LogicalType::Date
7083    )
7084}
7085
7086fn table_integer_extremes(table: &Table) -> Vec<StoredIntegerExtremes> {
7087    table
7088        .fields
7089        .iter()
7090        .enumerate()
7091        .map(|(column, field)| {
7092            if !integer_or_date(&field.ty) {
7093                return None;
7094            }
7095            let mut low: Option<i128> = None;
7096            let mut high: Option<i128> = None;
7097            for stripe in &table.stripes {
7098                let range = stripe.zone.column(column)?;
7099                if !range.exact {
7100                    return None;
7101                }
7102                match (range.low.as_ref(), range.high.as_ref()) {
7103                    (Some(Bound::Int(small)), Some(Bound::Int(large))) => {
7104                        low = Some(low.map_or(*small, |held| held.min(*small)));
7105                        high = Some(high.map_or(*large, |held| held.max(*large)));
7106                    }
7107                    (None, None) if stripe.rows == range.nulls => {}
7108                    _ => return None,
7109                }
7110            }
7111            Some(low.zip(high))
7112        })
7113        .collect()
7114}
7115
7116fn reader_integer_extremes(reader: &Reader) -> Result<Vec<StoredIntegerExtremes>> {
7117    reader
7118        .table
7119        .fields
7120        .iter()
7121        .enumerate()
7122        .map(|(column, field)| {
7123            if !integer_or_date(&field.ty) {
7124                return Ok(None);
7125            }
7126            match reader.exact_extremes(column)? {
7127                Some((Bound::Int(low), Bound::Int(high))) => Ok(Some(Some((low, high)))),
7128                None if reader.null_count(column)? == reader.table.rows as u64 => Ok(Some(None)),
7129                _ => Ok(None),
7130            }
7131        })
7132        .collect()
7133}
7134
7135fn table_complete_numeric_frequencies(table: &Table) -> Vec<StoredNumericFrequencies> {
7136    table
7137        .fields
7138        .iter()
7139        .enumerate()
7140        .map(|(column, field)| {
7141            if !integer_or_date(&field.ty) {
7142                return None;
7143            }
7144            let Some(Frequencies::Held(summary)) = table.frequencies.get(column)?.as_ref() else {
7145                return None;
7146            };
7147            if summary.omitted_max != 0 || summary.entries.len() > MAX_CATALOG_FREQUENCIES {
7148                return None;
7149            }
7150            let entries = summary
7151                .entries
7152                .iter()
7153                .map(|entry| {
7154                    let value = match entry.value {
7155                        FrequencyValue::Null => None,
7156                        FrequencyValue::Integer(value) => Some(value),
7157                        FrequencyValue::Code(_) => return None,
7158                    };
7159                    Some((value, entry.count))
7160                })
7161                .collect::<Option<Vec<_>>>()?;
7162            let rows = entries.iter().try_fold(0_u64, |sum, (_, count)| sum.checked_add(*count))?;
7163            (rows == table.rows as u64).then_some(entries)
7164        })
7165        .collect()
7166}
7167
7168fn numeric_frequency_value(value: &Value) -> Option<Option<i128>> {
7169    Some(match value {
7170        Value::Null => None,
7171        Value::TinyInt(value) => Some(i128::from(*value)),
7172        Value::SmallInt(value) => Some(i128::from(*value)),
7173        Value::Integer(value) | Value::Date(value) => Some(i128::from(*value)),
7174        Value::BigInt(value) => Some(i128::from(*value)),
7175        Value::UTinyInt(value) => Some(i128::from(*value)),
7176        Value::USmallInt(value) => Some(i128::from(*value)),
7177        Value::UInteger(value) => Some(i128::from(*value)),
7178        Value::UBigInt(value) => Some(i128::from(*value)),
7179        _ => return None,
7180    })
7181}
7182
7183fn reader_complete_numeric_frequencies(reader: &Reader) -> Result<Vec<StoredNumericFrequencies>> {
7184    reader
7185        .table
7186        .fields
7187        .iter()
7188        .enumerate()
7189        .map(|(column, field)| {
7190            if !integer_or_date(&field.ty) {
7191                return Ok(None);
7192            }
7193            let Some(summary) = reader.frequency_summary(column)? else { return Ok(None) };
7194            if summary.omitted_max != 0 || summary.entries.len() > MAX_CATALOG_FREQUENCIES {
7195                return Ok(None);
7196            }
7197            let entries = reader.decode_frequencies(column, &field.ty, &summary.entries)?;
7198            let Some(entries) = entries
7199                .iter()
7200                .map(|(value, count)| Some((numeric_frequency_value(value)?, *count)))
7201                .collect::<Option<Vec<_>>>()
7202            else {
7203                return Ok(None);
7204            };
7205            let rows = entries.iter().try_fold(0_u64, |sum, (_, count)| sum.checked_add(*count));
7206            Ok((rows == Some(reader.table.rows as u64)).then_some(entries))
7207        })
7208        .collect()
7209}
7210
7211fn table_exact_sum(table: &Table, column: usize) -> Option<(i128, u64)> {
7212    table.stripes.iter().try_fold((0_i128, 0_u64), |(sum, count), stripe| {
7213        let range = stripe.zone.column(column)?;
7214        let sum = sum.checked_add(range.sum?)?;
7215        let nonnull = (stripe.rows as u64).checked_sub(range.nulls as u64)?;
7216        Some((sum, count.checked_add(nonnull)?))
7217    })
7218}
7219
7220fn table_aggregate_sums(table: &Table) -> Vec<Option<(i128, u64)>> {
7221    table
7222        .fields
7223        .iter()
7224        .enumerate()
7225        .map(|(column, field)| {
7226            signed_integer(&field.ty).then(|| table_exact_sum(table, column)).flatten()
7227        })
7228        .collect()
7229}
7230
7231fn reader_nonzero_counts(reader: &Reader) -> Result<Vec<Option<u64>>> {
7232    reader
7233        .table
7234        .fields
7235        .iter()
7236        .enumerate()
7237        .map(|(column, field)| {
7238            if !matches!(
7239                field.ty,
7240                LogicalType::TinyInt
7241                    | LogicalType::SmallInt
7242                    | LogicalType::Integer
7243                    | LogicalType::BigInt
7244                    | LogicalType::UTinyInt
7245                    | LogicalType::USmallInt
7246                    | LogicalType::UInteger
7247                    | LogicalType::UBigInt
7248            ) {
7249                return Ok(None);
7250            }
7251            let Some(summary) = reader.frequency_summary(column)? else {
7252                return Ok(None);
7253            };
7254            let zero = summary
7255                .entries
7256                .iter()
7257                .find(|entry| entry.value == FrequencyValue::Integer(0))
7258                .map(|entry| entry.count)
7259                .or_else(|| (summary.omitted_max == 0).then_some(0));
7260            let Some(zero) = zero else { return Ok(None) };
7261            let nulls = reader.null_count(column)?;
7262            Ok((reader.table.rows as u64)
7263                .checked_sub(nulls)
7264                .and_then(|count| count.checked_sub(zero)))
7265        })
7266        .collect()
7267}
7268
7269fn reader_aggregate_sums(reader: &Reader) -> Result<Vec<Option<(i128, u64)>>> {
7270    reader
7271        .table
7272        .fields
7273        .iter()
7274        .enumerate()
7275        .map(
7276            |(column, field)| {
7277                if signed_integer(&field.ty) { reader.exact_sum(column) } else { Ok(None) }
7278            },
7279        )
7280        .collect()
7281}
7282
7283fn encode_catalog(entries: &[Entry], views: &[ViewEntry]) -> Result<Vec<u8>> {
7284    let mut out = CATALOG.to_vec();
7285    put_u32(&mut out, u32::try_from(entries.len()).map_err(|_| invalid("too many tables"))?);
7286    for entry in entries {
7287        let name = entry.name.as_bytes();
7288        put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("table name too long"))?);
7289        out.extend_from_slice(name);
7290        put_u64(&mut out, u64::try_from(entry.rows).map_err(|_| invalid("row count overflow"))?);
7291        put_u16(
7292            &mut out,
7293            u16::try_from(entry.fields.len()).map_err(|_| invalid("too many columns"))?,
7294        );
7295        for field in &entry.fields {
7296            let name = field.name.as_bytes();
7297            put_u16(
7298                &mut out,
7299                u16::try_from(name.len()).map_err(|_| invalid("column name too long"))?,
7300            );
7301            out.extend_from_slice(name);
7302            put_type(&mut out, &field.ty)?;
7303            out.push(u8::from(field.not_null));
7304        }
7305        put_u64(&mut out, entry.directory.offset);
7306        put_u32(&mut out, entry.directory.length);
7307        put_u64(&mut out, entry.directory.hash);
7308    }
7309    put_u32(&mut out, u32::try_from(views.len()).map_err(|_| invalid("too many views"))?);
7310    for view in views {
7311        let name = view.name.as_bytes();
7312        put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("view name too long"))?);
7313        out.extend_from_slice(name);
7314        put_long_text(&mut out, &view.sql, "view body")?;
7315        put_long_text(&mut out, &view.statement, "view statement")?;
7316        put_u16(
7317            &mut out,
7318            u16::try_from(view.aliases.len()).map_err(|_| invalid("too many aliases"))?,
7319        );
7320        for alias in &view.aliases {
7321            let alias = alias.as_bytes();
7322            put_u16(
7323                &mut out,
7324                u16::try_from(alias.len()).map_err(|_| invalid("alias name too long"))?,
7325            );
7326            out.extend_from_slice(alias);
7327        }
7328        put_u16(
7329            &mut out,
7330            u16::try_from(view.columns.len()).map_err(|_| invalid("too many columns"))?,
7331        );
7332        for field in &view.columns {
7333            let name = field.name.as_bytes();
7334            put_u16(
7335                &mut out,
7336                u16::try_from(name.len()).map_err(|_| invalid("column name too long"))?,
7337            );
7338            out.extend_from_slice(name);
7339            put_type(&mut out, &field.ty)?;
7340            out.push(u8::from(field.not_null));
7341        }
7342    }
7343    out.extend_from_slice(NONZERO_COUNTS);
7344    for entry in entries {
7345        if entry.nonzero.len() != entry.fields.len() {
7346            return Err(invalid("nonzero count width differs from schema"));
7347        }
7348        for count in &entry.nonzero {
7349            match count {
7350                None => out.push(0),
7351                Some(count) => {
7352                    out.push(1);
7353                    put_u64(&mut out, *count);
7354                }
7355            }
7356        }
7357    }
7358    out.extend_from_slice(AGGREGATE_SUMS);
7359    for entry in entries {
7360        if entry.aggregates.len() != entry.fields.len() {
7361            return Err(invalid("aggregate sum width differs from schema"));
7362        }
7363        for summary in &entry.aggregates {
7364            match summary {
7365                None => out.push(0),
7366                Some((sum, count)) => {
7367                    out.push(1);
7368                    out.extend_from_slice(&sum.to_le_bytes());
7369                    put_u64(&mut out, *count);
7370                }
7371            }
7372        }
7373    }
7374    out.extend_from_slice(DISTINCT_COUNTS);
7375    for entry in entries {
7376        if entry.distincts.len() != entry.fields.len() {
7377            return Err(invalid("distinct count width differs from schema"));
7378        }
7379        for count in &entry.distincts {
7380            match count {
7381                None => out.push(0),
7382                Some(count) => {
7383                    if *count > entry.rows as u64 {
7384                        return Err(invalid("distinct count exceeds table rows"));
7385                    }
7386                    out.push(1);
7387                    put_u64(&mut out, *count);
7388                }
7389            }
7390        }
7391    }
7392    out.extend_from_slice(INTEGER_EXTREMES);
7393    for entry in entries {
7394        if entry.extremes.len() != entry.fields.len() {
7395            return Err(invalid("integer extremes width differs from schema"));
7396        }
7397        for (field, extremes) in entry.fields.iter().zip(&entry.extremes) {
7398            match extremes {
7399                None => out.push(0),
7400                Some(None) if integer_or_date(&field.ty) => out.push(1),
7401                Some(Some((low, high))) if integer_or_date(&field.ty) && low <= high => {
7402                    out.push(2);
7403                    out.extend_from_slice(&low.to_le_bytes());
7404                    out.extend_from_slice(&high.to_le_bytes());
7405                }
7406                _ => return Err(invalid("integer extremes type or range differs")),
7407            }
7408        }
7409    }
7410    out.extend_from_slice(COMPLETE_FREQUENCIES);
7411    for entry in entries {
7412        if entry.frequencies.len() != entry.fields.len() {
7413            return Err(invalid("numeric frequency width differs from schema"));
7414        }
7415        for (field, frequencies) in entry.fields.iter().zip(&entry.frequencies) {
7416            match frequencies {
7417                None => out.push(0),
7418                Some(entries)
7419                    if integer_or_date(&field.ty) && entries.len() <= MAX_CATALOG_FREQUENCIES =>
7420                {
7421                    let mut total = 0_u64;
7422                    for (at, (value, count)) in entries.iter().enumerate() {
7423                        if entries[..at].iter().any(|(held, _)| held == value) {
7424                            return Err(invalid("numeric frequency value repeats"));
7425                        }
7426                        total = total
7427                            .checked_add(*count)
7428                            .ok_or_else(|| invalid("numeric frequency count overflows"))?;
7429                    }
7430                    if total != entry.rows as u64 {
7431                        return Err(invalid("numeric frequencies do not cover table rows"));
7432                    }
7433                    out.push(1);
7434                    out.push(entries.len() as u8);
7435                    for (value, count) in entries {
7436                        match value {
7437                            None => out.push(0),
7438                            Some(value) => {
7439                                out.push(1);
7440                                out.extend_from_slice(&value.to_le_bytes());
7441                            }
7442                        }
7443                        put_u64(&mut out, *count);
7444                    }
7445                }
7446                _ => return Err(invalid("numeric frequency type or width differs")),
7447            }
7448        }
7449    }
7450    Ok(out)
7451}
7452
7453/// A length and that many bytes, for text that is allowed to be longer than a name.
7454fn put_long_text(out: &mut Vec<u8>, text: &str, what: &str) -> Result<()> {
7455    let bytes = text.as_bytes();
7456    put_u32(out, u32::try_from(bytes.len()).map_err(|_| invalid(&format!("{what} too long")))?);
7457    out.extend_from_slice(bytes);
7458    Ok(())
7459}
7460
7461/// Reads the catalog directory back, checking every span against the file before anything is
7462/// allocated for it.
7463fn decode_catalog(bytes: &[u8], size: u64) -> Result<(Vec<Entry>, Vec<ViewEntry>)> {
7464    let mut cur = Cursor::new(bytes);
7465    if cur.take(8)? != CATALOG {
7466        return Err(invalid("catalog magic differs"));
7467    }
7468    let count = cur.u32()? as usize;
7469    let mut entries: Vec<Entry> = Vec::with_capacity(count.min(1024));
7470    for _ in 0..count {
7471        let name = cur.text()?;
7472        let rows = usize::try_from(cur.u64()?).map_err(|_| invalid("row count does not fit"))?;
7473        let width = cur.u16()? as usize;
7474        let mut fields = Vec::with_capacity(width);
7475        for _ in 0..width {
7476            let name = cur.text()?;
7477            let ty = read_type(&mut cur)?;
7478            let not_null = match cur.u8()? {
7479                0 => false,
7480                1 => true,
7481                _ => return Err(invalid("nullability flag differs")),
7482            };
7483            fields.push(Field { name, ty, not_null });
7484        }
7485        let directory = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
7486        let end = directory
7487            .offset
7488            .checked_add(u64::from(directory.length))
7489            .ok_or_else(|| invalid("table directory offset overflow"))?;
7490        if directory.offset < HEADER
7491            || end > size
7492            || directory.length as usize > MAX_DIRECTORY
7493            || directory.length == 0
7494        {
7495            return Err(invalid("table directory range is outside the file"));
7496        }
7497        if entries.iter().any(|held| held.name == name) {
7498            return Err(invalid("two tables in the catalog have the same name"));
7499        }
7500        let nonzero = vec![None; fields.len()];
7501        let aggregates = vec![None; fields.len()];
7502        let distincts = vec![None; fields.len()];
7503        let extremes = vec![None; fields.len()];
7504        let frequencies = vec![None; fields.len()];
7505        entries.push(Entry {
7506            name,
7507            fields,
7508            rows,
7509            directory,
7510            nonzero,
7511            aggregates,
7512            distincts,
7513            extremes,
7514            frequencies,
7515        });
7516    }
7517    // A catalog that ends where the tables end is a catalog with no views in it, which is every
7518    // file written before format 25. That is why the count is allowed to be missing rather than
7519    // read as a zero that has to be there: an older file has nothing after the last table entry at
7520    // all, and [`READABLE`] says those files still open.
7521    let count = if cur.done() { 0 } else { cur.u32()? as usize };
7522    let mut views: Vec<ViewEntry> = Vec::with_capacity(count.min(1024));
7523    for _ in 0..count {
7524        let name = cur.text()?;
7525        let sql = cur.long_text()?;
7526        let statement = cur.long_text()?;
7527        let width = cur.u16()? as usize;
7528        let mut aliases = Vec::with_capacity(width);
7529        for _ in 0..width {
7530            aliases.push(cur.text()?);
7531        }
7532        let width = cur.u16()? as usize;
7533        let mut columns = Vec::with_capacity(width);
7534        for _ in 0..width {
7535            let name = cur.text()?;
7536            let ty = read_type(&mut cur)?;
7537            let not_null = match cur.u8()? {
7538                0 => false,
7539                1 => true,
7540                _ => return Err(invalid("nullability flag differs")),
7541            };
7542            columns.push(Field { name, ty, not_null });
7543        }
7544        // The same rule the tables above get, and for the same reason. Two entries under one name
7545        // is a catalog nothing can answer a lookup from, and finding that out here is better than
7546        // finding it out from whichever of the two a search happened to reach first.
7547        if views.iter().any(|held| held.name == name) {
7548            return Err(invalid("two views in the catalog have the same name"));
7549        }
7550        if entries.iter().any(|held| held.name == name) {
7551            return Err(invalid("a table and a view in the catalog have the same name"));
7552        }
7553        views.push(ViewEntry { name, sql, statement, aliases, columns });
7554    }
7555    if !cur.done() {
7556        if cur.take(8)? != NONZERO_COUNTS {
7557            return Err(invalid("catalog extension magic differs"));
7558        }
7559        for entry in &mut entries {
7560            for (field, count) in entry.fields.iter().zip(&mut entry.nonzero) {
7561                *count = match cur.u8()? {
7562                    0 => None,
7563                    1 if matches!(
7564                        field.ty,
7565                        LogicalType::TinyInt
7566                            | LogicalType::SmallInt
7567                            | LogicalType::Integer
7568                            | LogicalType::BigInt
7569                            | LogicalType::UTinyInt
7570                            | LogicalType::USmallInt
7571                            | LogicalType::UInteger
7572                            | LogicalType::UBigInt
7573                    ) =>
7574                    {
7575                        let value = cur.u64()?;
7576                        if value > entry.rows as u64 {
7577                            return Err(invalid("nonzero count exceeds rows"));
7578                        }
7579                        Some(value)
7580                    }
7581                    _ => return Err(invalid("nonzero count tag or column type differs")),
7582                };
7583            }
7584        }
7585    }
7586    if !cur.done() {
7587        if cur.take(8)? != AGGREGATE_SUMS {
7588            return Err(invalid("aggregate catalog extension magic differs"));
7589        }
7590        for entry in &mut entries {
7591            for (field, summary) in entry.fields.iter().zip(&mut entry.aggregates) {
7592                *summary = match cur.u8()? {
7593                    0 => None,
7594                    1 if signed_integer(&field.ty) => {
7595                        let sum = i128::from_le_bytes(
7596                            cur.take(16)?
7597                                .try_into()
7598                                .map_err(|_| invalid("aggregate sum is truncated"))?,
7599                        );
7600                        let count = cur.u64()?;
7601                        if count > entry.rows as u64 {
7602                            return Err(invalid("aggregate count exceeds table rows"));
7603                        }
7604                        Some((sum, count))
7605                    }
7606                    _ => return Err(invalid("aggregate sum tag or column type differs")),
7607                };
7608            }
7609        }
7610    }
7611    if !cur.done() {
7612        if cur.take(8)? != DISTINCT_COUNTS {
7613            return Err(invalid("distinct catalog extension magic differs"));
7614        }
7615        for entry in &mut entries {
7616            for count in &mut entry.distincts {
7617                *count = match cur.u8()? {
7618                    0 => None,
7619                    1 => {
7620                        let value = cur.u64()?;
7621                        if value > entry.rows as u64 {
7622                            return Err(invalid("distinct count exceeds table rows"));
7623                        }
7624                        Some(value)
7625                    }
7626                    _ => return Err(invalid("distinct count tag differs")),
7627                };
7628            }
7629        }
7630    }
7631    if !cur.done() {
7632        if cur.take(8)? != INTEGER_EXTREMES {
7633            return Err(invalid("integer extremes catalog extension magic differs"));
7634        }
7635        for entry in &mut entries {
7636            for (field, extremes) in entry.fields.iter().zip(&mut entry.extremes) {
7637                *extremes = match cur.u8()? {
7638                    0 => None,
7639                    1 if integer_or_date(&field.ty) => Some(None),
7640                    2 if integer_or_date(&field.ty) => {
7641                        let low = i128::from_le_bytes(
7642                            cur.take(16)?
7643                                .try_into()
7644                                .map_err(|_| invalid("minimum is truncated"))?,
7645                        );
7646                        let high = i128::from_le_bytes(
7647                            cur.take(16)?
7648                                .try_into()
7649                                .map_err(|_| invalid("maximum is truncated"))?,
7650                        );
7651                        if low > high {
7652                            return Err(invalid("integer extremes are reversed"));
7653                        }
7654                        Some(Some((low, high)))
7655                    }
7656                    _ => return Err(invalid("integer extremes tag or type differs")),
7657                };
7658            }
7659        }
7660    }
7661    if !cur.done() {
7662        if cur.take(8)? != COMPLETE_FREQUENCIES {
7663            return Err(invalid("numeric frequency catalog extension magic differs"));
7664        }
7665        for entry in &mut entries {
7666            for (field, frequencies) in entry.fields.iter().zip(&mut entry.frequencies) {
7667                *frequencies = match cur.u8()? {
7668                    0 => None,
7669                    1 if integer_or_date(&field.ty) => {
7670                        let len = cur.u8()? as usize;
7671                        if len > MAX_CATALOG_FREQUENCIES {
7672                            return Err(invalid("too many catalog numeric frequencies"));
7673                        }
7674                        let mut values = Vec::with_capacity(len);
7675                        let mut total = 0_u64;
7676                        for _ in 0..len {
7677                            let value = match cur.u8()? {
7678                                0 => None,
7679                                1 => Some(i128::from_le_bytes(cur.take(16)?.try_into().map_err(
7680                                    |_| invalid("numeric frequency value is truncated"),
7681                                )?)),
7682                                _ => return Err(invalid("numeric frequency value tag differs")),
7683                            };
7684                            if values.iter().any(|(held, _)| *held == value) {
7685                                return Err(invalid("numeric frequency value repeats"));
7686                            }
7687                            let count = cur.u64()?;
7688                            total = total
7689                                .checked_add(count)
7690                                .ok_or_else(|| invalid("numeric frequency count overflows"))?;
7691                            values.push((value, count));
7692                        }
7693                        if total != entry.rows as u64 {
7694                            return Err(invalid("numeric frequencies do not cover table rows"));
7695                        }
7696                        Some(values)
7697                    }
7698                    _ => return Err(invalid("numeric frequency tag or type differs")),
7699                };
7700            }
7701        }
7702    }
7703    if !cur.done() {
7704        return Err(invalid("catalog has trailing bytes"));
7705    }
7706    Ok((entries, views))
7707}
7708
7709/// Reads the fields of a directory or a catalog in order, off bytes in memory or out of the file.
7710///
7711/// A catalog is small and is read whole. A table directory is not: at ten million rows of `hits` it
7712/// is nearly a megabyte, and holding that buffer while the table it describes is built out of it
7713/// put both at the peak of every query. Out of the file, the cursor holds one window of
7714/// [`DIRECTORY_WINDOW`] bytes and moves it forward as the fields are read, so what a directory
7715/// costs at open is what it decodes into and not that plus its own bytes.
7716struct Cursor<'a> {
7717    bytes: &'a [u8],
7718    at: usize,
7719    window: Option<Window<'a>>,
7720}
7721
7722/// The part of a directory in the file that a [`Cursor`] has read in.
7723struct Window<'a> {
7724    file: &'a File,
7725    offset: u64,
7726    length: usize,
7727    /// Where `held` starts, counted from the start of the directory.
7728    start: usize,
7729    held: Vec<u8>,
7730    /// How much to read at once, which is [`DIRECTORY_WINDOW`] outside the tests.
7731    size: usize,
7732}
7733
7734/// How much of a directory a cursor reading one out of the file holds at once.
7735const DIRECTORY_WINDOW: usize = 64 << 10;
7736
7737impl<'a> Cursor<'a> {
7738    fn new(bytes: &'a [u8]) -> Self {
7739        Self { bytes, at: 0, window: None }
7740    }
7741
7742    /// A cursor over `length` bytes of `file` from `offset`, which it reads a window at a time.
7743    fn over(file: &'a File, offset: u64, length: usize) -> Self {
7744        let window =
7745            Window { file, offset, length, start: 0, held: Vec::new(), size: DIRECTORY_WINDOW };
7746        Self { bytes: &[], at: 0, window: Some(window) }
7747    }
7748
7749    /// How many bytes the cursor walks in all.
7750    fn len(&self) -> usize {
7751        self.window.as_ref().map_or(self.bytes.len(), |window| window.length)
7752    }
7753
7754    /// Makes sure the next `len` bytes are in memory.
7755    fn ensure(&mut self, len: usize) -> Result<()> {
7756        let end = self.at.checked_add(len).ok_or_else(|| invalid("directory offset overflow"))?;
7757        if end > self.len() {
7758            return Err(invalid("directory is truncated"));
7759        }
7760        let Some(window) = &mut self.window else { return Ok(()) };
7761        if self.at < window.start || end > window.start + window.held.len() {
7762            let want = len.max(window.size).min(window.length - self.at);
7763            window.start = self.at;
7764            window.held.resize(want, 0);
7765            read_at(window.file, window.offset + self.at as u64, &mut window.held)?;
7766        }
7767        Ok(())
7768    }
7769
7770    /// `len` bytes from `at`, which [`Self::ensure`] has already brought in.
7771    fn held(&self, at: usize, len: usize) -> &[u8] {
7772        match &self.window {
7773            Some(window) => &window.held[at - window.start..at - window.start + len],
7774            None => &self.bytes[at..at + len],
7775        }
7776    }
7777
7778    /// The next `len` bytes, without moving past them.
7779    #[inline]
7780    fn peek(&mut self, len: usize) -> Result<&[u8]> {
7781        if self.window.is_none() {
7782            let bytes = self.bytes;
7783            return Ok(&bytes[self.at..self.end(len)?]);
7784        }
7785        self.ensure(len)?;
7786        Ok(self.held(self.at, len))
7787    }
7788
7789    /// The next `len` bytes, moving past them.
7790    ///
7791    /// Every data page is decoded through this, a byte or a word at a time, so a cursor over bytes
7792    /// already in memory takes them here and never reaches [`Self::ensure`]. With the window check
7793    /// on every call, q06 on TPC-H spent a seventh of its instructions in it.
7794    #[inline]
7795    fn take(&mut self, len: usize) -> Result<&[u8]> {
7796        if self.window.is_none() {
7797            let bytes = self.bytes;
7798            let (at, end) = (self.at, self.end(len)?);
7799            self.at = end;
7800            return Ok(&bytes[at..end]);
7801        }
7802        self.take_windowed(len)
7803    }
7804
7805    /// Moves over a checked field without reading its payload from a windowed directory.
7806    fn skip(&mut self, len: usize) -> Result<()> {
7807        let end = self.at.checked_add(len).ok_or_else(|| invalid("directory offset overflow"))?;
7808        if end > self.len() {
7809            return Err(invalid("directory is truncated"));
7810        }
7811        self.at = end;
7812        Ok(())
7813    }
7814
7815    fn skip_bound(&mut self) -> Result<()> {
7816        match self.u8()? {
7817            0 => Ok(()),
7818            1 => self.skip(16),
7819            2 => self.skip(8),
7820            3 => {
7821                let length = self.u32()? as usize;
7822                self.skip(length)
7823            }
7824            4 => self.skip(17),
7825            _ => Err(invalid("a stored bound has an unknown tag")),
7826        }
7827    }
7828
7829    /// Where `len` bytes from here end, when they end inside the bytes.
7830    #[inline]
7831    fn end(&self, len: usize) -> Result<usize> {
7832        let end = self.at.checked_add(len).ok_or_else(|| invalid("directory offset overflow"))?;
7833        if end > self.bytes.len() {
7834            return Err(invalid("directory is truncated"));
7835        }
7836        Ok(end)
7837    }
7838
7839    /// [`Self::take`] out of the file, a window at a time.
7840    #[inline(never)]
7841    fn take_windowed(&mut self, len: usize) -> Result<&[u8]> {
7842        self.ensure(len)?;
7843        self.at += len;
7844        Ok(self.held(self.at - len, len))
7845    }
7846    #[inline]
7847    fn u8(&mut self) -> Result<u8> {
7848        Ok(self.take(1)?[0])
7849    }
7850    #[inline]
7851    fn u16(&mut self) -> Result<u16> {
7852        Ok(u16::from_le_bytes(self.take(2)?.try_into().expect("two bytes")))
7853    }
7854    #[inline]
7855    fn u32(&mut self) -> Result<u32> {
7856        Ok(u32::from_le_bytes(self.take(4)?.try_into().expect("four bytes")))
7857    }
7858    #[inline]
7859    fn u64(&mut self) -> Result<u64> {
7860        Ok(u64::from_le_bytes(self.take(8)?.try_into().expect("eight bytes")))
7861    }
7862    fn var_u64(&mut self) -> Result<u64> {
7863        let mut value = 0_u64;
7864        for shift in (0..=63).step_by(7) {
7865            let byte = self.u8()?;
7866            let part = u64::from(byte & 0x7f);
7867            if shift == 63 && part > 1 {
7868                return Err(invalid("frequency ordinal varint overflows"));
7869            }
7870            value |= part << shift;
7871            if byte & 0x80 == 0 {
7872                return Ok(value);
7873            }
7874        }
7875        Err(invalid("frequency ordinal varint is too long"))
7876    }
7877    /// A zone map's end, in the layout `rudb_common::bounds` defines.
7878    ///
7879    /// The bytes are the ones this directory has written since format 10 and the codec moved to
7880    /// rank zero rather than being copied, because a column summary now writes the same two ends
7881    /// and two encodings of one type is how the two quietly stop agreeing.
7882    ///
7883    /// A bound's length is in the bound, so out of the file the cursor offers the codec a few bytes
7884    /// and offers it twice as many whenever it runs out before the directory does.
7885    fn bound(&mut self) -> Result<Option<Bound>> {
7886        let rest = self.len().saturating_sub(self.at);
7887        let mut want = 32;
7888        loop {
7889            let offered = self.peek(want.min(rest))?;
7890            let mut used = 0;
7891            match bounds::get(offered, &mut used) {
7892                Ok(bound) => {
7893                    self.at += used;
7894                    return Ok(bound);
7895                }
7896                Err(_) if want < rest => want *= 2,
7897                Err(error) => return Err(error),
7898            }
7899        }
7900    }
7901    fn text(&mut self) -> Result<String> {
7902        let len = self.u16()? as usize;
7903        String::from_utf8(self.take(len)?.to_vec()).map_err(|_| invalid("name is not UTF-8"))
7904    }
7905    /// Whether everything has been read, which is how a section that an older file does not have at
7906    /// all is told from one that is there and empty.
7907    fn done(&self) -> bool {
7908        self.at >= self.len()
7909    }
7910    /// The same, for text that is a query rather than a name.
7911    ///
7912    /// A name fits in sixteen bits of length and a view body does not have to. Nobody writes a 64
7913    /// kilobyte identifier by accident and people do write generated queries that long, and a view
7914    /// that could not be written down because its body was too big would be a limit invented here
7915    /// rather than one anything else in the engine has.
7916    fn long_text(&mut self) -> Result<String> {
7917        let len = self.u32()? as usize;
7918        String::from_utf8(self.take(len)?.to_vec()).map_err(|_| invalid("text is not UTF-8"))
7919    }
7920}
7921
7922/// One column's frequency synopsis, or `None` for a column that has none, checked against the column.
7923fn decode_summary(
7924    cur: &mut Cursor<'_>,
7925    field: &Field,
7926    rows: usize,
7927    values: bool,
7928) -> Result<Option<FrequencySummary>> {
7929    Ok(match cur.u8()? {
7930        0 => None,
7931        1 => {
7932            let omitted_max = cur.u64()?;
7933            let count = cur.u32()? as usize;
7934            if count > FREQUENCY_ENTRIES {
7935                return Err(invalid("frequency entry count exceeds its bound"));
7936            }
7937            let mut entries = Vec::with_capacity(count);
7938            // row at a time: directory decoding validates each persisted bounded frequency entry.
7939            for _ in 0..count {
7940                let value = match cur.u8()? {
7941                    0 => FrequencyValue::Null,
7942                    1 => FrequencyValue::Integer(i128::from_le_bytes(
7943                        cur.take(16)?.try_into().expect("sixteen bytes"),
7944                    )),
7945                    2 => FrequencyValue::Code(cur.u32()?),
7946                    _ => return Err(invalid("frequency value tag differs")),
7947                };
7948                let valid = matches!(
7949                    (&field.ty, value),
7950                    (_, FrequencyValue::Null)
7951                        | (LogicalType::Varchar, FrequencyValue::Code(_))
7952                        | (
7953                            LogicalType::TinyInt
7954                                | LogicalType::SmallInt
7955                                | LogicalType::Integer
7956                                | LogicalType::BigInt
7957                                | LogicalType::UTinyInt
7958                                | LogicalType::USmallInt
7959                                | LogicalType::UInteger
7960                                | LogicalType::UBigInt
7961                                | LogicalType::Date
7962                                | LogicalType::Timestamp,
7963                            FrequencyValue::Integer(_),
7964                        )
7965                );
7966                if !valid {
7967                    return Err(invalid("frequency value does not match its column"));
7968                }
7969                let count = cur.u64()?;
7970                if count == 0 || count > rows as u64 {
7971                    return Err(invalid("frequency count is outside the table"));
7972                }
7973                entries.push(FrequencyEntry { value, count });
7974            }
7975            if entries.windows(2).any(|pair| pair[0].count < pair[1].count) {
7976                return Err(invalid("frequency entries are not descending"));
7977            }
7978            let ordinals = {
7979                let ordinal_count = cur.u32()? as usize;
7980                if ordinal_count > FREQUENCY_ORDINALS || ordinal_count > rows {
7981                    return Err(invalid("frequency ordinal count exceeds its bound"));
7982                }
7983                let mut ordinals = Vec::with_capacity(ordinal_count);
7984                let mut previous = 0_u64;
7985                for at in 0..ordinal_count {
7986                    let delta = cur.var_u64()?;
7987                    if at != 0 && delta == 0 {
7988                        return Err(invalid("frequency ordinals are not increasing"));
7989                    }
7990                    let ordinal = if at == 0 {
7991                        delta
7992                    } else {
7993                        previous
7994                            .checked_add(delta)
7995                            .ok_or_else(|| invalid("frequency ordinal overflows"))?
7996                    };
7997                    if ordinal >= rows as u64 {
7998                        return Err(invalid("frequency ordinal is outside the table"));
7999                    }
8000                    ordinals.push(ordinal);
8001                    previous = ordinal;
8002                }
8003                ordinals
8004            };
8005            let ordinal_entries = if values {
8006                let mut ordinal_entries = Vec::with_capacity(ordinals.len());
8007                for _ in 0..ordinals.len() {
8008                    let entry = cur.u16()?;
8009                    if entry as usize >= entries.len() {
8010                        return Err(invalid("frequency ordinal value is outside its entries"));
8011                    }
8012                    ordinal_entries.push(entry);
8013                }
8014                ordinal_entries
8015            } else {
8016                Vec::new()
8017            };
8018            Some(FrequencySummary { entries, omitted_max, ordinals, ordinal_entries })
8019        }
8020        _ => return Err(invalid("frequency summary tag differs")),
8021    })
8022}
8023
8024/// Skips a synopsis whose column the caller does not need. The directory checksum was checked
8025/// before this walk, and the fields still need their lengths and tags checked to find the next one.
8026fn skip_summary(cur: &mut Cursor<'_>, values: bool, rows: usize) -> Result<()> {
8027    match cur.u8()? {
8028        0 => Ok(()),
8029        1 => {
8030            cur.skip(8)?;
8031            let entries = cur.u32()? as usize;
8032            if entries > FREQUENCY_ENTRIES {
8033                return Err(invalid("frequency entry count exceeds its bound"));
8034            }
8035            for _ in 0..entries {
8036                match cur.u8()? {
8037                    0 => {}
8038                    1 => cur.skip(16)?,
8039                    2 => cur.skip(4)?,
8040                    _ => return Err(invalid("frequency value tag differs")),
8041                }
8042                cur.skip(8)?;
8043            }
8044            let ordinals = cur.u32()? as usize;
8045            if ordinals > FREQUENCY_ORDINALS || ordinals > rows {
8046                return Err(invalid("frequency ordinal count exceeds its bound"));
8047            }
8048            for _ in 0..ordinals {
8049                cur.var_u64()?;
8050            }
8051            if values {
8052                cur.skip(ordinals * 2)?;
8053            }
8054            Ok(())
8055        }
8056        _ => Err(invalid("frequency summary tag differs")),
8057    }
8058}
8059
8060/// Reads only the catalog, stripe null counts, and one frequency synopsis. This is the cold path
8061/// for a summary-backed count; constructing every page descriptor and zone map would make it cost
8062/// the size of the table directory even when no row is read.
8063fn quick_nonzero(
8064    mut cur: Cursor<'_>,
8065    name: &str,
8066    fields: &[Field],
8067    rows: usize,
8068    wanted: usize,
8069) -> Result<Option<u64>> {
8070    if cur.take(8)? != DIRECTORY || cur.text()? != name {
8071        return Err(invalid("table directory differs from the catalog"));
8072    }
8073    let width = cur.u16()? as usize;
8074    if width != fields.len() {
8075        return Err(invalid("table directory width differs from the catalog"));
8076    }
8077    for field in fields {
8078        let stored =
8079            Field { name: cur.text()?, ty: read_type(&mut cur)?, not_null: cur.u8()? != 0 };
8080        if &stored != field {
8081            return Err(invalid("table directory schema differs from the catalog"));
8082        }
8083    }
8084    let mut dictionaries = Vec::with_capacity(width);
8085    for _ in 0..width {
8086        let held = match cur.u8()? {
8087            0 => false,
8088            1 => {
8089                cur.skip(20)?;
8090                true
8091            }
8092            _ => return Err(invalid("dictionary page tag differs")),
8093        };
8094        dictionaries.push(held);
8095    }
8096    for _ in 0..width {
8097        match cur.u8()? {
8098            0 => {}
8099            1 => cur.skip(8)?,
8100            _ => return Err(invalid("distinct count tag differs")),
8101        }
8102    }
8103    if cur.u64()? != rows as u64 {
8104        return Err(invalid("table row count differs from the catalog"));
8105    }
8106    let stripes = cur.u32()? as usize;
8107    let mut total = 0_usize;
8108    let mut nulls = 0_u64;
8109    for _ in 0..stripes {
8110        let parts = cur.u32()? as usize;
8111        if parts == 0 || parts > STRIPE_PARTS {
8112            return Err(invalid("stripe part count is outside its bound"));
8113        }
8114        let mut stripe_rows = 0_usize;
8115        for _ in 0..parts {
8116            stripe_rows = stripe_rows
8117                .checked_add(cur.u32()? as usize)
8118                .ok_or_else(|| invalid("stripe row count overflow"))?;
8119        }
8120        total =
8121            total.checked_add(stripe_rows).ok_or_else(|| invalid("stripe row count overflow"))?;
8122        cur.skip(12 + width * 12)?;
8123        for (field, held) in fields.iter().zip(&dictionaries) {
8124            if field.ty == LogicalType::Varchar && *held {
8125                cur.skip(20)?;
8126            }
8127        }
8128        for _ in 0..width * 2 {
8129            match cur.u8()? {
8130                0 => {}
8131                1 => cur.skip(20)?,
8132                _ => return Err(invalid("stripe page tag differs")),
8133            }
8134        }
8135        for column in 0..width {
8136            cur.skip_bound()?;
8137            cur.skip_bound()?;
8138            let count = cur.u32()? as u64;
8139            if count > stripe_rows as u64 {
8140                return Err(invalid("null count exceeds stripe rows"));
8141            }
8142            if column == wanted {
8143                nulls = nulls.checked_add(count).ok_or_else(|| invalid("null count overflow"))?;
8144            }
8145            cur.skip(1)?;
8146            match cur.u8()? {
8147                0 => {}
8148                1 => cur.skip(16)?,
8149                _ => return Err(invalid("a stripe sum has an unknown tag")),
8150            }
8151        }
8152    }
8153    if total != rows {
8154        return Err(invalid("table row count differs from stripes"));
8155    }
8156    if cur.done() {
8157        return Ok(None);
8158    }
8159    let magic = cur.take(8)?;
8160    let values = magic == FREQUENCIES;
8161    if !values && magic != FREQUENCIES_V2 {
8162        return Err(invalid("directory extension magic differs"));
8163    }
8164    if cur.u16()? as usize != width {
8165        return Err(invalid("frequency column count differs"));
8166    }
8167    for _ in 0..wanted {
8168        skip_summary(&mut cur, values, rows)?;
8169    }
8170    let Some(summary) = decode_summary(&mut cur, &fields[wanted], rows, values)? else {
8171        return Ok(None);
8172    };
8173    let zero = summary
8174        .entries
8175        .iter()
8176        .find(|entry| entry.value == FrequencyValue::Integer(0))
8177        .map(|entry| entry.count)
8178        .or_else(|| (summary.omitted_max == 0).then_some(0));
8179    Ok(zero.and_then(|zero| (rows as u64).checked_sub(nulls)?.checked_sub(zero)))
8180}
8181
8182fn decode_directory(bytes: &[u8], size: u64) -> Result<Table> {
8183    read_directory(Cursor::new(bytes), size, None)
8184}
8185
8186/// A directory out of `cur`, which is a whole one in memory or one being read out of the file.
8187///
8188/// `stored_at` is where the directory starts in the file when it is being read out of it, and then
8189/// every frequency synopsis is checked and left there, as [`Frequencies::Stored`].
8190fn read_directory(mut cur: Cursor<'_>, size: u64, stored_at: Option<u64>) -> Result<Table> {
8191    if cur.take(8)? != DIRECTORY {
8192        return Err(invalid("directory magic differs"));
8193    }
8194    let name = cur.text()?;
8195    let width = cur.u16()? as usize;
8196    let mut fields = Vec::with_capacity(width);
8197    for _ in 0..width {
8198        let name = cur.text()?;
8199        let ty = read_type(&mut cur)?;
8200        let not_null = match cur.u8()? {
8201            0 => false,
8202            1 => true,
8203            _ => return Err(invalid("nullability flag differs")),
8204        };
8205        fields.push(Field { name, ty, not_null });
8206    }
8207    let mut dictionaries = Vec::with_capacity(width);
8208    for _ in 0..width {
8209        dictionaries.push(match cur.u8()? {
8210            0 => None,
8211            1 => {
8212                let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
8213                let end = page
8214                    .offset
8215                    .checked_add(u64::from(page.length))
8216                    .ok_or_else(|| invalid("dictionary page offset overflow"))?;
8217                // A global dictionary covers a whole column, not one bounded stripe. Its lazy
8218                // payload is intentionally allowed to grow past `MAX_PAGE`; only ordinary column
8219                // pages are capped there. `Writer::finish` has already bounded this length by the
8220                // on-disk `u32`, and the range check below keeps it inside the file.
8221                if page.offset < HEADER || end > size {
8222                    return Err(invalid("dictionary page range is outside the file"));
8223                }
8224                Some(page)
8225            }
8226            _ => return Err(invalid("dictionary page tag differs")),
8227        });
8228    }
8229    let mut distincts = Vec::with_capacity(width);
8230    for _ in 0..width {
8231        distincts.push(match cur.u8()? {
8232            0 => None,
8233            1 => Some(cur.u64()?),
8234            _ => return Err(invalid("distinct count tag differs")),
8235        });
8236    }
8237    let rows = usize::try_from(cur.u64()?).map_err(|_| invalid("row count does not fit"))?;
8238    let count = cur.u32()? as usize;
8239    let mut stripes = Vec::with_capacity(count);
8240    let mut total = 0_usize;
8241    for _ in 0..count {
8242        let count = cur.u32()? as usize;
8243        if count == 0 || count > STRIPE_PARTS {
8244            return Err(invalid("stripe part count is outside its bound"));
8245        }
8246        let mut parts = Vec::with_capacity(count);
8247        let mut stripe_rows = 0_usize;
8248        for _ in 0..count {
8249            let rows = cur.u32()?;
8250            if rows == 0 {
8251                return Err(invalid("empty part"));
8252            }
8253            parts.push(rows);
8254            stripe_rows = stripe_rows
8255                .checked_add(rows as usize)
8256                .ok_or_else(|| invalid("stripe row count overflow"))?;
8257        }
8258        total =
8259            total.checked_add(stripe_rows).ok_or_else(|| invalid("stripe row count overflow"))?;
8260        let index = Span { offset: cur.u64()?, length: cur.u32()? };
8261        let section = index_section(count)?;
8262        let wanted = section
8263            .checked_mul(width)
8264            .and_then(|bytes| u32::try_from(bytes).ok())
8265            .ok_or_else(|| invalid("index page length overflow"))?;
8266        let end = index
8267            .offset
8268            .checked_add(u64::from(index.length))
8269            .ok_or_else(|| invalid("index page offset overflow"))?;
8270        if index.offset < HEADER || end > size || index.length != wanted {
8271            return Err(invalid("index page range is outside the file"));
8272        }
8273        let mut pages = Vec::with_capacity(width);
8274        for _ in 0..width {
8275            let offset = cur.u64()?;
8276            let length = cur.u32()?;
8277            let end = offset
8278                .checked_add(u64::from(length))
8279                .ok_or_else(|| invalid("page offset overflow"))?;
8280            if offset < HEADER || end > size || length as usize > MAX_PAGE {
8281                return Err(invalid("page range is outside the file"));
8282            }
8283            pages.push(Span { offset, length });
8284        }
8285        let mut memberships = vec![None; width];
8286        for (column, field) in fields.iter().enumerate() {
8287            if field.ty != LogicalType::Varchar || dictionaries[column].is_none() {
8288                continue;
8289            }
8290            let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
8291            let end = page
8292                .offset
8293                .checked_add(u64::from(page.length))
8294                .ok_or_else(|| invalid("membership page offset overflow"))?;
8295            if page.offset < HEADER || end > size || page.length as usize > MAX_PAGE {
8296                return Err(invalid("membership page range is outside the file"));
8297            }
8298            memberships[column] = Some(page);
8299        }
8300        let mut sieves = vec![None; width];
8301        for sieve in sieves.iter_mut().take(width) {
8302            match cur.u8()? {
8303                0 => continue,
8304                1 => {}
8305                _ => return Err(invalid("a sieve page has an unknown tag")),
8306            }
8307            let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
8308            let end = page
8309                .offset
8310                .checked_add(u64::from(page.length))
8311                .ok_or_else(|| invalid("sieve page offset overflow"))?;
8312            if page.offset < HEADER || end > size || page.length as usize > MAX_PAGE {
8313                return Err(invalid("sieve page range is outside the file"));
8314            }
8315            *sieve = Some(page);
8316        }
8317        let mut part_ranges = vec![None; width];
8318        for held in part_ranges.iter_mut().take(width) {
8319            match cur.u8()? {
8320                0 => continue,
8321                1 => {}
8322                _ => return Err(invalid("a part range page has an unknown tag")),
8323            }
8324            let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
8325            let end = page
8326                .offset
8327                .checked_add(u64::from(page.length))
8328                .ok_or_else(|| invalid("part range page offset overflow"))?;
8329            if page.offset < HEADER || end > size || page.length as usize > MAX_PAGE {
8330                return Err(invalid("part range page range is outside the file"));
8331            }
8332            *held = Some(page);
8333        }
8334        let mut ranges = Vec::with_capacity(width);
8335        for column in 0..width {
8336            let low = cur.bound()?;
8337            let high = cur.bound()?;
8338            let nulls = cur.u32()? as usize;
8339            if nulls > stripe_rows {
8340                return Err(invalid("null count exceeds stripe rows"));
8341            }
8342            let exact = cur.u8()? != 0;
8343            let sum = match cur.u8()? {
8344                0 => None,
8345                1 => Some(i128::from_le_bytes(
8346                    cur.take(16)?.try_into().map_err(|_| invalid("a stripe sum is truncated"))?,
8347                )),
8348                _ => return Err(invalid("a stripe sum has an unknown tag")),
8349            };
8350            // Files written before the ends of a decimal or a timestamp column carried their power
8351            // of ten hold a bare integer here, and that integer is the one the column holds, which
8352            // is what the power is over. So the type puts it back on the way in and an old file
8353            // prunes as well as a new one. A file that already wrote the power keeps it, because
8354            // this leaves anything that is not an integer alone.
8355            let ty = &fields.get(column).ok_or_else(|| invalid("a stripe range has no column"))?.ty;
8356            let low = low.map(|bound| scaled_as(bound, ty));
8357            let high = high.map(|bound| scaled_as(bound, ty));
8358            ranges.push(Range { low, high, nulls, exact, sum });
8359        }
8360        stripes.push(Stripe {
8361            rows: stripe_rows,
8362            parts,
8363            index,
8364            pages,
8365            memberships: Pages::from_slots(memberships)?,
8366            sieves: Pages::from_slots(sieves)?,
8367            part_ranges: Pages::from_slots(part_ranges)?,
8368            zone: Zone::from_ranges(ranges),
8369        });
8370    }
8371    if total != rows {
8372        return Err(invalid("table row count differs from stripes"));
8373    }
8374    // How many entries each column's synopsis lists, which is all a pair summary is checked against,
8375    // kept apart because the synopses themselves may be left in the file.
8376    let mut entry_counts = vec![0; width];
8377    let frequencies = if cur.done() {
8378        vec![None; width]
8379    } else {
8380        let frequency_magic = cur.take(8)?;
8381        let frequency_values = frequency_magic == FREQUENCIES;
8382        if !frequency_values && frequency_magic != FREQUENCIES_V2 {
8383            return Err(invalid("directory extension magic differs"));
8384        }
8385        if cur.u16()? as usize != width {
8386            return Err(invalid("frequency column count differs"));
8387        }
8388        let mut frequencies = Vec::with_capacity(width);
8389        for (field, entry_count) in fields.iter().zip(&mut entry_counts) {
8390            let start = cur.at;
8391            let summary = decode_summary(&mut cur, field, rows, frequency_values)?;
8392            *entry_count = summary.as_ref().map_or(0, |summary| summary.entries.len());
8393            frequencies.push(match (summary, stored_at) {
8394                (None, _) => None,
8395                (Some(summary), None) => Some(Frequencies::Held(summary)),
8396                (Some(_), Some(offset)) => Some(Frequencies::Stored {
8397                    span: Span {
8398                        offset: offset + start as u64,
8399                        length: u32::try_from(cur.at - start)
8400                            .map_err(|_| invalid("a frequency synopsis is too long"))?,
8401                    },
8402                    values: frequency_values,
8403                }),
8404            });
8405        }
8406        frequencies
8407    };
8408    // There are two optional trailing blocks now rather than one, so the reader dispatches on the
8409    // magic it finds rather than on where the bytes ran out. That is what lets the two arrive
8410    // independently: a format 22 directory ends here and has neither, a directory written before
8411    // the section table has only the clustering declaration, and each one still opens without a
8412    // rewrite. It is the G1 exit criterion, which is that a reader that knows about sections opens
8413    // a file that predates them and answers every query, only without the graph path.
8414    //
8415    // A repeated block is refused rather than allowed to win, because two clustering declarations
8416    // in one directory is a torn directory and the only question is which of them is the lie.
8417    let mut clustering = None;
8418    let mut sections = Vec::new();
8419    let mut pair_frequencies = Vec::new();
8420    let mut seen_pair_frequencies = false;
8421    let mut frequency_texts = vec![Vec::new(); width];
8422    let mut seen_frequency_texts = false;
8423    let mut host_groups = None;
8424    let mut seen_sections = false;
8425    let mut dictionary_payloads = Vec::new();
8426    let mut seen_payloads = false;
8427    // Zero until a section table says otherwise, which is what a format 22 table gets and what
8428    // makes every section stamp fail to match on one, because real generations start at one.
8429    let mut generation = 0;
8430    while !cur.done() {
8431        let mut tag = [0u8; 8];
8432        tag.copy_from_slice(cur.take(8)?);
8433        if &tag == PAIR_FREQUENCIES {
8434            if seen_pair_frequencies {
8435                return Err(invalid("directory names two pair frequency blocks"));
8436            }
8437            seen_pair_frequencies = true;
8438            let count = cur.u16()? as usize;
8439            if count > MAX_PAIR_FREQUENCIES {
8440                return Err(invalid("pair frequency count exceeds its bound"));
8441            }
8442            pair_frequencies = Vec::with_capacity(count);
8443            for _ in 0..count {
8444                let first = cur.u16()?;
8445                let second = cur.u16()?;
8446                let first_at = first as usize;
8447                let second_at = second as usize;
8448                if frequencies.get(first_at).and_then(Option::as_ref).is_none() {
8449                    return Err(invalid("pair frequency first column has no synopsis"));
8450                }
8451                let first_entries = entry_counts[first_at];
8452                if !matches!(fields.get(second_at), Some(field) if field.ty == LogicalType::Varchar)
8453                    || dictionaries.get(second_at).copied().flatten().is_none()
8454                {
8455                    return Err(invalid("pair frequency second column has no stable dictionary"));
8456                }
8457                if pair_frequencies
8458                    .iter()
8459                    .any(|held: &PairFrequencySummary| held.first == first && held.second == second)
8460                {
8461                    return Err(invalid("directory repeats a pair frequency summary"));
8462                }
8463                let omitted_max = cur.u64()?;
8464                if omitted_max > rows as u64 {
8465                    return Err(invalid("pair frequency omitted count exceeds the table"));
8466                }
8467                let entries_count = cur.u16()? as usize;
8468                if entries_count > FREQUENCY_ENTRIES {
8469                    return Err(invalid("pair frequency entry count exceeds its bound"));
8470                }
8471                let mut entries = Vec::with_capacity(entries_count);
8472                for _ in 0..entries_count {
8473                    let first_entry = cur.u16()?;
8474                    if first_entry as usize >= first_entries {
8475                        return Err(invalid("pair frequency anchor is outside its synopsis"));
8476                    }
8477                    let second = match cur.u8()? {
8478                        0 => None,
8479                        1 => Some(cur.u32()?),
8480                        _ => return Err(invalid("pair frequency string tag differs")),
8481                    };
8482                    let count = cur.u64()?;
8483                    if count == 0 || count > rows as u64 {
8484                        return Err(invalid("pair frequency count is outside the table"));
8485                    }
8486                    entries.push(PairFrequencyEntry { first_entry, second, count });
8487                }
8488                if entries.windows(2).any(|pair| pair[0].count < pair[1].count) {
8489                    return Err(invalid("pair frequency entries are not descending"));
8490                }
8491                pair_frequencies.push(PairFrequencySummary { first, second, entries, omitted_max });
8492            }
8493        } else if &tag == FREQUENCY_TEXTS {
8494            if seen_frequency_texts {
8495                return Err(invalid("directory names two frequency text blocks"));
8496            }
8497            seen_frequency_texts = true;
8498            let columns = cur.u16()? as usize;
8499            if columns > width {
8500                return Err(invalid("frequency text column count exceeds the schema"));
8501            }
8502            for _ in 0..columns {
8503                let column = cur.u16()? as usize;
8504                if !frequency_texts.get(column).is_some_and(Vec::is_empty) {
8505                    return Err(invalid("frequency text column is repeated or out of range"));
8506                }
8507                if !matches!(fields.get(column), Some(field) if field.ty == LogicalType::Varchar)
8508                    || dictionaries.get(column).copied().flatten().is_none()
8509                    || frequencies.get(column).and_then(Option::as_ref).is_none()
8510                {
8511                    return Err(invalid("frequency texts belong to a non-string synopsis"));
8512                }
8513                let count = cur.u16()? as usize;
8514                if count == 0 || count != entry_counts[column] {
8515                    return Err(invalid("frequency text count differs from its synopsis"));
8516                }
8517                let mut texts = Vec::with_capacity(count);
8518                for _ in 0..count {
8519                    texts.push(match cur.u8()? {
8520                        0 => None,
8521                        1 => {
8522                            let length = cur.u32()? as usize;
8523                            let bytes = cur.take(length)?.to_vec();
8524                            std::str::from_utf8(&bytes)
8525                                .map_err(|_| invalid("frequency text is not UTF-8"))?;
8526                            Some(bytes)
8527                        }
8528                        _ => return Err(invalid("frequency text tag differs")),
8529                    });
8530                }
8531                frequency_texts[column] = texts;
8532            }
8533        } else if &tag == HOST_GROUPS {
8534            if host_groups.is_some() {
8535                return Err(invalid("directory names two host group blocks"));
8536            }
8537            let column = cur.u16()? as usize;
8538            if !matches!(fields.get(column), Some(field) if field.ty == LogicalType::Varchar)
8539                || dictionaries.get(column).copied().flatten().is_none()
8540            {
8541                return Err(invalid("host groups belong to a non-string dictionary"));
8542            }
8543            let omitted_max = cur.u64()?;
8544            if omitted_max > rows as u64 {
8545                return Err(invalid("host group bound exceeds the table"));
8546            }
8547            let count = cur.u16()? as usize;
8548            if count > host::CAPACITY {
8549                return Err(invalid("host group count exceeds its bound"));
8550            }
8551            let mut entries = Vec::with_capacity(count);
8552            let mut bytes = 0_usize;
8553            for _ in 0..count {
8554                let host_len = cur.u32()? as usize;
8555                bytes =
8556                    bytes.checked_add(host_len).ok_or_else(|| invalid("host bytes overflow"))?;
8557                if bytes > host::BYTE_BUDGET {
8558                    return Err(invalid("host groups exceed their byte budget"));
8559                }
8560                let host = std::str::from_utf8(cur.take(host_len)?)
8561                    .map_err(|_| invalid("host is not UTF-8"))?
8562                    .to_owned();
8563                let count = cur.u64()?;
8564                if count == 0 || count > rows as u64 {
8565                    return Err(invalid("host group count exceeds the table"));
8566                }
8567                let bytes_sum = i128::from_le_bytes(
8568                    cur.take(16)?
8569                        .try_into()
8570                        .map_err(|_| invalid("host length sum is truncated"))?,
8571                );
8572                if bytes_sum < 0 {
8573                    return Err(invalid("host length sum is negative"));
8574                }
8575                let minimum_len = cur.u32()? as usize;
8576                bytes =
8577                    bytes.checked_add(minimum_len).ok_or_else(|| invalid("host bytes overflow"))?;
8578                if bytes > host::BYTE_BUDGET {
8579                    return Err(invalid("host groups exceed their byte budget"));
8580                }
8581                let minimum = std::str::from_utf8(cur.take(minimum_len)?)
8582                    .map_err(|_| invalid("host minimum is not UTF-8"))?
8583                    .to_owned();
8584                entries.push(host::HostEntry { host, count, bytes_sum, minimum });
8585            }
8586            if entries.windows(2).any(|pair| pair[0].count < pair[1].count)
8587                || entries.iter().any(|entry| entry.host.is_empty() || entry.minimum.is_empty())
8588            {
8589                return Err(invalid("host groups are not in certified order"));
8590            }
8591            host_groups = Some(host::HostSummary { column, omitted_max, entries });
8592        } else if &tag == CLUSTERING {
8593            if clustering.is_some() {
8594                return Err(invalid("directory names two clustering declarations"));
8595            }
8596            let bucket = Width::from_tag(cur.u8()?)
8597                .ok_or_else(|| invalid("clustering width tag differs"))?;
8598            let count = cur.u16()? as usize;
8599            let mut columns = Vec::with_capacity(count.min(fields.len()));
8600            for _ in 0..count {
8601                columns.push(u32::from(cur.u16()?));
8602            }
8603            // Through the constructor and not built by hand, so that a file claiming a column the
8604            // table does not have is caught at open rather than at the first scan that trusted it.
8605            clustering = Some(Clustering::new(columns, bucket, &fields).map_err(|_| {
8606                invalid("stored clustering declaration does not match the table it is on")
8607            })?);
8608        } else if &tag == SECTIONS {
8609            if seen_sections {
8610                return Err(invalid("directory names two section tables"));
8611            }
8612            seen_sections = true;
8613            generation = cur.u64()?;
8614            let count = cur.u16()? as usize;
8615            if count > MAX_SECTIONS {
8616                return Err(invalid("section count exceeds its bound"));
8617            }
8618            sections = Vec::with_capacity(count);
8619            // entry at a time: a malformed section entry is refused rather than turned into an
8620            // offset.
8621            for _ in 0..count {
8622                sections.push(Section::decode(cur.take(section::ENTRY_BYTES)?)?);
8623            }
8624            for held in &sections {
8625                let Some(end) = held.extent_page.checked_add(u64::from(held.extent_bytes)) else {
8626                    return Err(invalid("a section's extent table overflows the file"));
8627                };
8628                // The bound check is here and not in `section`, because only the caller knows how
8629                // big the file is. A section pointing past the end is a torn directory, and reading
8630                // the payload it names would be reading whatever else is at that offset.
8631                if held.extent_bytes != 0 && (held.extent_page < HEADER || end > size) {
8632                    return Err(invalid("a section's extent table is outside the file"));
8633                }
8634                if held.extents == 0 && held.extent_bytes != 0 {
8635                    return Err(invalid("a section with no extents names an extent table"));
8636                }
8637            }
8638        } else if &tag == DICTIONARY_PAYLOADS {
8639            if seen_payloads {
8640                return Err(invalid("directory names two dictionary payload blocks"));
8641            }
8642            seen_payloads = true;
8643            let count = cur.u16()? as usize;
8644            if count != fields.len() {
8645                return Err(invalid("dictionary payload block does not match the table's columns"));
8646            }
8647            dictionary_payloads = Vec::with_capacity(count);
8648            for _ in 0..count {
8649                let bytes = cur.u64()?;
8650                if bytes > size {
8651                    return Err(invalid("a dictionary payload is larger than the file"));
8652                }
8653                dictionary_payloads.push(bytes);
8654            }
8655        } else {
8656            return Err(invalid("directory extension magic differs"));
8657        }
8658    }
8659    if !cur.done() {
8660        return Err(invalid("directory has trailing bytes"));
8661    }
8662    Ok(Table {
8663        name,
8664        fields,
8665        stripes,
8666        rows,
8667        dictionaries,
8668        dictionary_payloads,
8669        distincts,
8670        frequencies,
8671        pair_frequencies,
8672        frequency_texts,
8673        host_groups,
8674        clustering,
8675        generation,
8676        sections,
8677    })
8678}
8679
8680/// A zone map's end, in the layout `rudb_common::bounds` defines. See [`Cursor::bound`].
8681fn put_bound(out: &mut Vec<u8>, bound: Option<&Bound>) -> Result<()> {
8682    bounds::put(out, bound)
8683}
8684
8685/// Which cascades are worth trying on a run of dictionary codes.
8686///
8687/// The exhaustive chooser encodes every candidate at every level of a cascade three deep and keeps
8688/// the smallest, which on a part of 1024 codes is around a hundred full encodes to decide something
8689/// three candidates were always going to win. It is the right default for a crate that does not
8690/// know what it is looking at. Here we do know. Codes are counted from zero in the order the values
8691/// were first seen, so a part of them is one value, or a narrow band, or a few long runs, and those
8692/// are constant, frame of reference and run length. Nothing else has ever come first on this data.
8693///
8694/// A dictionary of dictionary codes is the one candidate that can never pay, because the codes are
8695/// already the dictionary, and it is also the most expensive one to try. Below the top level the
8696/// streams are an RLE's run values and run lengths, which are integers in their own right with no
8697/// runs left in them, so only the two flat candidates go down there.
8698///
8699/// This is size given up for time on purpose, and the ablation is this chooser against
8700/// [`chooser::EXHAUSTIVE`] on the same file.
8701#[derive(Debug)]
8702struct Codes;
8703
8704impl chooser::Chooser for Codes {
8705    fn name(&self) -> &'static str {
8706        "codes"
8707    }
8708
8709    fn narrow_strings(
8710        &self,
8711        _values: &[&[u8]],
8712        offered: &[string::Kind],
8713        _depth: u8,
8714    ) -> Vec<string::Kind> {
8715        // Never reached, because nothing here encodes strings through the cascade. The trait asks
8716        // for it and the honest answer to a question we have no opinion on is the whole list.
8717        offered.to_vec()
8718    }
8719
8720    fn narrow_integers(
8721        &self,
8722        _values: &[i64],
8723        offered: &[integer::Kind],
8724        depth: u8,
8725    ) -> Vec<integer::Kind> {
8726        // The contract is a non empty subset, and a chunk that offers none of the three is a chunk
8727        // this has no opinion about rather than one that cannot be written.
8728        narrowed_to(Codes::keep(depth), offered)
8729    }
8730
8731    fn considers_integer(&self, kind: integer::Kind, depth: u8) -> bool {
8732        Codes::keep(depth).contains(&kind)
8733    }
8734}
8735
8736impl Codes {
8737    fn keep(depth: u8) -> &'static [integer::Kind] {
8738        if depth == 0 {
8739            &[integer::Kind::Constant, integer::Kind::Packed, integer::Kind::Rle]
8740        } else {
8741            &[integer::Kind::Constant, integer::Kind::Packed]
8742        }
8743    }
8744}
8745
8746/// The kinds of `offered` that are in `keep`, or all of `offered` when none of them are.
8747///
8748/// `Packed` applies to every chunk and both choosers keep it, so the fallback is never taken on a
8749/// chunk the cascade offers. It is there because the contract is a non empty subset and a chooser
8750/// that returned nothing would be a chunk that cannot be written. It is also why saying no to a kind
8751/// in `considers_integer` is safe: a kind that is never offered could only have been kept through
8752/// this fallback, and the fallback is never reached.
8753fn narrowed_to(keep: &[integer::Kind], offered: &[integer::Kind]) -> Vec<integer::Kind> {
8754    let narrowed: Vec<integer::Kind> =
8755        offered.iter().copied().filter(|kind| keep.contains(kind)).collect();
8756    if narrowed.is_empty() { offered.to_vec() } else { narrowed }
8757}
8758
8759/// Which cascades are worth trying on a part of plain integers.
8760///
8761/// Wider than [`Codes`] because the values are not codes and carry whatever shape the column has.
8762/// A timestamp column climbs, so delta is the one that matters and is the reason this exists at
8763/// all: three timestamp columns in ClickBench were coming out at exactly eight bytes a row with
8764/// nothing asked of them. The same three columns are why the stride is here, since a timestamp
8765/// loaded from a source that recorded whole seconds is microseconds with twenty zero bits under
8766/// every value. A column that is one value with a handful of exceptions is sparse. What is still
8767/// left out is the dictionary, for the same reason as in [`Codes`]: it is the most
8768/// expensive candidate to try and this file already puts the columns that want one through a
8769/// dictionary of their own before they ever reach here.
8770#[derive(Debug)]
8771struct Fixed;
8772
8773impl chooser::Chooser for Fixed {
8774    fn name(&self) -> &'static str {
8775        "fixed"
8776    }
8777
8778    fn narrow_strings(
8779        &self,
8780        _values: &[&[u8]],
8781        offered: &[string::Kind],
8782        _depth: u8,
8783    ) -> Vec<string::Kind> {
8784        offered.to_vec()
8785    }
8786
8787    fn narrow_integers(
8788        &self,
8789        _values: &[i64],
8790        offered: &[integer::Kind],
8791        depth: u8,
8792    ) -> Vec<integer::Kind> {
8793        narrowed_to(Fixed::keep(depth), offered)
8794    }
8795
8796    fn considers_integer(&self, kind: integer::Kind, depth: u8) -> bool {
8797        Fixed::keep(depth).contains(&kind)
8798    }
8799}
8800
8801impl Fixed {
8802    fn keep(depth: u8) -> &'static [integer::Kind] {
8803        if depth == 0 {
8804            &[
8805                integer::Kind::Constant,
8806                integer::Kind::Packed,
8807                integer::Kind::Delta,
8808                integer::Kind::Rle,
8809                integer::Kind::Sparse,
8810                integer::Kind::Strided,
8811            ]
8812        } else {
8813            &[integer::Kind::Constant, integer::Kind::Packed, integer::Kind::Delta]
8814        }
8815    }
8816}
8817
8818/// Every value of an integer part as an `i64`, or `None` for a part this cannot widen without
8819/// losing one.
8820///
8821/// `UBIGINT` is the only integer type left out, because half its range does not fit and a page that
8822/// silently wrapped would be worse than a page that stays plain. Booleans and strings are not
8823/// integers and have their own ways of being small.
8824fn widened(data: &Data) -> Option<Vec<i64>> {
8825    match data {
8826        Data::Int8(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
8827        Data::UInt8(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
8828        Data::Int16(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
8829        Data::UInt16(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
8830        Data::Int32(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
8831        Data::UInt32(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
8832        Data::Int64(values) => Some(values.to_vec()),
8833        _ => None,
8834    }
8835}
8836
8837/// An integer type a cascaded page can be read back into, and how to tell whether a value fits.
8838///
8839/// This exists so that the check and the conversion can be two loops instead of one. `TryFrom` puts
8840/// them together, which is the right shape for one value and the wrong one for a page: a fallible
8841/// conversion a value at a time is a branch a value at a time, the branch decides whether the loop
8842/// keeps going, and a loop like that is one no compiler will widen.
8843trait Narrow: Copy {
8844    /// How wide this type is, and what to add to a value to put its range at the bottom of a `u64`.
8845    ///
8846    /// Half the width for a signed type, which is what moves its smallest value to zero, and nothing
8847    /// for an unsigned one, whose smallest value is already there.
8848    const BIASED: (u32, u64);
8849
8850    /// The value narrowed, which the caller has already shown fits.
8851    fn narrow(value: i64) -> Self;
8852}
8853
8854/// The bits of `value` a `T` cannot hold, and zero when the value fits.
8855///
8856/// The question is asked this way round because the answers or together. A page fits when every
8857/// residue in it is zero, so the loop is an or into an accumulator and the decision is one test
8858/// after it, where asking whether each value is between a floor and a ceiling gives an answer that
8859/// does not combine and turns into a running minimum and maximum.
8860///
8861/// Biasing and shifting is what the answer is made of, rather than anything that reads more like the
8862/// question, because those are the operations a machine has four of. A 64 bit integer minimum is
8863/// AVX-512. So is a 64 bit arithmetic shift right, which is how the sign extension this could be
8864/// written as would have to be done. An add and a logical shift right are AVX2 and are on every
8865/// machine this runs on, so this is the form that gets four values a cycle instead of one.
8866///
8867/// Adding the bias moves the type's range to `0..=2^bits`, wrapping, so everything in range shifts
8868/// away to nothing and everything outside it leaves something behind. A negative value under an
8869/// unsigned type is caught by the same shift, because a negative `i64` read as a `u64` is enormous.
8870#[allow(clippy::cast_sign_loss, reason = "a residue is a bit pattern and not a number")]
8871fn residue<T: Narrow>(value: i64) -> u64 {
8872    let (bits, bias) = T::BIASED;
8873    (value as u64).wrapping_add(bias) >> bits
8874}
8875
8876/// Says a primitive integer narrows with `as`, and where the bottom of its range is.
8877///
8878/// `as` is a truncation and is the right operation here only because [`fit`] has already found every
8879/// residue zero, and it is what makes the second loop a narrowing store with no branch in it.
8880macro_rules! narrows {
8881    ($($ty:ty => $bias:expr),* $(,)?) => {$(
8882        impl Narrow for $ty {
8883            const BIASED: (u32, u64) = (<$ty>::BITS, $bias);
8884
8885            #[allow(
8886                clippy::cast_possible_truncation,
8887                clippy::cast_sign_loss,
8888                reason = "the caller has checked the bits this truncates away"
8889            )]
8890            fn narrow(value: i64) -> Self {
8891                value as Self
8892            }
8893        }
8894    )*};
8895}
8896
8897narrows! {
8898    i8 => 1 << 7,
8899    u8 => 0,
8900    i16 => 1 << 15,
8901    u16 => 0,
8902    i32 => 1 << 31,
8903    u32 => 0,
8904}
8905
8906/// Narrows a page's values, refusing the page if any of them does not fit.
8907///
8908/// The check first and the conversion second, rather than a fallible conversion a value at a time.
8909/// Both loops here are ones a compiler widens: [`residue`] is three instructions a lane and a
8910/// narrowing store is one. The version before this was a `TryFrom` and a `collect` into a `Result`,
8911/// which is a compare, a branch and a short circuit a value at a time, and on ClickBench 39 it was
8912/// seven percent of the query. The version after that kept a running minimum and maximum, which is
8913/// the obvious way to ask and needs a 64 bit integer minimum that AVX2 does not have, so it stayed
8914/// a value at a time and was still ten percent of the same query.
8915///
8916/// An empty page has nothing to refuse, which falls out of the accumulator starting at zero rather
8917/// than needing a case of its own.
8918fn fit<T: Narrow>(values: &[i64]) -> Result<Vec<T>> {
8919    let mut spilled = 0u64;
8920    for value in values {
8921        spilled |= residue::<T>(*value);
8922    }
8923    if spilled != 0 {
8924        return Err(invalid("page value is not of its type"));
8925    }
8926    Ok(values.iter().map(|value| T::narrow(*value)).collect())
8927}
8928
8929/// The same values back in the width the column is declared at.
8930///
8931/// A value that does not fit is a page that disagrees with the directory about what the column is,
8932/// which is a damaged file rather than a caller error, so it is refused rather than truncated.
8933fn narrowed(ty: &LogicalType, values: Vec<i64>) -> Result<Data> {
8934    Ok(match ty {
8935        LogicalType::TinyInt => Data::Int8(fit::<i8>(&values)?.into()),
8936        LogicalType::UTinyInt => Data::UInt8(fit::<u8>(&values)?.into()),
8937        LogicalType::SmallInt => Data::Int16(fit::<i16>(&values)?.into()),
8938        LogicalType::USmallInt => Data::UInt16(fit::<u16>(&values)?.into()),
8939        LogicalType::Integer | LogicalType::Date => Data::Int32(fit::<i32>(&values)?.into()),
8940        LogicalType::UInteger => Data::UInt32(fit::<u32>(&values)?.into()),
8941        LogicalType::BigInt
8942        | LogicalType::Timestamp
8943        | LogicalType::Time
8944        | LogicalType::TimeTz
8945        | LogicalType::TimestampTz
8946        | LogicalType::TimestampS
8947        | LogicalType::TimestampMs
8948        | LogicalType::TimestampNs => Data::Int64(values.into()),
8949        // A decimal is an integer of unscaled units, so the cascade reads back into whichever
8950        // integer the declared width says the column is stored as.
8951        LogicalType::Decimal { .. } => match ty.physical() {
8952            PhysicalType::Int16 => Data::Int16(fit::<i16>(&values)?.into()),
8953            PhysicalType::Int32 => Data::Int32(fit::<i32>(&values)?.into()),
8954            PhysicalType::Int64 => Data::Int64(values.into()),
8955            _ => return Err(invalid("cascade codec belongs to a decimal that is not an integer")),
8956        },
8957        _ => return Err(invalid("cascade codec belongs to a page that is not integers")),
8958    })
8959}
8960
8961/// How many bytes a part of this type costs written out plainly, which is what the cascade has to
8962/// beat before it is worth the decode.
8963fn plain_width(ty: &LogicalType) -> Option<usize> {
8964    Some(match ty {
8965        LogicalType::TinyInt | LogicalType::UTinyInt => 1,
8966        LogicalType::SmallInt | LogicalType::USmallInt => 2,
8967        LogicalType::Integer | LogicalType::UInteger | LogicalType::Date => 4,
8968        LogicalType::BigInt
8969        | LogicalType::Timestamp
8970        | LogicalType::Time
8971        | LogicalType::TimeTz
8972        | LogicalType::TimestampTz
8973        | LogicalType::TimestampS
8974        | LogicalType::TimestampMs
8975        | LogicalType::TimestampNs => 8,
8976        LogicalType::Decimal { .. } => match ty.physical() {
8977            PhysicalType::Int16 => 2,
8978            PhysicalType::Int32 => 4,
8979            PhysicalType::Int64 => 8,
8980            // The widest decimals are stored as `i128`, which the cascade does not widen into, so
8981            // they take the plain path and there is nothing here to compare against.
8982            _ => return None,
8983        },
8984        _ => return None,
8985    })
8986}
8987
8988/// A part's plain integers through the cascade, or `None` when nothing it offers is worth it.
8989///
8990/// What it has to beat is whatever the page would otherwise have cost, which is the bit packed form
8991/// where there is one and the plain width where there is not. Both are cheaper to decode than a
8992/// cascade, so a tie goes to them.
8993fn cascaded(
8994    flat: &Vector,
8995    ty: &LogicalType,
8996    packed: Option<&Packed<'_>>,
8997    settling: &mut Settling,
8998) -> Result<Option<Vec<u8>>> {
8999    let (Some(width), Some(data)) = (plain_width(ty), flat.data()) else { return Ok(None) };
9000    let Some(values) = widened(data) else { return Ok(None) };
9001    let plain = values.len().saturating_mul(width);
9002    let best = match packed {
9003        // The tag, the base, the word count and the words, which is what the codec 2 branch writes.
9004        Some(packed) => plain.min(21 + size_of_val(packed.words())),
9005        None => plain,
9006    };
9007    let out = settling.encode(&values)?;
9008    Ok((out.len() < best).then_some(out))
9009}
9010
9011/// How often the parts of one column in one stripe search the cascade again, in parts.
9012///
9013/// A stripe is 64 parts, so this is four searches a stripe where there were 64. The search is
9014/// what the cascade costs: on ClickBench `hits` the integer cascade was about a tenth of the load's
9015/// CPU and nearly all of it under `encode_pages`, trying six trees on every part to keep the one
9016/// the part before had kept.
9017const SEARCH_EVERY: usize = 16;
9018
9019/// What the parts of one column in one stripe have settled on in the integer cascade.
9020///
9021/// One of these per column per stripe, used in part order, so what a part comes out as depends on
9022/// the stripe and not on which thread wrote it or on how many there were.
9023#[derive(Debug, Default)]
9024struct Settling {
9025    /// The shape of the last part that was searched, with what its top level offered, its length
9026    /// and its row count, which is the size a replay is held to.
9027    shape: Option<Shape>,
9028    /// Parts replayed since that search.
9029    since: usize,
9030}
9031
9032impl Settling {
9033    /// A part's integers through the cascade, replaying the settled shape where there is one.
9034    ///
9035    /// The replay is kept when it held and came out no more than a quarter bigger a row than the
9036    /// part the shape was searched on. Past that the column has changed under it and the part is
9037    /// searched. A replay that stopped fitting partway has already searched from where it stopped,
9038    /// so its shape is taken as the new one rather than searched a second time.
9039    fn encode(&mut self, values: &[i64]) -> Result<Vec<u8>> {
9040        if let Some(shape) = self.shape.as_ref().filter(|_| self.since < SEARCH_EVERY) {
9041            let replay = chooser::Replay::new(&shape.kinds, &Fixed).expecting(&shape.offered);
9042            let out = integer::encode_with(values, &replay)?;
9043            if !replay.held() {
9044                self.settle(&out, values.len(), replay.first_offered())?;
9045                return Ok(out);
9046            }
9047            let grown = (out.len() as u128) * (shape.rows as u128) * 4;
9048            if grown <= (shape.len as u128) * (values.len() as u128) * 5 {
9049                self.since += 1;
9050                return Ok(out);
9051            }
9052        }
9053        // A replay of nothing is the search, and says what the top level offered on the way.
9054        let search = chooser::Replay::new(&[], &Fixed);
9055        let out = integer::encode_with(values, &search)?;
9056        self.settle(&out, values.len(), search.first_offered())?;
9057        Ok(out)
9058    }
9059
9060    fn settle(&mut self, out: &[u8], rows: usize, offered: Vec<integer::Kind>) -> Result<()> {
9061        let kinds = integer::shape(out)?;
9062        self.shape = Some(Shape { kinds, offered, len: out.len().max(1), rows: rows.max(1) });
9063        self.since = 0;
9064        Ok(())
9065    }
9066}
9067
9068/// A searched part's cascade, what its top level was offered, and what it came to.
9069#[derive(Debug)]
9070struct Shape {
9071    kinds: Vec<integer::Kind>,
9072    offered: Vec<integer::Kind>,
9073    len: usize,
9074    rows: usize,
9075}
9076
9077/// A part's dictionary codes through the integer cascade, or `None` when the cascade did not pay.
9078///
9079/// Until now this stream was a `u32` a row with nothing asked of it, and on ClickBench that was
9080/// 400,185,326 bytes for every one of the 28 varchar columns, the same count for `URL` as for a
9081/// column holding the empty string in nearly every row. Codes are dense integers counted from zero
9082/// and a part holds 1024 of them, which is the shape frame of reference is best at, and a column
9083/// with one value everywhere comes back a constant costing nothing per row rather than four bytes.
9084///
9085/// The result is taken only when it is smaller than the plain form. A cascade is allowed to come
9086/// out larger on a part whose codes are genuinely wide, `URL` has about sixty million distinct
9087/// values, and there is no reason to pay for the decode when it does.
9088/// A varchar page as one FSST layer, or `None` when it did not pay.
9089///
9090/// Until now a varchar page that neither the global dictionary nor the per page dictionary claimed
9091/// was written out raw: four bytes of offset a row and then the bytes. That is the right answer for
9092/// a page of values with nothing in common and the wrong one for a page of English, and a column of
9093/// comments is the case this exists for.
9094///
9095/// One layer and not the full string cascade, which is what the payload blocks of a global
9096/// dictionary go through. The cascade is a search: it encodes the page under every candidate it has
9097/// and recurses into the integer cascade for the lengths of each one, and on TPC-H `orders` that
9098/// took the write from 6.9 s to 48.3 s. It reads back no faster than the dictionary it replaced
9099/// either, 1.807 G instructions against 1.810 G for `select o_comment from orders`, because
9100/// unpicking a nest of layers a value at a time costs what the dictionary's payload block decode
9101/// cost. Raw pages of the same column read in 0.686 G, which says the whole of the difference is
9102/// what the page has to be put back together from.
9103///
9104/// FSST alone keeps most of what the cascade found and gives all of that back. Decoding it is one
9105/// pass over the payload into one buffer, the values are laid end to end in it the way the raw form
9106/// already lays them out, and what the reader hands a chunk is views over that buffer.
9107///
9108/// The page dictionary gets first refusal because it is cheaper still, and it wins on a page whose
9109/// values repeat. What is left for this is the page whose values mostly do not, which is exactly the
9110/// page that was being written raw.
9111///
9112/// Taken only when it comes out smaller than the raw form, so a page of incompressible values pays
9113/// nothing at read time for having been offered.
9114fn text_compressed(flat: &Vector) -> Result<Option<Vec<u8>>> {
9115    let mut values: Vec<&[u8]> = Vec::with_capacity(flat.len());
9116    let mut payload = 0_usize;
9117    for row in 0..flat.len() {
9118        let text = flat.text_at(row).unwrap_or("").as_bytes();
9119        payload = payload.saturating_add(text.len());
9120        values.push(text);
9121    }
9122    // What codec 0 writes for a varchar page: an offset a row and one more, then the payload.
9123    let plain = (flat.len() + 1).saturating_mul(4).saturating_add(payload);
9124    let Some(out) = string::encode_only(string::Kind::Fsst, &values)? else {
9125        return Ok(None);
9126    };
9127    Ok((out.len() < plain).then_some(out))
9128}
9129
9130fn encoded_codes(codes: &[u32]) -> Result<Option<Vec<u8>>> {
9131    let wide: Vec<i64> = codes.iter().map(|code| i64::from(*code)).collect();
9132    let coded = integer::encode_with(&wide, &Codes)?;
9133    let plain = codes.len().saturating_mul(size_of::<u32>());
9134    Ok((coded.len() < plain).then_some(coded))
9135}
9136
9137/// The validity of a page, which is a flag and then, when some rows are null and some are not, a
9138/// bit a row with the valid ones set.
9139fn push_validity(out: &mut Vec<u8>, flat: &Vector) {
9140    let flag = match flat.validity() {
9141        Validity::AllValid => 0,
9142        Validity::AllInvalid => 1,
9143        Validity::Mask(_) => 2,
9144    };
9145    out.push(flag);
9146    if flag == 2 {
9147        for group in (0..flat.len()).step_by(8) {
9148            let mut bits = 0_u8;
9149            for bit in 0..8 {
9150                if group + bit < flat.len() && !flat.is_null_at(group + bit) {
9151                    bits |= 1 << bit;
9152                }
9153            }
9154            out.push(bits);
9155        }
9156    }
9157}
9158
9159/// One part of a column coded against its global dictionary as a page, from the codes and the
9160/// validity [`push_validity`] wrote for it.
9161///
9162/// The codes go through the integer cascade when that comes out smaller than four bytes a code,
9163/// which on a column that repeats itself it nearly always does, and are written as they are when it
9164/// does not.
9165fn coded_page(codes: &[u32], validity: &[u8]) -> Result<Vec<u8>> {
9166    let coded = encoded_codes(codes)?;
9167    let mut out = Vec::with_capacity(
9168        1 + validity.len() + coded.as_ref().map_or(size_of_val(codes), Vec::len),
9169    );
9170    out.push(if coded.is_some() { 4 } else { 3 });
9171    out.extend_from_slice(validity);
9172    match coded {
9173        Some(coded) => out.extend_from_slice(&coded),
9174        None => {
9175            for &code in codes {
9176                put_u32(&mut out, code);
9177            }
9178        }
9179    }
9180    Ok(out)
9181}
9182
9183/// One part of one column as a page, for every column that is not coded against a global
9184/// dictionary. Those are built by [`coded_page`] from codes [`prepare`] handed out.
9185fn encode(vector: &Vector, settling: &mut Settling) -> Result<Vec<u8>> {
9186    let ty = vector.logical_type();
9187    // flatten: the file writer needs a uniform scalar page and does it once per loaded chunk.
9188    let flat = vector.flatten()?;
9189    let mut out = Vec::new();
9190    let dictionary = if ty == &LogicalType::Varchar { string_dictionary(&flat)? } else { None };
9191    let compressed_text = if dictionary.is_none() && ty == &LogicalType::Varchar {
9192        text_compressed(&flat)?
9193    } else {
9194        None
9195    };
9196    let packed_vector = if dictionary.is_none() { Some(flat.bit_packed()?) } else { None };
9197    let packed = packed_vector.as_ref().and_then(Vector::packed_parts);
9198    // Only where nothing else has claimed the page, which is the plain integer case. A packed part
9199    // is still on the table because the cascade has to beat it too: the bit pack takes a part only
9200    // when it halves it, so a column that shrinks by a third was coming out whole.
9201    let cascade =
9202        if dictionary.is_none() { cascaded(&flat, ty, packed.as_ref(), settling)? } else { None };
9203    out.push(if cascade.is_some() {
9204        5
9205    } else if dictionary.is_some() {
9206        1
9207    } else if compressed_text.is_some() {
9208        6
9209    } else if packed.is_some() {
9210        2
9211    } else {
9212        0
9213    });
9214    push_validity(&mut out, &flat);
9215    if let Some(cascade) = cascade {
9216        out.extend_from_slice(&cascade);
9217        return Ok(out);
9218    }
9219    if let Some(dictionary) = dictionary {
9220        out.extend_from_slice(&dictionary);
9221        return Ok(out);
9222    }
9223    if let Some(compressed_text) = compressed_text {
9224        out.extend_from_slice(&compressed_text);
9225        return Ok(out);
9226    }
9227    if let Some(packed) = packed {
9228        if packed.offset() != 0 {
9229            return Err(invalid("writer received a sliced packed vector"));
9230        }
9231        out.push(u8::try_from(packed.width()).map_err(|_| invalid("packed width overflow"))?);
9232        out.extend_from_slice(&packed.base().to_le_bytes());
9233        put_u32(
9234            &mut out,
9235            u32::try_from(packed.words().len()).map_err(|_| invalid("too many packed words"))?,
9236        );
9237        for word in packed.words() {
9238            put_u64(&mut out, *word);
9239        }
9240        return Ok(out);
9241    }
9242    let data = flat.data().ok_or_else(|| invalid("scalar column did not flatten"))?;
9243    match (ty, data) {
9244        (LogicalType::TinyInt, Data::Int8(values)) => {
9245            for value in &**values {
9246                out.extend_from_slice(&value.to_le_bytes());
9247            }
9248        }
9249        (LogicalType::UTinyInt, Data::UInt8(values)) => {
9250            for value in &**values {
9251                out.extend_from_slice(&value.to_le_bytes());
9252            }
9253        }
9254        (LogicalType::SmallInt, Data::Int16(values)) => {
9255            for value in &**values {
9256                out.extend_from_slice(&value.to_le_bytes());
9257            }
9258        }
9259        (LogicalType::USmallInt, Data::UInt16(values)) => {
9260            for value in &**values {
9261                out.extend_from_slice(&value.to_le_bytes());
9262            }
9263        }
9264        (LogicalType::UInteger, Data::UInt32(values)) => {
9265            for value in &**values {
9266                out.extend_from_slice(&value.to_le_bytes());
9267            }
9268        }
9269        (LogicalType::UBigInt, Data::UInt64(values)) => {
9270            for value in &**values {
9271                out.extend_from_slice(&value.to_le_bytes());
9272            }
9273        }
9274        (LogicalType::Integer | LogicalType::Date, Data::Int32(values)) => {
9275            for value in &**values {
9276                out.extend_from_slice(&value.to_le_bytes());
9277            }
9278        }
9279        (
9280            LogicalType::BigInt
9281            | LogicalType::Timestamp
9282            | LogicalType::Time
9283            | LogicalType::TimeTz
9284            | LogicalType::TimestampTz
9285            | LogicalType::TimestampS
9286            | LogicalType::TimestampMs
9287            | LogicalType::TimestampNs,
9288            Data::Int64(values),
9289        ) => {
9290            for value in &**values {
9291                out.extend_from_slice(&value.to_le_bytes());
9292            }
9293        }
9294        // A hugeint and a uuid are both the 128 bit lane, and a uuid's bits are the ones the rest of
9295        // the engine already carries it in, so nothing about the value changes on the way down.
9296        (LogicalType::HugeInt | LogicalType::Uuid, Data::Int128(values)) => {
9297            for value in &**values {
9298                out.extend_from_slice(&value.to_le_bytes());
9299            }
9300        }
9301        (LogicalType::UHugeInt, Data::UInt128(values)) => {
9302            for value in &**values {
9303                out.extend_from_slice(&value.to_le_bytes());
9304            }
9305        }
9306        // Plainly, in the IEEE bytes. The integer encodings do not apply to a float and none of the
9307        // float codecs is worth having before somebody has measured a corpus of them.
9308        (LogicalType::Float, Data::Float32(values)) => {
9309            for value in &**values {
9310                out.extend_from_slice(&value.to_le_bytes());
9311            }
9312        }
9313        (LogicalType::Double, Data::Float64(values)) => {
9314            for value in &**values {
9315                out.extend_from_slice(&value.to_le_bytes());
9316            }
9317        }
9318        // Three counts and not one number. Months, days and microseconds stay apart on disk because
9319        // they are apart in the value: a month is not a fixed number of days and a day is not a
9320        // fixed number of microseconds, which is the whole reason the type has three fields.
9321        (LogicalType::Interval, Data::Interval(values)) => {
9322            for (months, days, micros) in &**values {
9323                out.extend_from_slice(&months.to_le_bytes());
9324                out.extend_from_slice(&days.to_le_bytes());
9325                out.extend_from_slice(&micros.to_le_bytes());
9326            }
9327        }
9328        (LogicalType::Boolean, Data::Bool(values)) => {
9329            for value in &**values {
9330                out.push(u8::from(*value));
9331            }
9332        }
9333        // The unscaled integer and nothing else. Scale is a property of the column and it is in the
9334        // directory already, so writing it a value at a time would be paying for it twice.
9335        (LogicalType::Decimal { .. }, Data::Int16(values)) => {
9336            for value in &**values {
9337                out.extend_from_slice(&value.to_le_bytes());
9338            }
9339        }
9340        (LogicalType::Decimal { .. }, Data::Int32(values)) => {
9341            for value in &**values {
9342                out.extend_from_slice(&value.to_le_bytes());
9343            }
9344        }
9345        (LogicalType::Decimal { .. }, Data::Int64(values)) => {
9346            for value in &**values {
9347                out.extend_from_slice(&value.to_le_bytes());
9348            }
9349        }
9350        (LogicalType::Decimal { .. }, Data::Int128(values)) => {
9351            for value in &**values {
9352                out.extend_from_slice(&value.to_le_bytes());
9353            }
9354        }
9355        // A blob and a bit string go down the way a varchar does, because the layout is the same
9356        // one: an offset a value and then the bytes. What is not the same is that nothing here may
9357        // read the payload as text, which is why this arm asks the column for bytes rather than for
9358        // a string, and why the codecs above that do read text are all asked of a varchar by name.
9359        (LogicalType::Varchar | LogicalType::Blob | LogicalType::Bit, Data::Varlen(values)) => {
9360            let mut bytes = Vec::new();
9361            put_u32(&mut out, 0);
9362            for row in 0..vector.len() {
9363                let value = values.bytes(row).ok_or_else(|| invalid("string view is invalid"))?;
9364                bytes.extend_from_slice(value);
9365                put_u32(
9366                    &mut out,
9367                    u32::try_from(bytes.len())
9368                        .map_err(|_| invalid("string payload exceeds 4GiB"))?,
9369                );
9370            }
9371            out.extend_from_slice(&bytes);
9372        }
9373        _ => return Err(Error::not_implemented(format!("native page for {ty}"))),
9374    }
9375    Ok(out)
9376}
9377
9378fn put_varint(out: &mut Vec<u8>, mut value: u32) {
9379    while value >= 0x80 {
9380        out.push((value as u8 & 0x7f) | 0x80);
9381        value >>= 7;
9382    }
9383    out.push(value as u8);
9384}
9385
9386/// The distinct codes of one part, which is what a stripe's membership index is merged from.
9387fn unique_codes(codes: &[u32]) -> Vec<u32> {
9388    let mut unique = codes.to_vec();
9389    unique.sort_unstable();
9390    unique.dedup();
9391    unique
9392}
9393
9394/// The union of the sorted distinct codes of every part in a stripe.
9395///
9396/// Pairwise up a tree rather than one long list concatenated and sorted. Both are the same order of
9397/// work on paper and the tree is the one that does not sort what is already in order: sixty four
9398/// sorted lists become one in six passes over the values.
9399fn merged_codes(lists: Vec<Vec<u32>>) -> Vec<u32> {
9400    let mut lists = lists;
9401    while lists.len() > 1 {
9402        let mut next = Vec::with_capacity(lists.len().div_ceil(2));
9403        for pair in lists.chunks(2) {
9404            match pair {
9405                [left, right] => next.push(merged_pair(left, right)),
9406                [only] => next.push(only.clone()),
9407                _ => {}
9408            }
9409        }
9410        lists = next;
9411    }
9412    lists.pop().unwrap_or_default()
9413}
9414
9415fn merged_pair(left: &[u32], right: &[u32]) -> Vec<u32> {
9416    let mut out = Vec::with_capacity(left.len().saturating_add(right.len()));
9417    let mut at = 0;
9418    let mut to = 0;
9419    while at < left.len() && to < right.len() {
9420        match left[at].cmp(&right[to]) {
9421            Ordering::Less => {
9422                out.push(left[at]);
9423                at += 1;
9424            }
9425            Ordering::Greater => {
9426                out.push(right[to]);
9427                to += 1;
9428            }
9429            Ordering::Equal => {
9430                out.push(left[at]);
9431                at += 1;
9432                to += 1;
9433            }
9434        }
9435    }
9436    out.extend_from_slice(&left[at..]);
9437    out.extend_from_slice(&right[to..]);
9438    out
9439}
9440
9441/// The widest bounds and the total null count of a stripe, from the bounds of its parts.
9442///
9443/// A bound that is missing from any part is missing from the stripe, because a missing bound means
9444/// nothing is known and a stripe that holds an unknown cannot claim one.
9445fn merged_range(ranges: impl Iterator<Item = Range>) -> Range {
9446    let mut merged = Range::default();
9447    let mut first = true;
9448    for range in ranges {
9449        merged.nulls = merged.nulls.saturating_add(range.nulls);
9450        // Both of these have to survive every part, so one part that could not say anything makes
9451        // the stripe unable to say it either. A sum is dropped on overflow rather than wrapped,
9452        // which leaves the stripe with exact ends and no total, which is a true thing to say.
9453        merged.sum = match (merged.sum.take(), range.sum) {
9454            (Some(held), Some(next)) if !first => held.checked_add(next),
9455            (_, next) if first => next,
9456            _ => None,
9457        };
9458        merged.exact = if first { range.exact } else { merged.exact && range.exact };
9459        if first {
9460            merged.low = range.low;
9461            merged.high = range.high;
9462            first = false;
9463            continue;
9464        }
9465        merged.low = match (merged.low.take(), range.low) {
9466            (Some(held), Some(next)) => Some(held.smaller(next)),
9467            _ => None,
9468        };
9469        merged.high = match (merged.high.take(), range.high) {
9470            (Some(held), Some(next)) => Some(held.larger(next)),
9471            _ => None,
9472        };
9473    }
9474    merged
9475}
9476
9477/// One stripe's sieves for one column: the part count, a length for each part, then their bytes.
9478///
9479/// One page for the whole stripe rather than one per part, because a part's sieve is a few hundred
9480/// bytes and sixty four of those are sixty four directory entries and sixty four reads for something
9481/// a scan walks straight through. A part with no sieve writes a length of zero and costs four bytes.
9482/// `bound` cut down to [`PART_BOUND_BYTES`], still a bound of the side it was.
9483///
9484/// A prefix of a string sorts at or before the string, so cutting one down leaves a low end that is
9485/// still a low end. A high end has to go the other way, so the cut prefix is stepped up at the last
9486/// byte that can carry it, and a prefix of nothing but `0xFF` has no such byte and gives up the
9487/// bound rather than claiming one that is too small. Anything that is not a string is already a
9488/// fixed width and is left alone.
9489fn shortened(bound: Option<Bound>, high: bool) -> Option<Bound> {
9490    match bound {
9491        Some(Bound::Bytes(mut value)) if value.len() > PART_BOUND_BYTES => {
9492            value.truncate(PART_BOUND_BYTES);
9493            if !high {
9494                return Some(Bound::Bytes(value));
9495            }
9496            while let Some(last) = value.pop() {
9497                if last < u8::MAX {
9498                    value.push(last + 1);
9499                    return Some(Bound::Bytes(value));
9500                }
9501            }
9502            None
9503        }
9504        other => other,
9505    }
9506}
9507
9508/// The ranges of one column's parts of one stripe, as a page.
9509///
9510/// The two ends and the null count, and not `exact` or the total. Those two answer a `MIN` or a
9511/// `SUM` out of the directory, and the directory already answers those per stripe, where the same
9512/// number costs sixty times less to keep. What a part range is for is skipping the part, and
9513/// skipping needs the ends. So a range read back from here says it is not exact, which is true of a
9514/// string end that was cut down anyway.
9515fn encode_part_ranges(ranges: &[Range]) -> Result<Vec<u8>> {
9516    let mut out = Vec::new();
9517    put_u32(
9518        &mut out,
9519        u32::try_from(ranges.len()).map_err(|_| invalid("too many parts in a stripe"))?,
9520    );
9521    for range in ranges {
9522        put_bound(&mut out, shortened(range.low.clone(), false).as_ref())?;
9523        put_bound(&mut out, shortened(range.high.clone(), true).as_ref())?;
9524        put_u32(&mut out, u32::try_from(range.nulls).map_err(|_| invalid("null count overflow"))?);
9525    }
9526    Ok(out)
9527}
9528
9529/// The ranges one encoded page holds, one entry per part of the stripe.
9530fn decode_part_ranges(bytes: &[u8]) -> Result<Vec<Range>> {
9531    let mut cur = Cursor::new(bytes);
9532    let parts = cur.u32()? as usize;
9533    let mut out = Vec::new();
9534    for _ in 0..parts {
9535        let low = cur.bound()?;
9536        let high = cur.bound()?;
9537        let nulls = cur.u32()? as usize;
9538        out.push(Range { low, high, nulls, exact: false, sum: None });
9539    }
9540    Ok(out)
9541}
9542
9543fn encode_sieves<'a>(sieves: impl Iterator<Item = &'a Option<Sieve>>) -> Result<Vec<u8>> {
9544    let held: Vec<&Option<Sieve>> = sieves.collect();
9545    let mut out = Vec::new();
9546    put_u32(
9547        &mut out,
9548        u32::try_from(held.len()).map_err(|_| invalid("too many parts in a stripe"))?,
9549    );
9550    for sieve in &held {
9551        let length = sieve.as_ref().map_or(0, Sieve::len);
9552        put_u32(&mut out, u32::try_from(length).map_err(|_| invalid("sieve length overflow"))?);
9553    }
9554    // flatten: a part with no sieve wrote a length of zero above and contributes no bytes here.
9555    for sieve in held.into_iter().flatten() {
9556        out.extend_from_slice(&sieve.to_bytes());
9557    }
9558    Ok(out)
9559}
9560
9561/// The sieves one encoded page holds, one entry per part of the stripe.
9562///
9563/// A part whose bytes are not a sieve this version understands comes back as `None`, which is a part
9564/// that gets read. That is how a file written by a later version of the sieve stays readable rather
9565/// than being a corrupt page.
9566fn decode_sieves(bytes: &[u8]) -> Result<Vec<Option<Sieve>>> {
9567    let parts = u32::from_le_bytes(
9568        bytes
9569            .get(..4)
9570            .ok_or_else(|| invalid("sieve page is truncated"))?
9571            .try_into()
9572            .map_err(|_| invalid("sieve page is truncated"))?,
9573    ) as usize;
9574    let mut lengths = Vec::with_capacity(parts);
9575    for part in 0..parts {
9576        let at = 4 + part * 4;
9577        let field = bytes.get(at..at + 4).ok_or_else(|| invalid("sieve page is truncated"))?;
9578        lengths.push(u32::from_le_bytes(
9579            field.try_into().map_err(|_| invalid("sieve page is truncated"))?,
9580        ) as usize);
9581    }
9582    let mut at = 4 + parts * 4;
9583    let mut out = Vec::with_capacity(parts);
9584    for length in lengths {
9585        if length == 0 {
9586            out.push(None);
9587            continue;
9588        }
9589        let end = at.checked_add(length).ok_or_else(|| invalid("sieve page is truncated"))?;
9590        let field = bytes.get(at..end).ok_or_else(|| invalid("sieve page is truncated"))?;
9591        out.push(Sieve::from_bytes(field));
9592        at = end;
9593    }
9594    if at != bytes.len() {
9595        return Err(invalid("sieve page has trailing bytes"));
9596    }
9597    Ok(out)
9598}
9599
9600/// One stripe's membership index: the code count and then the codes as ascending deltas.
9601///
9602/// The codes have to be sorted and distinct already, which is what [`unique_codes`] and
9603/// [`merged_codes`] hand over. Anything else decodes as different codes, so neither of those two is
9604/// a step a caller can skip.
9605fn encode_membership(unique: &[u32]) -> Vec<u8> {
9606    let mut out = Vec::with_capacity(unique.len().saturating_mul(2).saturating_add(5));
9607    put_varint(&mut out, u32::try_from(unique.len()).unwrap_or(u32::MAX));
9608    let mut previous = 0;
9609    for (at, &code) in unique.iter().enumerate() {
9610        put_varint(&mut out, if at == 0 { code } else { code - previous });
9611        previous = code;
9612    }
9613    out
9614}
9615
9616fn take_varint(bytes: &[u8], at: &mut usize) -> Result<u32> {
9617    let mut value = 0_u32;
9618    for shift in (0..35).step_by(7) {
9619        let byte = *bytes.get(*at).ok_or_else(|| invalid("membership varint is truncated"))?;
9620        *at += 1;
9621        let part = u32::from(byte & 0x7f);
9622        if shift == 28 && part > 0x0f {
9623            return Err(invalid("membership varint overflow"));
9624        }
9625        value = value
9626            .checked_add(
9627                part.checked_shl(shift).ok_or_else(|| invalid("membership varint overflow"))?,
9628            )
9629            .ok_or_else(|| invalid("membership varint overflow"))?;
9630        if byte & 0x80 == 0 {
9631            return Ok(value);
9632        }
9633    }
9634    Err(invalid("membership varint is too long"))
9635}
9636
9637fn decode_membership(bytes: &[u8]) -> Result<Vec<u32>> {
9638    let mut at = 0;
9639    let count = take_varint(bytes, &mut at)? as usize;
9640    let mut codes = Vec::with_capacity(count);
9641    let mut previous = 0_u32;
9642    for index in 0..count {
9643        let delta = take_varint(bytes, &mut at)?;
9644        let code = if index == 0 {
9645            delta
9646        } else {
9647            previous.checked_add(delta).ok_or_else(|| invalid("membership code overflow"))?
9648        };
9649        if index > 0 && code <= previous {
9650            return Err(invalid("membership codes are not increasing"));
9651        }
9652        codes.push(code);
9653        previous = code;
9654    }
9655    if at != bytes.len() {
9656        return Err(invalid("membership page has trailing bytes"));
9657    }
9658    Ok(codes)
9659}
9660
9661fn string_dictionary(vector: &Vector) -> Result<Option<Vec<u8>>> {
9662    let mut by_text = HashMap::new();
9663    let mut values = Vec::new();
9664    let mut codes = Vec::with_capacity(vector.len());
9665    let mut plain_bytes = 0_usize;
9666    for row in 0..vector.len() {
9667        let text = vector.text_at(row).unwrap_or("");
9668        plain_bytes = plain_bytes.saturating_add(text.len());
9669        let code = match by_text.get(text) {
9670            Some(&code) => code,
9671            None => {
9672                let code = u32::try_from(values.len())
9673                    .map_err(|_| invalid("too many dictionary values"))?;
9674                by_text.insert(text, code);
9675                values.push(text);
9676                code
9677            }
9678        };
9679        codes.push(code);
9680    }
9681    let dictionary_bytes = values.iter().map(|value| value.len()).sum::<usize>();
9682    let encoded = 8_usize
9683        .saturating_add((values.len() + 1).saturating_mul(4))
9684        .saturating_add(dictionary_bytes)
9685        .saturating_add(codes.len().saturating_mul(4));
9686    let plain = (vector.len() + 1).saturating_mul(4).saturating_add(plain_bytes);
9687    if encoded >= plain {
9688        return Ok(None);
9689    }
9690    let mut out = Vec::with_capacity(encoded);
9691    put_u32(
9692        &mut out,
9693        u32::try_from(values.len()).map_err(|_| invalid("too many dictionary values"))?,
9694    );
9695    put_u32(
9696        &mut out,
9697        u32::try_from(dictionary_bytes).map_err(|_| invalid("dictionary payload exceeds 4GiB"))?,
9698    );
9699    let mut offset = 0_u32;
9700    put_u32(&mut out, offset);
9701    for value in &values {
9702        offset = offset
9703            .checked_add(
9704                u32::try_from(value.len()).map_err(|_| invalid("dictionary value is too long"))?,
9705            )
9706            .ok_or_else(|| invalid("dictionary payload exceeds 4GiB"))?;
9707        put_u32(&mut out, offset);
9708    }
9709    for value in values {
9710        out.extend_from_slice(value.as_bytes());
9711    }
9712    for code in codes {
9713        put_u32(&mut out, code);
9714    }
9715    Ok(Some(out))
9716}
9717
9718/// The room one closing dictionary takes under [`CLOSE_DICTIONARY_BYTES`], given back when dropped.
9719struct Room<'a, T> {
9720    state: &'a Mutex<(T, usize)>,
9721    finished: &'a Condvar,
9722    bytes: usize,
9723}
9724
9725impl<T> Drop for Room<'_, T> {
9726    fn drop(&mut self) {
9727        let mut held = self.state.lock().unwrap_or_else(PoisonError::into_inner);
9728        held.1 -= self.bytes;
9729        drop(held);
9730        self.finished.notify_all();
9731    }
9732}
9733
9734/// What [`Writer::close_dictionary`] builds for one column and [`Writer::close`] writes.
9735struct ClosedDictionary {
9736    distinct: u64,
9737    frequencies: FrequencySummary,
9738    texts: Vec<Option<Vec<u8>>>,
9739    hosts: Option<host::HostSummary>,
9740    encoded: EncodedDictionary,
9741    /// The bytes of the column's payload blocks, which are already in the file.
9742    payload: u64,
9743}
9744
9745struct EncodedDictionary {
9746    index: Vec<u8>,
9747    ranks: Vec<u8>,
9748    grams: Vec<u8>,
9749}
9750
9751/// Sorts codes into the byte order of the values they name, eight bytes of depth at a time.
9752///
9753/// # What the shape of the data does to a comparison sort
9754///
9755/// Distinct values against distinct prefixes, on the eight million row `hits`:
9756///
9757/// ```text
9758///   distinct   first 8   first 16   first 32   column
9759///  2,266,417        50      8,892    232,630   URL
9760///  2,346,025        49      8,534    204,060   Referer
9761///  1,357,764    81,362    348,340    861,579   Title
9762/// ```
9763///
9764/// Two and a quarter million URLs have fifty distinct first eight bytes between them, because they
9765/// all begin `http://` and then a host and there are not many hosts. So a sort that leads with
9766/// those eight bytes settles almost nothing on `URL` and `Referer`, whatever the comment on it used
9767/// to say, and almost every pair falls through to a comparison of whole values that agree for most
9768/// of their length. `Title` is free text and separates at eight bytes, which is why the design
9769/// looked right when it was written.
9770///
9771/// # What is done about it
9772///
9773/// Sort on eight bytes of the value at the current depth, held beside the code, and then take each
9774/// run that those eight bytes leave tied and sort it again on the next eight. A value is fetched
9775/// from the payload once per eight bytes of depth rather than once per comparison, and the sort
9776/// itself runs over an array of integers that is in cache rather than over pointers into a payload
9777/// that is hundreds of megabytes.
9778///
9779/// That is the whole trick, and it matters because the payload touch is the expensive part. The
9780/// bytes themselves are nearly free once the line is in cache, so reading eight at a time and
9781/// throwing away the ones that were not needed beats going back for each one.
9782///
9783/// # Why the length has to be carried
9784///
9785/// The eight bytes are padded with zero when the value has fewer than eight left, and a zero byte
9786/// can appear in a value, so equal keys do not mean equal bytes. What is true is that a value which
9787/// ran out inside the window is a prefix of any other value with the same key, and a prefix sorts
9788/// first, so how many of the eight bytes were real is the tie break and nothing further is needed.
9789/// A run is only worth another pass when all eight were real, because otherwise the run is one
9790/// value: a dictionary holds a value once.
9791fn sort_by_value<'a>(codes: &mut [u32], values: impl Fn(u32) -> &'a [u8]) {
9792    let mut work = vec![(0, codes.len(), 0)];
9793    let mut keyed: Vec<(u64, u8, u32)> = Vec::new();
9794    while let Some((from, to, depth)) = work.pop() {
9795        let part = &mut codes[from..to];
9796        keyed.clear();
9797        keyed.extend(part.iter().map(|&code| {
9798            let value = values(code);
9799            let rest = value.get(depth..).unwrap_or_default();
9800            (head(rest), rest.len().min(8) as u8, code)
9801        }));
9802        keyed.sort_unstable();
9803        for (slot, entry) in part.iter_mut().zip(keyed.iter()) {
9804            *slot = entry.2;
9805        }
9806        let mut start = 0;
9807        while start < keyed.len() {
9808            let (key, taken, _) = keyed[start];
9809            let mut end = start + 1;
9810            while end < keyed.len() && keyed[end].0 == key && keyed[end].1 == taken {
9811                end += 1;
9812            }
9813            if taken == 8 && end - start > 1 {
9814                work.push((from + start, from + end, depth + 8));
9815            }
9816            start = end;
9817        }
9818    }
9819}
9820
9821/// How few codes are worth sorting on more than one thread.
9822const PARALLEL_SORT_MIN: usize = 1 << 16;
9823
9824/// How many buckets a thread gets in [`sort_by_value_across`], so that a thread that drew a slow
9825/// bucket is not what the others wait for.
9826const BUCKETS_PER_WORKER: usize = 4;
9827
9828/// How many sampled codes stand for each bucket when the splitters are picked.
9829const SAMPLES_PER_BUCKET: usize = 32;
9830
9831/// [`sort_by_value`] over `workers` threads, with the same answer.
9832///
9833/// A sample sort. A sample of the codes is sorted and cut into as many equal runs as there are
9834/// buckets, and the values at the cuts are the splitters. Every code goes to the bucket its value
9835/// falls in by a binary search of the splitters, the buckets are laid end to end in splitter order,
9836/// and each bucket is then sorted on its own by whichever thread takes it. Every value in a bucket
9837/// sorts after every value in the bucket before, so the buckets sorted one by one are the codes
9838/// sorted.
9839///
9840/// The answer is the one [`sort_by_value`] gives down to the order of equal values, not only the
9841/// order of different ones. A global dictionary holds each value once, so there are none, but the
9842/// sort does not rely on it: equal values land in the same bucket in code order, which is the order
9843/// [`sort_by_value`] leaves them in, since the code is the last thing it sorts on.
9844///
9845/// On the 10m ClickBench sample the close sorts five columns of one to three and a half million
9846/// distinct values, one column at a time, and until this each sort ran on one thread while the
9847/// other thirty one waited for it.
9848fn sort_by_value_across<'a>(
9849    codes: &mut [u32],
9850    values: impl Fn(u32) -> &'a [u8] + Sync,
9851    workers: usize,
9852) {
9853    if workers <= 1 || codes.len() < PARALLEL_SORT_MIN {
9854        sort_by_value(codes, values);
9855        return;
9856    }
9857    let buckets = workers * BUCKETS_PER_WORKER;
9858    let wanted = buckets * SAMPLES_PER_BUCKET;
9859    let mut sample = (0..wanted).map(|at| codes[at * codes.len() / wanted]).collect::<Vec<_>>();
9860    sort_by_value(&mut sample, &values);
9861    let splitters =
9862        (1..buckets).map(|cut| values(sample[cut * sample.len() / buckets])).collect::<Vec<_>>();
9863    let values = &values;
9864    let splitters = &splitters;
9865    let per = codes.len().div_ceil(workers);
9866    // Which bucket each code goes to, a run of the codes per thread.
9867    let places = std::thread::scope(|scope| {
9868        codes
9869            .chunks(per)
9870            .map(|run| {
9871                scope.spawn(move || {
9872                    run.iter()
9873                        .map(|&code| {
9874                            let value = values(code);
9875                            splitters.partition_point(|splitter| *splitter <= value) as u32
9876                        })
9877                        .collect::<Vec<_>>()
9878                })
9879            })
9880            .collect::<Vec<_>>()
9881            .into_iter()
9882            .flat_map(|handle| {
9883                handle.join().unwrap_or_else(|panic| std::panic::resume_unwind(panic))
9884            })
9885            .collect::<Vec<_>>()
9886    });
9887    let mut starts = vec![0_usize; buckets + 1];
9888    for &place in &places {
9889        starts[place as usize + 1] += 1;
9890    }
9891    for bucket in 0..buckets {
9892        starts[bucket + 1] += starts[bucket];
9893    }
9894    let mut laid = vec![0_u32; codes.len()];
9895    let mut next = starts.clone();
9896    for (&code, &place) in codes.iter().zip(&places) {
9897        laid[next[place as usize]] = code;
9898        next[place as usize] += 1;
9899    }
9900    drop(places);
9901    let mut runs = Vec::with_capacity(buckets);
9902    let mut rest = laid.as_mut_slice();
9903    for bucket in 0..buckets {
9904        let (run, after) = rest.split_at_mut(starts[bucket + 1] - starts[bucket]);
9905        runs.push(run);
9906        rest = after;
9907    }
9908    // The largest buckets first, since they are taken from the back.
9909    runs.sort_by_key(|run| run.len());
9910    let queue = Mutex::new(runs);
9911    std::thread::scope(|scope| {
9912        for _ in 0..workers {
9913            scope.spawn(|| {
9914                loop {
9915                    let taken = queue.lock().unwrap_or_else(PoisonError::into_inner).pop();
9916                    let Some(run) = taken else { break };
9917                    sort_by_value(run, values);
9918                }
9919            });
9920        }
9921    });
9922    codes.copy_from_slice(&laid);
9923}
9924
9925/// The first eight bytes of a value as an integer that sorts the way the bytes sort.
9926fn head(bytes: &[u8]) -> u64 {
9927    let mut word = [0; 8];
9928    let take = bytes.len().min(8);
9929    word[..take].copy_from_slice(&bytes[..take]);
9930    u64::from_be_bytes(word)
9931}
9932
9933/// One column's dictionary page, which is its index and its sorted order.
9934///
9935/// The payload is not in it. Its blocks are in the file already, written as each was encoded, and
9936/// `places` says where, in block order. With `scattered` set the index records each block's start
9937/// and length, so a reader can find one wherever it went.
9938///
9939/// `scattered` false lays the blocks out the way a file written before format 26 has them, one
9940/// behind the next with only the ends recorded. Nothing in the writer asks for that any more. It is
9941/// kept because [`open_global_dictionary`] still reads those files and a reading path that nothing
9942/// can produce is a reading path nothing tests.
9943fn encode_global_dictionary(
9944    dictionary: &GlobalDictionary,
9945    order: &[(u64, u32)],
9946    places: &[Placed],
9947    scattered: bool,
9948) -> Result<EncodedDictionary> {
9949    let values = dictionary.values();
9950    if order.len() != values {
9951        return Err(invalid("global dictionary order does not cover its values"));
9952    }
9953    let blocks = values.div_ceil(TEXT_PAYLOAD_VALUES);
9954    if places.len() != blocks {
9955        return Err(invalid("global dictionary payload is not the blocks it says it is"));
9956    }
9957    if dictionary.grams.len() != blocks {
9958        return Err(invalid("global dictionary signatures do not cover its blocks"));
9959    }
9960    let (ranks, rank_ends) = encode_ranks(order, code_width(values))?;
9961    let rank_blocks = values.div_ceil(TEXT_RANK_BLOCK);
9962    let offset_bits = offset_width(&dictionary.ends);
9963    let payload_words = if scattered { 3 } else { 2 };
9964    let index_len = DICTIONARY_HEADER
9965        .checked_add(offset_bytes(values, offset_bits))
9966        .and_then(|len| len.checked_add(blocks.checked_mul(payload_words * 8)?))
9967        .and_then(|len| len.checked_add(rank_blocks.checked_mul(16)?))
9968        .and_then(|len| len.checked_add(8))
9969        .ok_or_else(|| invalid("global dictionary index length overflow"))?;
9970    let mut index = Vec::with_capacity(index_len);
9971    put_u32(
9972        &mut index,
9973        u32::try_from(values).map_err(|_| invalid("global dictionary has too many values"))?,
9974    );
9975    put_u32(&mut index, TEXT_PAYLOAD_VALUES as u32);
9976    put_u32(
9977        &mut index,
9978        u32::try_from(blocks).map_err(|_| invalid("global dictionary has too many blocks"))?,
9979    );
9980    let flag = (if scattered { DICTIONARY_SCATTERED } else { 0 }) | DICTIONARY_GRAMS;
9981    put_u32(&mut index, offset_bits as u32 | flag);
9982    encode_offsets(&dictionary.ends, offset_bits, &mut index)?;
9983    // Where each block is and how long it is, so a reader can find one. The stored blocks are
9984    // shorter than the decoded ones and by a different amount each, so their lengths are the one
9985    // thing the offsets above no longer say, and where they start is no longer arithmetic on the
9986    // block before once a block is written the moment it is encoded.
9987    let mut end = 0_u64;
9988    for place in places {
9989        if scattered {
9990            put_u64(&mut index, place.start);
9991            put_u64(&mut index, place.length);
9992        } else {
9993            end = end
9994                .checked_add(place.length)
9995                .ok_or_else(|| invalid("global dictionary payload overflow"))?;
9996            put_u64(&mut index, end);
9997        }
9998    }
9999    for place in places {
10000        put_u64(&mut index, place.hash);
10001    }
10002    // The same two lists for the sorted order. A rank block is packed at whatever width its own
10003    // heads need, so where one ends is no longer arithmetic on the block number.
10004    if rank_ends.len() != rank_blocks {
10005        return Err(invalid("global dictionary order is not the blocks it says it is"));
10006    }
10007    for end in &rank_ends {
10008        put_u64(&mut index, *end);
10009    }
10010    let mut at = 0_usize;
10011    for end in &rank_ends {
10012        let end = usize::try_from(*end).map_err(|_| invalid("global dictionary order overflow"))?;
10013        put_u64(&mut index, checksum(&ranks[at..end]));
10014        at = end;
10015    }
10016    let gram_len = blocks
10017        .checked_mul(TEXT_GRAM_BYTES)
10018        .ok_or_else(|| invalid("global dictionary signature count overflow"))?;
10019    let mut grams = Vec::with_capacity(gram_len);
10020    for block in &dictionary.grams {
10021        grams.extend_from_slice(block);
10022    }
10023    put_u64(&mut index, checksum(&grams));
10024    if index.len() != index_len {
10025        return Err(invalid("global dictionary index is not the length it was laid out for"));
10026    }
10027    Ok(EncodedDictionary { index, ranks, grams })
10028}
10029
10030/// How many blocks of the payload the shape is settled on.
10031///
10032/// Eight blocks is 8,192 values, which is the sample `chooser::Sampled` draws and is that size for
10033/// the same reason. They are spread across the dictionary rather than taken off the front, because
10034/// a dictionary is in the order values were first seen and the front of it is the first morsel of
10035/// the load.
10036const PAYLOAD_SAMPLE_BLOCKS: usize = 8;
10037
10038/// The shapes the payload encoder picks between.
10039///
10040/// Narrow on purpose. The exhaustive search encodes every candidate at every level and runs at two
10041/// to six megabytes a second on this data, which over the twelve gigabytes of dictionary `hits`
10042/// carries is about an hour of processor time, so it cannot be what a load does. Each of these
10043/// settles the outer level and the one below it, which is where almost all of that hour goes, and
10044/// leaves the levels under them to the exhaustive search where the chunks are small enough for it
10045/// to cost nothing.
10046///
10047/// Measured on the five ClickBench columns that have a dictionary worth the name, at 1,024 values a
10048/// block, against the exhaustive search over the same blocks:
10049///
10050/// | column | exhaustive | FRONT then LZ | LZ then FSST | LZ then PLAIN |
10051/// |---|---|---|---|---|
10052/// | 2 | 2.923 at 4.3 MB/s | 2.587 at 21.2 | 2.593 at 36.1 | 2.538 at 53.6 |
10053/// | 13 | 3.093 at 3.1 | 3.029 at 36.4 | 2.921 at 35.7 | 2.770 at 82.9 |
10054/// | 14 | 2.330 at 2.1 | 2.283 at 24.3 | 2.213 at 23.5 | 2.113 at 67.6 |
10055/// | 39 | 2.459 at 5.3 | 2.147 at 10.6 | 2.145 at 29.3 | 2.088 at 43.1 |
10056/// | 56 | 4.694 at 6.3 | 4.381 at 51.0 | 4.172 at 50.6 | 3.983 at 86.8 |
10057///
10058/// The best of the three per column is 98 percent of the exhaustive ratio for a tenth of the time.
10059/// `FSST` and `PLAIN` on their own are in the list as a floor rather than to win. `FSST` is the
10060/// right answer for text that does not share prefixes with its neighbours, and `PLAIN` is there so
10061/// that a column nothing compresses is found out in the sample and written at a gigabyte a second
10062/// rather than searched for an answer that does not exist.
10063fn payload_shapes() -> Vec<chooser::Settled> {
10064    let integers = vec![integer::Kind::Packed];
10065    [
10066        vec![string::Kind::Front, string::Kind::Lz],
10067        vec![string::Kind::Lz, string::Kind::Fsst],
10068        vec![string::Kind::Lz, string::Kind::Plain],
10069        vec![string::Kind::Fsst],
10070        vec![string::Kind::Plain],
10071    ]
10072    .into_iter()
10073    .map(|strings| chooser::Settled::new(strings, integers.clone()))
10074    .collect()
10075}
10076
10077/// Syncs the file, and counts the sync and how long it took as a publish wait when a load is being
10078/// profiled.
10079///
10080/// A wait rather than time, because the time is already in the publish span around it. What the
10081/// wait columns add is how much of publish was the device, which on the WSL2 disk of the gaming PC
10082/// is most of it: a sync there costs about two milliseconds (see `rudb_device_card`).
10083fn synced(file: &File, profile: Option<&LoadProfile>) -> Result<()> {
10084    let started = profile.map(|_| std::time::Instant::now());
10085    file.sync_all().map_err(io)?;
10086    if let (Some(profile), Some(started)) = (profile, started) {
10087        profile.waited(
10088            Stage::Publish,
10089            u64::try_from(started.elapsed().as_nanos()).unwrap_or(u64::MAX),
10090        );
10091    }
10092    Ok(())
10093}
10094
10095/// One sealed dictionary block on its way to being encoded outside the writer's lock.
10096///
10097/// Handed out by the merge that sealed it and encoded with the pages of the same stripe. See
10098/// [`GlobalDictionary::hand_out`].
10099#[derive(Debug)]
10100pub(crate) struct Unencoded {
10101    column: usize,
10102    at: usize,
10103    ends: Vec<u32>,
10104    bytes: Vec<u8>,
10105    shape: chooser::Settled,
10106}
10107
10108impl Unencoded {
10109    /// The encoded block and its signature.
10110    pub(crate) fn encode(&self) -> Result<EncodedBlock> {
10111        let values = block_values(&self.ends, &self.bytes);
10112        Ok((string::encode_with(&values, &self.shape)?, block_grams(&values)))
10113    }
10114
10115    /// The column and the block number the encoded block goes back to.
10116    pub(crate) fn place(&self) -> (usize, usize) {
10117        (self.column, self.at)
10118    }
10119}
10120
10121/// One encoded dictionary block and the signature of the values in it.
10122///
10123/// Boxed because it is carried around in things that are otherwise small.
10124pub(crate) type EncodedBlock = (Vec<u8>, Box<[u8; TEXT_GRAM_BYTES]>);
10125
10126/// The conservative four-byte substring signature of one block's values.
10127fn block_grams(values: &[&[u8]]) -> Box<[u8; TEXT_GRAM_BYTES]> {
10128    let mut grams = Box::new([0_u8; TEXT_GRAM_BYTES]);
10129    for value in values {
10130        for gram in value.windows(4) {
10131            for bit in gram_bits(gram) {
10132                grams[bit / 8] |= 1 << (bit % 8);
10133            }
10134        }
10135    }
10136    grams
10137}
10138
10139/// The values of one block, given where each of them ends relative to the block.
10140fn block_values<'a>(ends: &[u32], bytes: &'a [u8]) -> Vec<&'a [u8]> {
10141    let mut out = Vec::with_capacity(ends.len());
10142    let mut from = 0;
10143    for &to in ends {
10144        out.push(&bytes[from..to as usize]);
10145        from = to as usize;
10146    }
10147    out
10148}
10149
10150/// Encodes every block still raw at the end of a load: the part block each column ends on and,
10151/// for a column too small to have settled a shape, every block it has.
10152///
10153/// Across threads, the way [`encode_ready`] does it. This ran one column at a time on the thread
10154/// closing the table, and a column that never settled a shape encodes each block by trying every
10155/// candidate, so on a million rows of `hits` it was most of the load's CPU on one core.
10156fn finish_dictionaries(dictionaries: &mut [Option<GlobalDictionary>]) -> Result<()> {
10157    for dictionary in dictionaries.iter_mut().flatten() {
10158        if !dictionary.early.is_empty() {
10159            return Err(Error::internal("a dictionary block handed out never came back"));
10160        }
10161        dictionary.seal_rest();
10162    }
10163    encode_waiting(dictionaries)?;
10164    // A block handed out and never given back leaves a gap nothing above would notice when it was
10165    // the last one, so the count is checked against the values as well.
10166    if dictionaries
10167        .iter()
10168        .flatten()
10169        .any(|dictionary| dictionary.encoded() != dictionary.values().div_ceil(TEXT_PAYLOAD_VALUES))
10170    {
10171        return Err(Error::internal("a dictionary block handed out never came back"));
10172    }
10173    Ok(())
10174}
10175
10176/// Encodes the waiting blocks of every dictionary across threads, and appends them to their columns
10177/// in order.
10178fn encode_waiting(dictionaries: &mut [Option<GlobalDictionary>]) -> Result<()> {
10179    let jobs = dictionaries
10180        .iter()
10181        .enumerate()
10182        .flat_map(|(column, held)| {
10183            (0..held.as_ref().map_or(0, |held| held.waiting.len())).map(move |at| (column, at))
10184        })
10185        .collect::<Vec<_>>();
10186    if jobs.is_empty() {
10187        return Ok(());
10188    }
10189    let one = |column: usize, at: usize| -> Result<(usize, usize, EncodedBlock)> {
10190        let held = dictionaries[column].as_ref().ok_or_else(|| Error::internal("no dictionary"))?;
10191        Ok((column, at, held.encode_waiting(at)?))
10192    };
10193    let workers = std::thread::available_parallelism()
10194        .map_or(1, usize::from)
10195        .min(MAX_FREQUENCY_WORKERS)
10196        .min(jobs.len());
10197    let made = if workers <= 1 {
10198        jobs.iter().map(|&(column, at)| one(column, at)).collect::<Result<Vec<_>>>()?
10199    } else {
10200        let next = AtomicUsize::new(0);
10201        let jobs = &jobs;
10202        let pieces = std::thread::scope(|scope| {
10203            (0..workers)
10204                .map(|_| {
10205                    scope.spawn(|| {
10206                        let mut mine = Vec::new();
10207                        loop {
10208                            let job = next.fetch_add(1, Atomic::Relaxed);
10209                            let Some(&(column, at)) = jobs.get(job) else { break };
10210                            mine.push(one(column, at)?);
10211                        }
10212                        Ok(mine)
10213                    })
10214                })
10215                .collect::<Vec<_>>()
10216                .into_iter()
10217                .map(|handle| {
10218                    handle
10219                        .join()
10220                        .map_err(|_| Error::internal("a dictionary encode worker panicked"))?
10221                })
10222                .collect::<Result<Vec<_>>>()
10223        })?;
10224        pieces.into_iter().flatten().collect()
10225    };
10226    let mut done: Vec<Vec<(usize, EncodedBlock)>> =
10227        (0..dictionaries.len()).map(|_| Vec::new()).collect();
10228    for (column, at, bytes) in made {
10229        done[column].push((at, bytes));
10230    }
10231    for (column, mut made) in done.into_iter().enumerate() {
10232        if made.is_empty() {
10233            continue;
10234        }
10235        let Some(held) = dictionaries[column].as_mut() else { continue };
10236        made.sort_by_key(|(at, _)| *at);
10237        let waiting = std::mem::take(&mut held.waiting);
10238        for ((at, _), (_, block)) in waiting.into_iter().zip(made) {
10239            if held.encoded() != at {
10240                return Err(Error::internal("a dictionary block was encoded out of order"));
10241            }
10242            held.push_block(block);
10243        }
10244    }
10245    Ok(())
10246}
10247
10248/// Which of [`payload_shapes`] comes out smallest over a sample of the blocks.
10249///
10250/// Every shape is encoded over the same sample and the smallest wins, which is the exhaustive
10251/// search moved up a level: over shapes of a column rather than over candidates of a chunk. The
10252/// sample is spread across the dictionary so that the first and last blocks are both in it, because
10253/// a dictionary written in first seen order has its common values at the front and its long tail at
10254/// the back, and those do not compress alike. Which blocks those are is
10255/// [`GlobalDictionary::seal`]'s to decide, because by the time this is called the rest of them have
10256/// been encoded and the raw bytes are gone.
10257fn settle_shape(sample: &[Vec<&[u8]>]) -> Result<chooser::Settled> {
10258    let mut best: Option<(chooser::Settled, usize)> = None;
10259    for shape in payload_shapes() {
10260        let mut size = 0;
10261        for block in sample {
10262            size += string::encode_with(block, &shape)?.len();
10263        }
10264        if best.as_ref().is_none_or(|(_, smallest)| size < *smallest) {
10265            best = Some((shape, size));
10266        }
10267    }
10268    best.map(|(shape, _)| shape)
10269        .ok_or_else(|| invalid("no shape applies to a global dictionary payload"))
10270}
10271
10272/// The sorted order laid out the way a reader reads it, in blocks of [`TEXT_RANK_BLOCK`] entries.
10273///
10274/// Each block holds its heads first and then its codes, rather than pairing them, because a search
10275/// asks for a head at every probe and for a code about once a search. Keeping the heads together
10276/// means a probe touches eight bytes of a block rather than twelve spread over it, and the last few
10277/// probes of a search, which are the ones that land in the same block, touch the same cache line.
10278fn encode_ranks(order: &[(u64, u32)], code_bits: usize) -> Result<(Vec<u8>, Vec<u64>)> {
10279    let mut out = Vec::with_capacity(order.len() * 4);
10280    let mut ends = Vec::with_capacity(order.len().div_ceil(TEXT_RANK_BLOCK));
10281    let mut heads = Vec::with_capacity(TEXT_RANK_BLOCK);
10282    let mut codes = Vec::with_capacity(TEXT_RANK_BLOCK);
10283    for block in order.chunks(TEXT_RANK_BLOCK) {
10284        // The order is sorted by value and a head is a prefix of a value, so the heads of a block
10285        // rise, the smallest is the first and the largest is the last.
10286        let base = block.first().map_or(0, |&(head, _)| head);
10287        let span = block.last().map_or(0, |&(head, _)| head.wrapping_sub(base));
10288        let width = (u64::BITS - span.leading_zeros()) as usize;
10289        heads.clear();
10290        codes.clear();
10291        for &(head, code) in block {
10292            heads.push(head.wrapping_sub(base));
10293            codes.push(u64::from(code));
10294        }
10295        put_u64(&mut out, base);
10296        out.push(width as u8);
10297        bitpack::pack_tail(&heads, width, &mut out)
10298            .map_err(|_| invalid("global dictionary heads do not pack"))?;
10299        bitpack::pack_tail(&codes, code_bits, &mut out)
10300            .map_err(|_| invalid("global dictionary codes do not pack"))?;
10301        ends.push(out.len() as u64);
10302    }
10303    Ok((out, ends))
10304}
10305
10306/// Opens a column's global dictionary, which reads its index and none of its payload.
10307///
10308/// `keep_budget` is how many decoded payload bytes this dictionary may hold on to, and every
10309/// caller bar the test of the ceiling passes [`TEXT_KEEP_BUDGET`]. It is a parameter rather than
10310/// the constant read where it is used because a test of a ceiling that cannot be moved has to build
10311/// a quarter of a gigabyte of dictionary to reach it.
10312fn open_global_dictionary(
10313    file: Arc<File>,
10314    page: Page,
10315    ty: &LogicalType,
10316    keep_budget: usize,
10317) -> Result<Vector> {
10318    if ty != &LogicalType::Varchar {
10319        return Err(invalid("global dictionary belongs to a non-string column"));
10320    }
10321    let mut header = [0; DICTIONARY_HEADER];
10322    read_at(&file, page.offset, &mut header)?;
10323    let count = u32::from_le_bytes(header[0..4].try_into().expect("four bytes")) as usize;
10324    let per_block = u32::from_le_bytes(header[4..8].try_into().expect("four bytes")) as usize;
10325    let blocks = u32::from_le_bytes(header[8..12].try_into().expect("four bytes")) as usize;
10326    let width = u32::from_le_bytes(header[12..16].try_into().expect("four bytes"));
10327    let scattered = width & DICTIONARY_SCATTERED != 0;
10328    let has_grams = width & DICTIONARY_GRAMS != 0;
10329    let offset_bits = (width & !(DICTIONARY_SCATTERED | DICTIONARY_GRAMS)) as usize;
10330    if per_block != TEXT_PAYLOAD_VALUES {
10331        return Err(invalid("global dictionary block width differs"));
10332    }
10333    if blocks != count.div_ceil(TEXT_PAYLOAD_VALUES) {
10334        return Err(invalid("global dictionary block count differs from its value count"));
10335    }
10336    if offset_bits > u32::BITS as usize {
10337        return Err(invalid("global dictionary packs offsets past a payload"));
10338    }
10339    let offset_len = offset_bytes(count, offset_bits);
10340    // The sorted order is kept out of the index on purpose. The index is read and checksummed in
10341    // full the moment the column is first touched, and the order is half again the size of the
10342    // offsets, so putting it there would make every query that reads a string column pay for a
10343    // search that most of them never make.
10344    let ranks = count;
10345    let rank_blocks = ranks.div_ceil(TEXT_RANK_BLOCK);
10346    // Three words a payload block, for where it starts, how long it is and what it hashes to, or
10347    // two of them on a file that has the blocks back to back and needs no start. Two a rank block
10348    // either way, since those are still one run.
10349    let payload_words = if scattered { 3 } else { 2 };
10350    let hash_len = blocks
10351        .checked_mul(payload_words * 8)
10352        .and_then(|len| len.checked_add(rank_blocks.checked_mul(16)?))
10353        .and_then(|len| len.checked_add(usize::from(has_grams) * 8))
10354        .ok_or_else(|| invalid("global dictionary block count overflow"))?;
10355    let gram_len = if has_grams {
10356        blocks
10357            .checked_mul(TEXT_GRAM_BYTES)
10358            .ok_or_else(|| invalid("global dictionary signature count overflow"))?
10359    } else {
10360        0
10361    };
10362    let index_len = DICTIONARY_HEADER
10363        .checked_add(offset_len)
10364        .and_then(|len| len.checked_add(hash_len))
10365        .ok_or_else(|| invalid("global dictionary header overflow"))?;
10366    if index_len > page.length as usize {
10367        return Err(invalid("global dictionary offset index exceeds its page"));
10368    }
10369    let mut index = vec![0; index_len];
10370    index[..DICTIONARY_HEADER].copy_from_slice(&header);
10371    read_at(&file, page.offset + DICTIONARY_HEADER as u64, &mut index[DICTIONARY_HEADER..])?;
10372    if checksum(&index) != page.hash {
10373        return Err(invalid("global dictionary index checksum differs"));
10374    }
10375    let offsets = index[DICTIONARY_HEADER..DICTIONARY_HEADER + offset_len].to_vec();
10376    let word_end = index_len - usize::from(has_grams) * 8;
10377    let gram_hash = has_grams
10378        .then(|| u64::from_le_bytes(index[word_end..index_len].try_into().expect("eight bytes")));
10379    let mut words = index[DICTIONARY_HEADER + offset_len..word_end]
10380        .chunks_exact(8)
10381        .map(|part| u64::from_le_bytes(part.try_into().expect("eight bytes")))
10382        .collect::<Vec<_>>();
10383    let mut rest = words.split_off(blocks * payload_words);
10384    let rank_hashes = rest.split_off(rank_blocks);
10385    let rank_ends = rest;
10386    // A rank block packs its heads at whatever width its own values need, so its length is no longer
10387    // arithmetic on the block number and the reader has to be told where each one ends.
10388    if rank_ends.windows(2).any(|pair| pair[0] >= pair[1]) {
10389        return Err(invalid("global dictionary order blocks do not rise"));
10390    }
10391    let rank_len = usize::try_from(rank_ends.last().copied().unwrap_or_default())
10392        .map_err(|_| invalid("global dictionary rank overflow"))?;
10393    let body_len = index_len
10394        .checked_add(rank_len)
10395        .ok_or_else(|| invalid("global dictionary header overflow"))?;
10396    if body_len > page.length as usize {
10397        return Err(invalid("global dictionary order exceeds its page"));
10398    }
10399    let gram_end = body_len
10400        .checked_add(gram_len)
10401        .ok_or_else(|| invalid("global dictionary signature length overflow"))?;
10402    if gram_end > page.length as usize {
10403        return Err(invalid("global dictionary signatures exceed their page"));
10404    }
10405    let grams = gram_hash.map(|hash| NativeGrams {
10406        start: page.offset + body_len as u64,
10407        length: gram_len,
10408        hash,
10409        loaded: OnceLock::new(),
10410    });
10411    let hashes = words.split_off(blocks * (payload_words - 1));
10412    let (starts, lengths) = if scattered {
10413        let mut starts = Vec::with_capacity(blocks);
10414        let mut lengths = Vec::with_capacity(blocks);
10415        for pair in words.chunks_exact(2) {
10416            starts.push(pair[0]);
10417            lengths.push(pair[1]);
10418        }
10419        (starts, lengths)
10420    } else {
10421        // A file written before the blocks said where they were has them behind one another at the
10422        // end of the page, so the base is where the sorted order stops and each end is the start of
10423        // the one after it. Turning them round here is what lets everything below take one shape.
10424        let base = page.offset + gram_end as u64;
10425        let mut starts = Vec::with_capacity(blocks);
10426        let mut lengths = Vec::with_capacity(blocks);
10427        let mut at = 0_u64;
10428        for &end in &words {
10429            let len = end
10430                .checked_sub(at)
10431                .ok_or_else(|| invalid("global dictionary block ends before it starts"))?;
10432            starts.push(base + at);
10433            lengths.push(len);
10434            at = end;
10435        }
10436        (starts, lengths)
10437    };
10438    // What the offsets bound is the decoded payload, and what the page length counts is the stored
10439    // one, so on a format 26 file the block lengths adding up to the rest of the page is the one
10440    // thing that ties the index to the page. From format 27 the blocks are written during the load
10441    // and the page is only the index and the order, so there the most that can be said is that
10442    // every block is somewhere in the file past its header.
10443    let stored_len = page.length as u64 - gram_end as u64;
10444    if scattered && stored_len == 0 {
10445        let size = file.metadata().map_err(io)?.len();
10446        let inside = starts.iter().zip(&lengths).all(|(&start, &len)| {
10447            start >= HEADER && start.checked_add(len).is_some_and(|end| end <= size)
10448        });
10449        if !inside {
10450            return Err(invalid("global dictionary block lies outside the file"));
10451        }
10452    } else if lengths.iter().try_fold(0_u64, |sum, len| sum.checked_add(*len)) != Some(stored_len) {
10453        return Err(invalid("global dictionary blocks do not bound the payload"));
10454    }
10455    Vector::external_text(
10456        LogicalType::Varchar,
10457        Arc::new(NativeText {
10458            file,
10459            values: count,
10460            offsets,
10461            offset_bits,
10462            value_ends: OnceLock::new(),
10463            value_lens: OnceLock::new(),
10464            ends_asked: AtomicUsize::new(0),
10465            ranks,
10466            rank_at: page.offset + index_len as u64,
10467            rank_ends,
10468            rank_hashes,
10469            rank_blocks: (0..rank_blocks).map(|_| OnceLock::new()).collect(),
10470            code_bits: code_width(count),
10471            code_ranks: OnceLock::new(),
10472            starts,
10473            lengths,
10474            hashes,
10475            grams,
10476            blocks: (0..blocks).map(|_| OnceLock::new()).collect(),
10477            keep_budget,
10478            payload_kept: AtomicUsize::new(0),
10479            swept: (0..blocks).map(|_| AtomicBool::new(false)).collect(),
10480            searched: Mutex::new(HashMap::new()),
10481        }),
10482    )
10483}
10484
10485/// What a stored page is, without decoding a value out of it.
10486///
10487/// Two layers, and both of them belong in the answer. The codec byte at the front of every page is
10488/// the format's own choice, and it is what says whether the column came back as codes into a table
10489/// wide dictionary, as a bit packed page, as an encoding cascade or as the bytes themselves. Under
10490/// the cascade codecs there is a second choice the encoder made per chunk, and that is what
10491/// [`integer::describe`] and [`string::describe`] already write out as `DICT(PACKED, PACKED)`.
10492///
10493/// This mirrors the tags [`decode`] reads and has to be kept beside it. A page whose header this
10494/// cannot walk comes back as text rather than as an error, because a caller asking what a file
10495/// looks like is usually asking because something is wrong with it, and a report that stops at the
10496/// first bad page is a report that says nothing about the other nine hundred.
10497fn page_encoding(ty: &LogicalType, rows: usize, bytes: &[u8]) -> String {
10498    /// The page header is the codec, the validity tag and, for a page that stores a mask, the mask.
10499    fn cascade_at(rows: usize, bytes: &[u8]) -> Result<(u8, usize)> {
10500        let mut cur = Cursor::new(bytes);
10501        let codec = cur.u8()?;
10502        if cur.u8()? == 2 {
10503            cur.take(rows.div_ceil(8))?;
10504        }
10505        Ok((codec, cur.at))
10506    }
10507    let Ok((codec, at)) = cascade_at(rows, bytes) else {
10508        return "UNREADABLE".to_string();
10509    };
10510    let tail = &bytes[at..];
10511    let described = |described: Result<String>| described.unwrap_or_else(|_| "UNREADABLE".into());
10512    match codec {
10513        0 => match ty {
10514            LogicalType::Varchar | LogicalType::Blob => "PLAIN".to_string(),
10515            _ => "FIXED".to_string(),
10516        },
10517        1 => "DICT(PLAIN)".to_string(),
10518        2 => "FOR+BITPACK".to_string(),
10519        3 => "TABLE DICT".to_string(),
10520        4 => format!("TABLE DICT({})", described(integer::describe(tail))),
10521        5 => described(integer::describe(tail)),
10522        6 => described(string::describe(tail)),
10523        other => format!("CODEC {other}"),
10524    }
10525}
10526
10527/// Selected stable dictionary codes from one page.
10528///
10529/// Pair-frequency construction needs at most the bounded heavy-hitter rows. Reading those code
10530/// positions directly avoids materializing every code in each part that contains a candidate.
10531fn decode_selected_stable_codes(
10532    rows: usize,
10533    bytes: &[u8],
10534    positions: &[usize],
10535    out: &mut Vec<Option<u32>>,
10536) -> Result<bool> {
10537    if positions.windows(2).any(|pair| pair[0] >= pair[1])
10538        || positions.last().is_some_and(|&position| position >= rows)
10539    {
10540        return Err(invalid("selected code positions are not sorted and in range"));
10541    }
10542    let mut cur = Cursor::new(bytes);
10543    let codec = cur.u8()?;
10544    if codec != 3 && codec != 4 {
10545        return Ok(false);
10546    }
10547    let flag = cur.u8()?;
10548    let mask = match flag {
10549        0 | 1 => None,
10550        2 => {
10551            let at = cur.at;
10552            let len = rows.div_ceil(8);
10553            cur.take(len)?;
10554            Some((at, len))
10555        }
10556        _ => return Err(invalid("page validity tag differs")),
10557    };
10558    let valid = |row: usize| match flag {
10559        0 => true,
10560        1 => false,
10561        2 => mask.is_some_and(|(at, _)| bytes[at + row / 8] >> (row % 8) & 1 == 1),
10562        _ => unreachable!("the validity tag was checked"),
10563    };
10564    if codec == 4 {
10565        let wide = integer::decode_selected(&bytes[cur.at..], positions)?;
10566        for (&row, code) in positions.iter().zip(wide) {
10567            let code = u32::try_from(code).map_err(|_| invalid("code is not a code"))?;
10568            out.push(valid(row).then_some(code));
10569        }
10570        return Ok(true);
10571    }
10572    let codes_at = cur.at;
10573    let codes_len = rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?;
10574    cur.take(codes_len)?;
10575    if cur.at != bytes.len() {
10576        return Err(invalid("global code page has trailing bytes"));
10577    }
10578    let codes = &bytes[codes_at..codes_at + codes_len];
10579    for &row in positions {
10580        let at = row.checked_mul(4).ok_or_else(|| invalid("dictionary code offset overflow"))?;
10581        let code = u32::from_le_bytes(
10582            codes[at..at + 4].try_into().map_err(|_| invalid("dictionary code is truncated"))?,
10583        );
10584        out.push(valid(row).then_some(code));
10585    }
10586    Ok(true)
10587}
10588
10589fn decode(
10590    ty: &LogicalType,
10591    rows: usize,
10592    bytes: &[u8],
10593    global: Option<Arc<Vector>>,
10594) -> Result<Vector> {
10595    let mut cur = Cursor::new(bytes);
10596    let codec = cur.u8()?;
10597    let flag = cur.u8()?;
10598    let validity = match flag {
10599        0 => Validity::AllValid,
10600        1 => Validity::AllInvalid,
10601        2 => {
10602            let mask = cur.take(rows.div_ceil(8))?;
10603            Validity::from_iter(rows, |row| mask[row / 8] >> (row % 8) & 1 == 1)
10604        }
10605        _ => return Err(invalid("page validity tag differs")),
10606    };
10607    if codec == 1 {
10608        if ty != &LogicalType::Varchar {
10609            return Err(invalid("dictionary codec belongs to a non-string page"));
10610        }
10611        let count = cur.u32()? as usize;
10612        let payload_len = cur.u32()? as usize;
10613        let offset_bytes = cur.take(
10614            (count + 1)
10615                .checked_mul(4)
10616                .ok_or_else(|| invalid("dictionary offset count overflow"))?,
10617        )?;
10618        let offsets = offset_bytes
10619            .chunks_exact(4)
10620            .map(|part| u32::from_le_bytes(part.try_into().expect("four bytes")))
10621            .collect::<Vec<_>>();
10622        let payload = cur.take(payload_len)?.to_vec();
10623        if offsets.first() != Some(&0)
10624            || offsets.last().copied().map(|last| last as usize) != Some(payload.len())
10625            || offsets.windows(2).any(|pair| pair[0] > pair[1])
10626        {
10627            return Err(invalid("dictionary offsets do not bound the payload"));
10628        }
10629        // A page, because every chunk cut out of this dictionary points at the same payload and a
10630        // page is what lets a cut be the views and nothing else.
10631        let mut strings = StringColumn::over(Buffer::from_vec(payload).into_page());
10632        for pair in offsets.windows(2) {
10633            strings.push_in_place(pair[0] as usize, (pair[1] - pair[0]) as usize)?;
10634        }
10635        let mut codes = Vec::with_capacity(rows);
10636        for _ in 0..rows {
10637            codes.push(cur.u32()?);
10638        }
10639        if codes.iter().any(|code| *code as usize >= count) {
10640            return Err(invalid("dictionary code is out of range"));
10641        }
10642        if cur.at != bytes.len() {
10643            return Err(invalid("dictionary page has trailing bytes"));
10644        }
10645        let dictionary = Vector::flat(LogicalType::Varchar, Data::Varlen(strings))?;
10646        return Ok(Vector::dictionary(codes, dictionary)?.with_validity(validity));
10647    }
10648    if codec == 3 || codec == 4 {
10649        let dictionary = global.ok_or_else(|| invalid("global code page has no dictionary"))?;
10650        let codes = if codec == 4 {
10651            // The cascade holds the whole tail of the page and says how long it is itself, so the
10652            // check that nothing is left over is the one the decoder already makes.
10653            let wide = integer::decode(&bytes[cur.at..])?;
10654            if wide.len() != rows {
10655                return Err(invalid("encoded code page holds the wrong number of rows"));
10656            }
10657            // Checked once for the page rather than a fallible conversion per code. Every code a
10658            // file holds is inside a `u32` or the file is corrupt, so or the codes together and the
10659            // answer has a bit set above the low thirty two, or the sign bit, exactly when one of
10660            // them did. The or and the narrowing are two passes because each is then a vector
10661            // loop. As one loop with a `push` a code, the length check and the store kept it scalar,
10662            // and it was sixteen instructions a row on the two flag columns of q1.
10663            let seen = wide.iter().fold(0_i64, |seen, &code| seen | code);
10664            if seen < 0 || seen > i64::from(u32::MAX) {
10665                return Err(invalid("code is not a code"));
10666            }
10667            wide.iter().map(|&code| code as u32).collect()
10668        } else {
10669            let mut codes = Vec::with_capacity(rows);
10670            for _ in 0..rows {
10671                codes.push(cur.u32()?);
10672            }
10673            if cur.at != bytes.len() {
10674                return Err(invalid("global code page has trailing bytes"));
10675            }
10676            codes
10677        };
10678        let highest = codes.iter().copied().max();
10679        return Ok(Vector::stable_dictionary_validated(codes, dictionary, highest)?
10680            .with_validity(validity));
10681    }
10682    if codec == 6 {
10683        if ty != &LogicalType::Varchar {
10684            return Err(invalid("compressed text codec belongs to a non-string page"));
10685        }
10686        // As codec 5, the layer holds the whole tail of the page and says how long it is itself.
10687        // It comes back as one buffer with the values laid end to end and where each one ends, which
10688        // is the raw form's layout, so what is left to do here is what codec 0 does.
10689        let (payload, ends) = string::decode_flat(&bytes[cur.at..])?.into_parts();
10690        if ends.len() != rows {
10691            return Err(invalid("compressed text page holds the wrong number of rows"));
10692        }
10693        // A page, because this is read once and handed out a chunk at a time, and a cut of a paged
10694        // payload moves views rather than bytes.
10695        let mut values = StringColumn::over(Buffer::from_vec(payload).into_page());
10696        let mut start = 0;
10697        for end in ends {
10698            let len = end
10699                .checked_sub(start)
10700                .ok_or_else(|| invalid("compressed text value ends before it starts"))?;
10701            values.push_in_place(start, len)?;
10702            start = end;
10703        }
10704        return Ok(Vector::flat(ty.clone(), Data::Varlen(values))?.with_validity(validity));
10705    }
10706    if codec == 5 {
10707        // The cascade holds the whole tail of the page and says how long it is itself.
10708        let values = integer::decode(&bytes[cur.at..])?;
10709        if values.len() != rows {
10710            return Err(invalid("cascade page holds the wrong number of rows"));
10711        }
10712        let data = narrowed(ty, values)?;
10713        return Ok(Vector::flat(ty.clone(), data)?.with_validity(validity));
10714    }
10715    if codec == 2 {
10716        let width = u32::from(cur.u8()?);
10717        let base = i128::from_le_bytes(cur.take(16)?.try_into().expect("sixteen bytes"));
10718        let count = cur.u32()? as usize;
10719        let length = count.checked_mul(8).ok_or_else(|| invalid("packed page is too long"))?;
10720        let words: Vec<u64> = cur
10721            .take(length)?
10722            .chunks_exact(8)
10723            .map(|word| u64::from_le_bytes(word.try_into().expect("eight bytes")))
10724            .collect();
10725        if cur.at != bytes.len() {
10726            return Err(invalid("packed page has trailing bytes"));
10727        }
10728        return Ok(Vector::packed(ty.clone(), words, width, base, rows)?.with_validity(validity));
10729    }
10730    if codec != 0 {
10731        return Err(invalid("page codec is unknown"));
10732    }
10733    let data = match ty {
10734        LogicalType::TinyInt => {
10735            let values = cur.take(rows)?;
10736            Data::Int8(values.iter().map(|item| *item as i8).collect::<Vec<_>>().into())
10737        }
10738        LogicalType::UTinyInt => Data::UInt8(cur.take(rows)?.to_vec().into()),
10739        LogicalType::SmallInt => {
10740            let values =
10741                cur.take(rows.checked_mul(2).ok_or_else(|| invalid("page size overflow"))?)?;
10742            Data::Int16(
10743                values
10744                    .chunks_exact(2)
10745                    .map(|item| i16::from_le_bytes(item.try_into().expect("two bytes")))
10746                    .collect::<Vec<_>>()
10747                    .into(),
10748            )
10749        }
10750        LogicalType::USmallInt => {
10751            let values =
10752                cur.take(rows.checked_mul(2).ok_or_else(|| invalid("page size overflow"))?)?;
10753            Data::UInt16(
10754                values
10755                    .chunks_exact(2)
10756                    .map(|item| u16::from_le_bytes(item.try_into().expect("two bytes")))
10757                    .collect::<Vec<_>>()
10758                    .into(),
10759            )
10760        }
10761        LogicalType::UInteger => {
10762            let values =
10763                cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
10764            Data::UInt32(
10765                values
10766                    .chunks_exact(4)
10767                    .map(|item| u32::from_le_bytes(item.try_into().expect("four bytes")))
10768                    .collect::<Vec<_>>()
10769                    .into(),
10770            )
10771        }
10772        LogicalType::UBigInt => {
10773            let values =
10774                cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
10775            Data::UInt64(
10776                values
10777                    .chunks_exact(8)
10778                    .map(|item| u64::from_le_bytes(item.try_into().expect("eight bytes")))
10779                    .collect::<Vec<_>>()
10780                    .into(),
10781            )
10782        }
10783        LogicalType::Integer | LogicalType::Date => {
10784            let values =
10785                cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
10786            Data::Int32(
10787                values
10788                    .chunks_exact(4)
10789                    .map(|item| i32::from_le_bytes(item.try_into().expect("four bytes")))
10790                    .collect::<Vec<_>>()
10791                    .into(),
10792            )
10793        }
10794        LogicalType::BigInt
10795        | LogicalType::Timestamp
10796        | LogicalType::Time
10797        | LogicalType::TimeTz
10798        | LogicalType::TimestampTz
10799        | LogicalType::TimestampS
10800        | LogicalType::TimestampMs
10801        | LogicalType::TimestampNs => {
10802            let values =
10803                cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
10804            Data::Int64(
10805                values
10806                    .chunks_exact(8)
10807                    .map(|item| i64::from_le_bytes(item.try_into().expect("eight bytes")))
10808                    .collect::<Vec<_>>()
10809                    .into(),
10810            )
10811        }
10812        LogicalType::HugeInt | LogicalType::Uuid => {
10813            let values =
10814                cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
10815            Data::Int128(
10816                values
10817                    .chunks_exact(16)
10818                    .map(|item| i128::from_le_bytes(item.try_into().expect("sixteen bytes")))
10819                    .collect::<Vec<_>>()
10820                    .into(),
10821            )
10822        }
10823        LogicalType::UHugeInt => {
10824            let values =
10825                cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
10826            Data::UInt128(
10827                values
10828                    .chunks_exact(16)
10829                    .map(|item| u128::from_le_bytes(item.try_into().expect("sixteen bytes")))
10830                    .collect::<Vec<_>>()
10831                    .into(),
10832            )
10833        }
10834        LogicalType::Float => {
10835            let values =
10836                cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
10837            Data::Float32(
10838                values
10839                    .chunks_exact(4)
10840                    .map(|item| f32::from_le_bytes(item.try_into().expect("four bytes")))
10841                    .collect::<Vec<_>>()
10842                    .into(),
10843            )
10844        }
10845        LogicalType::Double => {
10846            let values =
10847                cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
10848            Data::Float64(
10849                values
10850                    .chunks_exact(8)
10851                    .map(|item| f64::from_le_bytes(item.try_into().expect("eight bytes")))
10852                    .collect::<Vec<_>>()
10853                    .into(),
10854            )
10855        }
10856        LogicalType::Interval => {
10857            let values =
10858                cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
10859            Data::Interval(
10860                values
10861                    .chunks_exact(16)
10862                    .map(|item| {
10863                        (
10864                            i32::from_le_bytes(item[..4].try_into().expect("four bytes")),
10865                            i32::from_le_bytes(item[4..8].try_into().expect("four bytes")),
10866                            i64::from_le_bytes(item[8..].try_into().expect("eight bytes")),
10867                        )
10868                    })
10869                    .collect::<Vec<_>>()
10870                    .into(),
10871            )
10872        }
10873        LogicalType::Boolean => {
10874            let values = cur.take(rows)?;
10875            if values.iter().any(|value| *value > 1) {
10876                return Err(invalid("boolean page has another value"));
10877            }
10878            Data::Bool(values.iter().map(|value| *value == 1).collect::<Vec<_>>().into())
10879        }
10880        // Whichever integer the declared width says, which is the mapping the rest of the engine
10881        // already uses for a decimal in memory.
10882        LogicalType::Decimal { .. } => match ty.physical() {
10883            PhysicalType::Int16 => {
10884                let values =
10885                    cur.take(rows.checked_mul(2).ok_or_else(|| invalid("page size overflow"))?)?;
10886                Data::Int16(
10887                    values
10888                        .chunks_exact(2)
10889                        .map(|item| i16::from_le_bytes(item.try_into().expect("two bytes")))
10890                        .collect::<Vec<_>>()
10891                        .into(),
10892                )
10893            }
10894            PhysicalType::Int32 => {
10895                let values =
10896                    cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
10897                Data::Int32(
10898                    values
10899                        .chunks_exact(4)
10900                        .map(|item| i32::from_le_bytes(item.try_into().expect("four bytes")))
10901                        .collect::<Vec<_>>()
10902                        .into(),
10903                )
10904            }
10905            PhysicalType::Int64 => {
10906                let values =
10907                    cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
10908                Data::Int64(
10909                    values
10910                        .chunks_exact(8)
10911                        .map(|item| i64::from_le_bytes(item.try_into().expect("eight bytes")))
10912                        .collect::<Vec<_>>()
10913                        .into(),
10914                )
10915            }
10916            _ => {
10917                let values =
10918                    cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
10919                Data::Int128(
10920                    values
10921                        .chunks_exact(16)
10922                        .map(|item| i128::from_le_bytes(item.try_into().expect("sixteen bytes")))
10923                        .collect::<Vec<_>>()
10924                        .into(),
10925                )
10926            }
10927        },
10928        LogicalType::Varchar | LogicalType::Blob | LogicalType::Bit => {
10929            let offset_bytes = cur
10930                .take((rows + 1).checked_mul(4).ok_or_else(|| invalid("offset count overflow"))?)?;
10931            let offsets = offset_bytes
10932                .chunks_exact(4)
10933                .map(|part| u32::from_le_bytes(part.try_into().expect("four bytes")))
10934                .collect::<Vec<_>>();
10935            let payload = cur.take(bytes.len() - cur.at)?.to_vec();
10936            if offsets.first() != Some(&0)
10937                || offsets.last().copied().map(|last| last as usize) != Some(payload.len())
10938                || offsets.windows(2).any(|pair| pair[0] > pair[1])
10939            {
10940                return Err(invalid("string offsets do not bound the payload"));
10941            }
10942            // A page for the reason the dictionary payload above is one: the page is read once and
10943            // handed out a chunk at a time, and a cut of a paged payload moves views rather than
10944            // bytes.
10945            //
10946            // A varchar is checked for text on the way in and a blob and a bit string are not,
10947            // because the second pair never claimed to hold any. Reading them through the checking
10948            // seam would refuse a column for holding exactly what it was told to hold.
10949            let mut values = StringColumn::over(Buffer::from_vec(payload).into_page());
10950            let text = ty == &LogicalType::Varchar;
10951            for pair in offsets.windows(2) {
10952                let (at, len) = (pair[0] as usize, (pair[1] - pair[0]) as usize);
10953                if text {
10954                    values.push_in_place(at, len)?;
10955                } else {
10956                    values.push_bytes_in_place(at, len)?;
10957                }
10958            }
10959            Data::Varlen(values)
10960        }
10961        _ => return Err(Error::not_implemented(format!("native page for {ty}"))),
10962    };
10963    if cur.at != bytes.len() {
10964        return Err(invalid("page has trailing bytes"));
10965    }
10966    Ok(Vector::flat(ty.clone(), data)?.with_validity(validity))
10967}
10968
10969#[cfg(test)]
10970mod tests {
10971    use std::fs;
10972    use std::io::{Seek, SeekFrom, Write};
10973    use std::path::PathBuf;
10974    use std::time::{SystemTime, UNIX_EPOCH};
10975
10976    use rudb_common::Stat;
10977    use rudb_common::Value;
10978    use rudb_common::bounds::{Frequencies, Op, Remainder, Zones};
10979    use rudb_common::stat::Provenance;
10980
10981    use super::*;
10982
10983    #[test]
10984    fn a_name_taken_in_pieces_is_the_name_of_the_pieces_joined() {
10985        let bytes: Vec<u8> =
10986            (0..300_u32).map(|at| (at.wrapping_mul(2_654_435_761) >> 13) as u8).collect();
10987        for length in [0, 1, 7, 31, 32, 33, 63, 64, 65, 100, 300] {
10988            let whole = content_name(&bytes[..length]);
10989            for step in [1, 3, 8, 31, 32, 33, 64, 301] {
10990                let mut namer = ContentNamer::default();
10991                bytes[..length].chunks(step).for_each(|piece| namer.update(piece));
10992                assert_eq!(namer.finish(), whole, "{length} bytes in pieces of {step}");
10993            }
10994        }
10995    }
10996
10997    /// The chooser as it was before it could rule kinds out up front: the same narrowing, with every
10998    /// kind tested for. What it writes is what the file used to hold.
10999    #[derive(Debug)]
11000    struct TestsEverything<'a>(&'a dyn chooser::Chooser);
11001
11002    impl chooser::Chooser for TestsEverything<'_> {
11003        fn name(&self) -> &'static str {
11004            "tests everything"
11005        }
11006
11007        fn narrow_strings(
11008            &self,
11009            values: &[&[u8]],
11010            offered: &[string::Kind],
11011            depth: u8,
11012        ) -> Vec<string::Kind> {
11013            self.0.narrow_strings(values, offered, depth)
11014        }
11015
11016        fn narrow_integers(
11017            &self,
11018            values: &[i64],
11019            offered: &[integer::Kind],
11020            depth: u8,
11021        ) -> Vec<integer::Kind> {
11022            self.0.narrow_integers(values, offered, depth)
11023        }
11024    }
11025
11026    #[test]
11027    fn ruling_kinds_out_before_testing_for_them_writes_the_same_bytes() {
11028        let columns: Vec<Vec<i64>> = vec![
11029            vec![],
11030            vec![5; 1000],
11031            (0..1000).collect(),
11032            (0..1000).map(|row| 1_600_000_000_000_000 + row * 1_000_000).collect(),
11033            (0..1000).map(|row| row / 50).collect(),
11034            (0..1000).map(|row| if row % 97 == 0 { row } else { 0 }).collect(),
11035            (0..1000).map(|row| (row * 7919) % 13).collect(),
11036            (0..1000).map(|row| (row * 2_654_435_761) % 1_000_003).collect(),
11037            (0..1000).map(|row| [3, 3, 3, 9, 9, 1][row as usize % 6]).collect(),
11038            (0..1000).map(|row| i64::MIN + row % 3).collect(),
11039        ];
11040        let choosers: [&dyn chooser::Chooser; 2] = [&Fixed, &Codes];
11041        for column in &columns {
11042            for chooser in choosers {
11043                let quick = integer::encode_with(column, chooser).unwrap();
11044                let full = integer::encode_with(column, &TestsEverything(chooser)).unwrap();
11045                assert_eq!(
11046                    quick,
11047                    full,
11048                    "{} on {:?}",
11049                    chooser.name(),
11050                    &column[..column.len().min(8)]
11051                );
11052            }
11053        }
11054    }
11055
11056    /// Parts of a column that all look alike come out of a settled shape byte for byte as they
11057    /// come out of a search, because the search would have kept the same tree on every one.
11058    #[test]
11059    fn parts_that_look_alike_replay_to_the_bytes_a_search_writes() {
11060        let mut settling = Settling::default();
11061        for part in 0..STRIPE_PARTS as i64 {
11062            let values: Vec<i64> = (0..2048)
11063                .map(|row| 1_600_000_000_000_000 + (part * 2048 + row) * 1_000_000 + row % 7)
11064                .collect();
11065            let searched = integer::encode_with(&values, &Fixed).unwrap();
11066            assert_eq!(settling.encode(&values).unwrap(), searched, "part {part}");
11067        }
11068    }
11069
11070    /// A column that changes shape partway through a stripe still reads back, and no part comes
11071    /// out much bigger than a search would have made it, because a replay that stops fitting or
11072    /// grows past a quarter a row is searched.
11073    #[test]
11074    fn a_column_that_changes_under_the_shape_is_searched_again() {
11075        let mut state = 0x9e37_79b9_7f4a_7c15_u64;
11076        let mut noise = move || {
11077            state ^= state << 13;
11078            state ^= state >> 7;
11079            state ^= state << 17;
11080            (state % 1_000_000) as i64
11081        };
11082        let mut settling = Settling::default();
11083        for part in 0..STRIPE_PARTS as i64 {
11084            let values: Vec<i64> = match part / 16 {
11085                0 => (0..2048).map(|row| (part * 2048 + row) / 300).collect(),
11086                1 => (0..2048).map(|_| noise()).collect(),
11087                2 => (0..2048).map(|row| if row % 97 == 0 { row } else { 42 }).collect(),
11088                _ => (0..2048).map(|row| 5 + (part * 2048 + row) * 1_000_000).collect(),
11089            };
11090            let settled = settling.encode(&values).unwrap();
11091            assert_eq!(integer::decode(&settled).unwrap(), values, "part {part}");
11092            let searched = integer::encode_with(&values, &Fixed).unwrap();
11093            assert!(
11094                settled.len() * 4 <= searched.len() * 5,
11095                "part {part}: {} settled against {} searched, {} against {}",
11096                settled.len(),
11097                searched.len(),
11098                integer::describe(&settled).unwrap(),
11099                integer::describe(&searched).unwrap(),
11100            );
11101        }
11102    }
11103
11104    #[test]
11105    fn checksum_matches_fixed_vectors() {
11106        assert_eq!(checksum(b""), 0xef46_db37_51d8_e999);
11107        assert_eq!(checksum(b"a"), 0xd24e_c4f1_a98c_6e5b);
11108        assert_eq!(checksum(b"abc"), 0x44bc_2cf5_ad77_0999);
11109    }
11110
11111    #[test]
11112    fn sorting_across_threads_matches_sorting_on_one() {
11113        let mut state = 0x9e37_79b9_7f4a_7c15_u64;
11114        let mut next = move || {
11115            state ^= state << 13;
11116            state ^= state >> 7;
11117            state ^= state << 17;
11118            state
11119        };
11120        let mut values = Vec::new();
11121        for at in 0..150_000_u64 {
11122            let value = match next() % 6 {
11123                0 => Vec::new(),
11124                1 => format!("https://example.com/{}", next() % 5_000).into_bytes(),
11125                2 => format!("https://example.com/path/{at}").into_bytes(),
11126                3 => b"same".to_vec(),
11127                4 => vec![0xff; (next() % 12) as usize],
11128                _ => (0..next() % 20).map(|_| (next() % 3) as u8).collect(),
11129            };
11130            values.push(value);
11131        }
11132        let value = |code: u32| values[code as usize].as_slice();
11133        for workers in [1, 2, 3, 8, 32] {
11134            let mut one = (0..values.len() as u32).rev().collect::<Vec<_>>();
11135            let mut across = one.clone();
11136            sort_by_value(&mut one, value);
11137            sort_by_value_across(&mut across, value, workers);
11138            assert_eq!(one, across, "{workers} workers");
11139        }
11140        let mut sorted = (0..values.len() as u32).collect::<Vec<_>>();
11141        sort_by_value_across(&mut sorted, value, 8);
11142        assert!(sorted.windows(2).all(|pair| value(pair[0]) <= value(pair[1])));
11143    }
11144
11145    fn path(label: &str) -> PathBuf {
11146        let stamp = SystemTime::now().duration_since(UNIX_EPOCH).expect("time advances").as_nanos();
11147        std::env::temp_dir().join(format!("rudb-native-{label}-{}-{stamp}.rdb", std::process::id()))
11148    }
11149
11150    /// Every value of a dictionary in code order, which the tests have no other way to ask for now
11151    /// that a dictionary does not keep the bytes of the values it has seen.
11152    ///
11153    /// Only valid once `finish_blocks` has run, because until then the last part block is still raw.
11154    fn dictionary_values(dictionary: &GlobalDictionary) -> Vec<Vec<u8>> {
11155        let (flat, bases) = dictionary.decoded(None).expect("the blocks decode");
11156        (0..dictionary.values())
11157            .map(|code| {
11158                let (from, to) = GlobalDictionary::value_span(&dictionary.ends, &bases, code);
11159                flat[from..to].to_vec()
11160            })
11161            .collect()
11162    }
11163
11164    /// The sections a test put in the table, which is every one the writer did not.
11165    ///
11166    /// A table now carries a summary and a sketch per column out of the write itself, and a test
11167    /// about the section table is not about those. Filtering by kind rather than by count, so a
11168    /// table that turns out to have no room for its summaries does not quietly change what these
11169    /// tests are asserting over.
11170    fn attached(table: &Table) -> Vec<&Section> {
11171        table.sections().iter().filter(|held| !held.among(section::STATISTICS_KINDS)).collect()
11172    }
11173
11174    /// A read names the offset it wants, so a cursor somebody else moved cannot reach it.
11175    #[test]
11176    fn a_read_at_an_offset_ignores_where_another_thread_left_the_cursor() {
11177        const SPANS: usize = 64;
11178        const SPAN: usize = 512;
11179        let path = path("positional");
11180        let content: Vec<u8> =
11181            (0..SPANS).flat_map(|span| std::iter::repeat_n(span as u8, SPAN)).collect();
11182        fs::write(&path, &content).expect("the file is written");
11183        let file = Arc::new(File::open(&path).expect("the file opens"));
11184        std::thread::scope(|scope| {
11185            for _ in 0..8 {
11186                let file = Arc::clone(&file);
11187                scope.spawn(move || {
11188                    for _ in 0..64 {
11189                        for span in 0..SPANS {
11190                            let mut bytes = [0_u8; SPAN];
11191                            read_at(&file, (span * SPAN) as u64, &mut bytes)
11192                                .expect("the span reads");
11193                            assert!(
11194                                bytes.iter().all(|byte| *byte == span as u8),
11195                                "span {span} came back as {}",
11196                                bytes[0],
11197                            );
11198                        }
11199                    }
11200                });
11201            }
11202        });
11203        let mut past = [0_u8; SPAN];
11204        let end = (SPANS * SPAN) as u64;
11205        let error = read_at(&file, end, &mut past).expect_err("a read past the end is refused");
11206        assert!(error.message().contains("ends before its declared length"), "{error}");
11207        drop(file);
11208        let _ = fs::remove_file(&path);
11209    }
11210
11211    /// The writer records where it put a page and puts it there, whatever the cursor is doing.
11212    ///
11213    /// The cursor is moved between the steps that record an offset, which is what reading the pages
11214    /// back to build the frequencies does on a platform with no `pread`. Without the fix the
11215    /// directory lands on top of a page and the file fails to reopen.
11216    #[test]
11217    fn a_writer_puts_a_page_where_it_said_it_did_wherever_the_cursor_has_got_to() {
11218        let path = path("cursor");
11219        let mut writer = Writer::create(
11220            &path,
11221            "items",
11222            vec![
11223                Field::required("id", LogicalType::Integer),
11224                Field::new("text", LogicalType::Varchar),
11225            ],
11226        )
11227        .expect("new file");
11228        writer.append(&sample()).expect("first part");
11229        writer.file.seek(SeekFrom::Start(0)).expect("the cursor goes back to the header");
11230        writer.append(&sample()).expect("second part");
11231        writer.file.seek(SeekFrom::Start(1)).expect("and somewhere useless again");
11232        writer.finish().expect("commit");
11233        let reader = Reader::open(&path).expect("reopen from disk");
11234        assert_eq!(reader.table().rows(), 6);
11235        let ids = reader.read(0, &[0]).expect("the integer page reads back");
11236        assert_eq!(ids.value_at(0, 0), Value::Integer(4));
11237        assert_eq!(ids.value_at(2, 0), Value::Integer(-2));
11238        let text = reader.read(1, &[1]).expect("the text page reads back");
11239        assert_eq!(text.value_at(1, 0), Value::Null);
11240        assert_eq!(text.value_at(2, 0), Value::Varchar("long text after a slash".into()));
11241        // Nothing the directory points at may run past the end of the file, which is the shape the
11242        // failure took: a page recorded at an offset the directory had already been written over.
11243        let end = reader.table().stripes().iter().flat_map(|stripe| {
11244            stripe
11245                .pages
11246                .iter()
11247                .map(|page| page.offset + u64::from(page.length))
11248                .chain(std::iter::once(stripe.index.offset + u64::from(stripe.index.length)))
11249        });
11250        let last = end.fold(HEADER, u64::max);
11251        let directory = fs::metadata(&path).expect("the file is there").len();
11252        assert!(last <= directory, "a page runs to {last} in a file of {directory} bytes");
11253        fs::remove_file(path).expect("remove scratch file");
11254    }
11255
11256    /// How long a global dictionary index is, read out of the page's own header.
11257    ///
11258    /// The tests below damage a byte of the order or of the payload, so they need to know where each
11259    /// one starts, and working it out here rather than writing a number down means adding something
11260    /// to the index does not quietly turn one of them into a test that damages the index instead.
11261    fn dictionary_index_len(header: &[u8; DICTIONARY_HEADER]) -> u64 {
11262        let count = u64::from(u32::from_le_bytes(header[0..4].try_into().expect("four bytes")));
11263        let blocks = u64::from(u32::from_le_bytes(header[8..12].try_into().expect("four bytes")));
11264        let width = u32::from_le_bytes(header[12..16].try_into().expect("four bytes"));
11265        let bits = (width & !(DICTIONARY_SCATTERED | DICTIONARY_GRAMS)) as usize;
11266        let payload_words = if width & DICTIONARY_SCATTERED == 0 { 2 } else { 3 };
11267        let rank_blocks = count.div_ceil(TEXT_RANK_BLOCK as u64);
11268        DICTIONARY_HEADER as u64
11269            + offset_bytes(count as usize, bits) as u64
11270            + blocks * payload_words * 8
11271            + rank_blocks * 16
11272            + if width & DICTIONARY_GRAMS == 0 { 0 } else { 8 }
11273    }
11274
11275    fn sample() -> Chunk {
11276        Chunk::new(vec![
11277            Vector::from_values(
11278                LogicalType::Integer,
11279                &[Value::Integer(4), Value::Integer(9), Value::Integer(-2)],
11280            )
11281            .expect("integers"),
11282            Vector::from_values(
11283                LogicalType::Varchar,
11284                &[
11285                    Value::Varchar("alpha".into()),
11286                    Value::Null,
11287                    Value::Varchar("long text after a slash".into()),
11288                ],
11289            )
11290            .expect("strings"),
11291        ])
11292        .expect("matching rows")
11293    }
11294
11295    fn sample_ids() -> Chunk {
11296        Chunk::new(vec![
11297            Vector::flat(LogicalType::Integer, Data::Int32(vec![7, 8, 9].into()))
11298                .expect("integers"),
11299        ])
11300        .expect("one column")
11301    }
11302
11303    #[test]
11304    fn the_planner_gets_the_null_count_off_the_same_directory_the_bounds_are_in() {
11305        // Six rows, two of them null. `IS NULL` used to get the same fifth any unreadable
11306        // condition gets, and the number was in the stripe entry next to the bounds all along.
11307        let path = path("nulls_for_the_planner");
11308        let mut writer =
11309            Writer::create(&path, "items", vec![Field::new("a", LogicalType::Integer)])
11310                .expect("new file");
11311        let rows = Chunk::new(vec![
11312            Vector::from_values(
11313                LogicalType::Integer,
11314                &[
11315                    Value::Integer(4),
11316                    Value::Null,
11317                    Value::Integer(9),
11318                    Value::Null,
11319                    Value::Integer(1),
11320                    Value::Integer(2),
11321                ],
11322            )
11323            .expect("integers"),
11324        ])
11325        .expect("one column");
11326        writer.append(&rows).expect("the only part");
11327        writer.finish().expect("commit");
11328        let reader = Reader::open(&path).expect("reopen from disk");
11329        let stripes = Stripes::new(reader);
11330        let column = stripes.column("a").expect("the file has that column");
11331        assert_eq!(stripes.nulls(column), Stat::exact(2, Provenance::NullCount));
11332        // A column the file does not have. Zero here would be a fact about a column that is not
11333        // there, which the planner would then divide by.
11334        assert_eq!(stripes.nulls(column + 1), Stat::Unknown);
11335        fs::remove_file(&path).expect("clean up");
11336    }
11337
11338    #[test]
11339    fn the_planner_gets_a_row_count_per_value_off_a_complete_synopsis() {
11340        // The whole of the frequency half of #1106, end to end over a real file. Six rows, three
11341        // of one value and two of another, and a complete synopsis because six rows is well inside
11342        // what the writer can account for. The estimate for `id = 4` is three rows rather than a
11343        // sixth of the table, and for a value the file does not hold it is none.
11344        let path = path("frequencies_for_the_planner");
11345        let mut writer =
11346            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
11347                .expect("new file");
11348        let rows = Chunk::new(vec![
11349            Vector::from_values(
11350                LogicalType::Integer,
11351                &[
11352                    Value::Integer(4),
11353                    Value::Integer(4),
11354                    Value::Integer(4),
11355                    Value::Integer(9),
11356                    Value::Integer(9),
11357                    Value::Integer(1),
11358                ],
11359            )
11360            .expect("integers"),
11361        ])
11362        .expect("one column");
11363        writer.append(&rows).expect("the only part");
11364        writer.finish().expect("commit");
11365        let reader = Reader::open(&path).expect("reopen from disk");
11366        let common = Common::new(reader);
11367        assert_eq!(common.rows(), 6);
11368        let column = common.column("id").expect("the file has that column");
11369        assert_eq!(common.column("nothing"), None);
11370        assert_eq!(
11371            common.rows_with(column, &Bound::Int(4)),
11372            Stat::exact(3, Provenance::FrequencySynopsis)
11373        );
11374        // Not in the file, and a synopsis that accounts for all six rows proves it.
11375        assert_eq!(
11376            common.rows_with(column, &Bound::Int(7)),
11377            Stat::exact(0, Provenance::FrequencySynopsis)
11378        );
11379        // A constant of another domain against an integer column. Nothing in the list compares
11380        // with it, so the zero above would be an artefact of the mismatch rather than a fact.
11381        assert_eq!(common.rows_with(column, &Bound::Bytes(b"four".to_vec())), Stat::Unknown);
11382        // A complete list has no remainder. Answering one of no rows over no values would hand the
11383        // caller a division to special case, and the counts above already answer this column.
11384        assert_eq!(common.remainder(column), None);
11385        fs::remove_file(&path).expect("clean up");
11386    }
11387
11388    #[test]
11389    fn string_frequency_estimates_do_not_open_the_global_dictionary() {
11390        let path = path("string_frequencies_for_the_planner");
11391        let mut writer =
11392            Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
11393                .expect("new file");
11394        let rows = Chunk::new(vec![
11395            Vector::from_values(
11396                LogicalType::Varchar,
11397                &[
11398                    Value::Varchar(String::new()),
11399                    Value::Varchar("alpha".into()),
11400                    Value::Varchar(String::new()),
11401                    Value::Varchar("beta".into()),
11402                    Value::Varchar(String::new()),
11403                ],
11404            )
11405            .expect("strings"),
11406        ])
11407        .expect("one column");
11408        writer.append(&rows).expect("the only part");
11409        writer.finish().expect("commit");
11410
11411        let reader = Reader::open(&path).expect("reopen from disk");
11412        assert_eq!(reader.reads().dictionaries, 0, "open reads only the directory");
11413        let common = Common::new(reader.clone());
11414        let column = common.column("text").expect("the file has that column");
11415        assert_eq!(
11416            common.rows_with(column, &Bound::Bytes(Vec::new())),
11417            Stat::exact(3, Provenance::FrequencySynopsis)
11418        );
11419        assert_eq!(
11420            common.rows_with(column, &Bound::Bytes(b"missing".to_vec())),
11421            Stat::exact(0, Provenance::FrequencySynopsis)
11422        );
11423        assert_eq!(
11424            reader.reads().dictionaries,
11425            0,
11426            "the bounded spellings answer without opening the dictionary index"
11427        );
11428        fs::remove_file(&path).expect("clean up");
11429    }
11430
11431    #[test]
11432    fn host_groups_certify_omitted_hosts_and_keep_exact_aggregates() {
11433        let path = path("certified_host_groups");
11434        let mut writer =
11435            Writer::create(&path, "hits", vec![Field::required("Referer", LogicalType::Varchar)])
11436                .expect("new file");
11437        let mut values = vec![Value::Varchar("http://www.example.com/a".into()); 150];
11438        values.extend(vec![Value::Varchar("https://example.com/b".into()); 70]);
11439        values.extend((0..550).map(|at| Value::Varchar(format!("https://site{at}.test/x"))));
11440        values.push(Value::Varchar(String::new()));
11441        for part in values.chunks(512) {
11442            writer
11443                .append(
11444                    &Chunk::new(vec![
11445                        Vector::from_values(LogicalType::Varchar, part).expect("strings"),
11446                    ])
11447                    .expect("one column"),
11448                )
11449                .expect("part written");
11450        }
11451        writer.finish().expect("commit");
11452        let reader = Reader::open(&path).expect("reopen");
11453        let summary = reader.table.host_groups.as_ref().expect("bounded host metadata");
11454        assert!(summary.omitted_max < 220);
11455        assert!(reader.host_groups(0, summary.omitted_max).expect("valid column").is_none());
11456        let groups = reader.host_groups(0, 220).expect("valid column").expect("certified");
11457        let example = groups.iter().find(|entry| entry.host == "example.com").expect("leader");
11458        assert_eq!(example.count, 220);
11459        assert_eq!(example.bytes_sum, 150 * 24 + 70 * 21);
11460        assert_eq!(example.minimum, "http://www.example.com/a");
11461        assert_eq!(reader.reads().dictionaries, 0, "the directory settles the question");
11462        fs::remove_file(&path).expect("clean up");
11463    }
11464
11465    /// A table directory with nothing in it but a name and one column, for the section tests.
11466    ///
11467    /// The section table is orthogonal to everything else in a directory, so the tests that pin it
11468    /// say so by starting from the emptiest table that encodes.
11469    fn bare_table(sections: Vec<Section>) -> Table {
11470        Table {
11471            name: "linked".to_owned(),
11472            fields: vec![Field::required("id", LogicalType::Integer)],
11473            stripes: Vec::new(),
11474            rows: 0,
11475            dictionaries: vec![None],
11476            dictionary_payloads: Vec::new(),
11477            distincts: vec![None],
11478            frequencies: vec![None],
11479            pair_frequencies: Vec::new(),
11480            frequency_texts: Vec::new(),
11481            host_groups: None,
11482            clustering: None,
11483            generation: 1,
11484            sections,
11485        }
11486    }
11487
11488    fn a_key_map_section() -> Section {
11489        Section {
11490            kind: *section::KEY_MAP,
11491            id: 1,
11492            generation: 3,
11493            extents: 1,
11494            extent_page: HEADER,
11495            extent_bytes: section::EXTENT_BYTES as u32,
11496            hash: 0x1234_5678_9abc_def0,
11497            flags: 0,
11498            header_bytes: 24,
11499        }
11500    }
11501
11502    #[test]
11503    fn a_section_table_round_trips_through_a_directory() {
11504        let mut later = a_key_map_section();
11505        later.kind = *b"RUDBZZ9\0";
11506        later.id = 2;
11507        let table = bare_table(vec![a_key_map_section(), later]);
11508        let directory = encode_directory(&table).expect("directory");
11509        let decoded = decode_directory(&directory, 1 << 20).expect("reopen");
11510        assert_eq!(decoded.sections(), &[a_key_map_section(), later]);
11511        // The second is a kind this build has no name for, and it survived the round trip anyway.
11512        // That is what keeps an old build from silently discarding a newer build's work when it
11513        // rewrites a directory.
11514        assert!(decoded.sections()[0].known());
11515        assert!(!decoded.sections()[1].known());
11516    }
11517
11518    #[test]
11519    fn a_directory_written_before_the_section_table_reads_as_a_table_with_none() {
11520        // The G1 exit criterion, at the directory level. A format 22 directory is exactly this
11521        // build's directory with the trailing section block cut off, so cutting it off is the
11522        // honest way to make one: no fixture to go stale, and no separate encoder to drift.
11523        let directory = encode_directory(&bare_table(Vec::new())).expect("directory");
11524        let block = SECTIONS.len() + size_of::<u64>() + size_of::<u16>();
11525        let older = &directory[..directory.len() - block];
11526        let decoded = decode_directory(older, 1 << 20).expect("a directory from before sections");
11527        assert!(decoded.sections().is_empty());
11528        assert_eq!(decoded.generation(), 0, "a format 22 table recorded no generation");
11529        assert_eq!(decoded.name(), "linked");
11530        assert_eq!(decoded.fields().len(), 1, "everything before the block still decodes");
11531    }
11532
11533    #[test]
11534    fn a_file_stamped_with_the_previous_format_still_opens_and_reads() {
11535        // The same criterion end to end, which is the one the milestone actually asks for: a build
11536        // that knows about sections opens a file written by a build that did not, with no rewrite
11537        // and no repair, and answers from it. The version field is patched rather than a file
11538        // committed by an old binary because the bytes either side of it are identical: format 22
11539        // and format 23 differ only in a trailing directory block, and a reader that stops before
11540        // that block gets a table with no sections.
11541        let path = path("format_twenty_two");
11542        let mut writer =
11543            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
11544                .expect("new file");
11545        let rows = Chunk::new(vec![
11546            Vector::from_values(
11547                LogicalType::Integer,
11548                &[Value::Integer(1), Value::Integer(2), Value::Integer(3)],
11549            )
11550            .expect("integers"),
11551        ])
11552        .expect("one column");
11553        writer.append(&rows).expect("the only part");
11554        writer.finish().expect("commit");
11555
11556        let file = OpenOptions::new().write(true).open(&path).expect("reopen to patch");
11557        write_at(&file, 8, &22_u32.to_le_bytes()).expect("stamp the older format");
11558        drop(file);
11559
11560        let reader = Reader::open(&path).expect("a format 22 file opens unchanged");
11561        assert_eq!(reader.table().rows(), 3);
11562        // The rows and not the section table, because the section block is found by the magic at
11563        // the end of the directory rather than by the number in the header, so stamping the header
11564        // back does not take away the summaries this writer put there. What the test is about is
11565        // that the version check accepts 22, and the rows coming back is what says it did.
11566        assert_eq!(reader.read(0, &[0]).expect("the part still reads").len(), 3);
11567
11568        // And a format this build has never written is still refused, so the accept set is a list
11569        // and not an absence of a check.
11570        let file = OpenOptions::new().write(true).open(&path).expect("reopen to patch");
11571        write_at(&file, 8, &21_u32.to_le_bytes()).expect("stamp an unreadable format");
11572        drop(file);
11573        let error = Reader::open(&path).expect_err("format 21 is not readable");
11574        assert!(error.to_string().contains("format 21"), "{error}");
11575
11576        fs::remove_file(&path).expect("clean up");
11577    }
11578
11579    #[test]
11580    fn a_section_whose_extent_table_is_outside_the_file_is_refused() {
11581        // The bound the format has to check and `section` cannot, because only the reader knows how
11582        // big the file is. Reading the payload a section like this names would be reading whatever
11583        // else happens to be at that offset, which is the one way a graph section could turn into a
11584        // wrong answer rather than a slow one.
11585        let mut past = a_key_map_section();
11586        past.extent_page = 1 << 30;
11587        let directory = encode_directory(&bare_table(vec![past])).expect("directory");
11588        let error = decode_directory(&directory, 1 << 20).expect_err("refused");
11589        assert!(error.to_string().contains("outside the file"), "{error}");
11590
11591        let mut inside_the_header = a_key_map_section();
11592        inside_the_header.extent_page = 8;
11593        let directory = encode_directory(&bare_table(vec![inside_the_header])).expect("directory");
11594        assert!(
11595            decode_directory(&directory, 1 << 20).is_err(),
11596            "a section may not overlap a header"
11597        );
11598    }
11599
11600    #[test]
11601    fn a_section_recorded_as_not_built_is_legal_and_names_no_bytes() {
11602        // Section 3.7: a relationship that does not fit the budget is recorded with its size so
11603        // that `rudb_links()` can report what a larger budget would buy. That record is a section
11604        // entry with no extents, so it has to survive a round trip while naming nothing.
11605        let not_built = Section {
11606            kind: *section::FORWARD_LINK,
11607            id: 9,
11608            generation: 3,
11609            extents: 0,
11610            extent_page: 0,
11611            extent_bytes: 0,
11612            hash: 0,
11613            flags: 0,
11614            header_bytes: 0,
11615        };
11616        let directory = encode_directory(&bare_table(vec![not_built])).expect("directory");
11617        let decoded = decode_directory(&directory, 1 << 20).expect("reopen");
11618        assert_eq!(decoded.sections(), &[not_built]);
11619
11620        // But a section with no extents that still names an extent table is incoherent, and an
11621        // incoherent entry is a torn directory rather than a relationship that was skipped.
11622        let mut incoherent = not_built;
11623        incoherent.extent_bytes = 28;
11624        incoherent.extent_page = HEADER;
11625        let directory = encode_directory(&bare_table(vec![incoherent])).expect("directory");
11626        assert!(decode_directory(&directory, 1 << 20).is_err());
11627    }
11628
11629    #[test]
11630    fn a_directory_naming_more_sections_than_the_bound_is_refused() {
11631        let directory = encode_directory(&bare_table(Vec::new())).expect("directory");
11632        let mut torn = directory.clone();
11633        let count_at = torn.len() - size_of::<u16>();
11634        torn[count_at..].copy_from_slice(&u16::MAX.to_le_bytes());
11635        // Not an allocation of sixty five thousand entries off a torn count: either the bound
11636        // refuses it or the bytes run out, and both are errors rather than a read past the end.
11637        assert!(decode_directory(&torn, 1 << 20).is_err());
11638    }
11639
11640    /// A committed one column file of `rows` integers, for the attach tests.
11641    fn linked_file(label: &str, rows: i32) -> PathBuf {
11642        let path = path(label);
11643        let mut writer =
11644            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
11645                .expect("new file");
11646        let values = (0..rows).map(Value::Integer).collect::<Vec<_>>();
11647        let chunk =
11648            Chunk::new(vec![Vector::from_values(LogicalType::Integer, &values).expect("integers")])
11649                .expect("one column");
11650        writer.append(&chunk).expect("the only part");
11651        writer.finish().expect("commit");
11652        path
11653    }
11654
11655    fn a_key_map_payload() -> Vec<u8> {
11656        // Shaped like one without being one: this crate never reads a payload, so what matters here
11657        // is that every byte comes back and that the header the entry measures is at the front.
11658        (0..512_u32).flat_map(u32::to_le_bytes).collect()
11659    }
11660
11661    #[test]
11662    fn a_section_attached_to_a_committed_file_reads_back_byte_for_byte() {
11663        let path = linked_file("attach", 64);
11664        let payload = a_key_map_payload();
11665        let table = attach(
11666            &path,
11667            "items",
11668            &[section::Attachment {
11669                kind: *section::KEY_MAP,
11670                id: 0,
11671                flags: 2,
11672                header_bytes: 40,
11673                bytes: &payload,
11674            }],
11675        )
11676        .expect("attach a key map");
11677        assert_eq!(attached(&table).len(), 1);
11678
11679        let reader = Reader::open(&path).expect("reopen after the attach");
11680        let held = attached(reader.table());
11681        assert_eq!(held.len(), 1);
11682        assert_eq!(held[0].kind, *section::KEY_MAP);
11683        assert_eq!(held[0].flags, 2, "the form a reader must not have to guess");
11684        assert_eq!(held[0].header_bytes, 40);
11685        // The generation is the one the pages were written at, not the one the attach committed at.
11686        // Attaching a section moved no row, so a section written by it is current, and a second
11687        // table added to this file later would not make it stale.
11688        assert_eq!(held[0].generation, 1);
11689        assert!(held[0].usable(reader.table().generation()));
11690        assert_eq!(reader.payload(held[0]).expect("read the payload"), payload);
11691        assert_eq!(reader.extents(held[0]).expect("extent table").len(), 1);
11692
11693        fs::remove_file(&path).expect("clean up");
11694    }
11695
11696    #[test]
11697    fn attaching_a_section_answers_every_row_exactly_as_before() {
11698        // Section 3.1 end to end, and the reason the whole layer is safe to build incrementally. A
11699        // file with a section in it and the same file without one have to agree row for row, so the
11700        // comparison is made against the answers taken before the attach rather than against a
11701        // constant somebody typed.
11702        let path = linked_file("attach_changes_nothing", 300);
11703        let before = Reader::open(&path).expect("open before");
11704        let rows = before.table().rows();
11705        let first = before.read(0, &[0]).expect("read before");
11706        let values = (0..rows).map(|at| first.value_at(at, 0)).collect::<Vec<_>>();
11707        let layout = before.layout().columns_total();
11708        drop(before);
11709
11710        let payload = a_key_map_payload();
11711        attach(
11712            &path,
11713            "items",
11714            &[section::Attachment {
11715                kind: *section::KEY_MAP,
11716                id: 0,
11717                flags: 0,
11718                header_bytes: 0,
11719                bytes: &payload,
11720            }],
11721        )
11722        .expect("attach");
11723
11724        let after = Reader::open(&path).expect("open after");
11725        assert_eq!(after.table().rows(), rows);
11726        let read = after.read(0, &[0]).expect("read after");
11727        for (at, value) in values.iter().enumerate() {
11728            assert_eq!(&read.value_at(at, 0), value, "row {at} moved");
11729        }
11730        assert_eq!(
11731            after.layout().columns_total(),
11732            layout,
11733            "an attach appends and does not rewrite a column page"
11734        );
11735
11736        fs::remove_file(&path).expect("clean up");
11737    }
11738
11739    #[test]
11740    fn a_rebuilt_section_replaces_the_one_it_supersedes() {
11741        // Rebuilding a key map has to be a write and not a question. If an attach added rather than
11742        // replaced, a table rebuilt a few times would name several maps for one column and a reader
11743        // would have to pick, which is a decision with no right answer in it.
11744        let path = linked_file("attach_twice", 32);
11745        let one = a_key_map_payload();
11746        let two = vec![7_u8; 1024];
11747        let entry = |bytes| section::Attachment {
11748            kind: *section::KEY_MAP,
11749            id: 4,
11750            flags: 1,
11751            header_bytes: 0,
11752            bytes,
11753        };
11754        attach(&path, "items", &[entry(&one)]).expect("first build");
11755        attach(&path, "items", &[entry(&two)]).expect("rebuild");
11756
11757        let reader = Reader::open(&path).expect("reopen");
11758        let held = attached(reader.table());
11759        assert_eq!(held.len(), 1, "one map per column and not one per build");
11760        assert_eq!(reader.payload(held[0]).expect("payload"), two);
11761
11762        fs::remove_file(&path).expect("clean up");
11763    }
11764
11765    #[test]
11766    fn an_attach_carries_through_a_kind_it_does_not_know() {
11767        // The first of section 3.2's three rules, at the point where it is easiest to break: a build
11768        // that rewrites a directory has to carry an entry it has no name for, or opening a file with
11769        // an older binary and attaching one section quietly deletes the work of a newer one.
11770        let path = linked_file("attach_unknown", 16);
11771        let payload = vec![3_u8; 96];
11772        attach(
11773            &path,
11774            "items",
11775            &[section::Attachment {
11776                kind: *b"RUDBZZ9\0",
11777                id: 1,
11778                flags: 0,
11779                header_bytes: 0,
11780                bytes: &payload,
11781            }],
11782        )
11783        .expect("a kind this build does not know still writes");
11784        let key_map = a_key_map_payload();
11785        attach(
11786            &path,
11787            "items",
11788            &[section::Attachment {
11789                kind: *section::KEY_MAP,
11790                id: 0,
11791                flags: 0,
11792                header_bytes: 0,
11793                bytes: &key_map,
11794            }],
11795        )
11796        .expect("attach beside it");
11797
11798        let reader = Reader::open(&path).expect("reopen");
11799        let held = attached(reader.table());
11800        assert_eq!(held.len(), 2, "the unfamiliar entry survived a directory rewrite");
11801        let unknown = held.iter().find(|one| !one.known()).expect("the unfamiliar one");
11802        assert_eq!(reader.payload(unknown).expect("its bytes are still there"), payload);
11803
11804        fs::remove_file(&path).expect("clean up");
11805    }
11806
11807    #[test]
11808    fn a_payload_of_nothing_is_a_relationship_recorded_as_not_built() {
11809        let path = linked_file("attach_not_built", 8);
11810        attach(
11811            &path,
11812            "items",
11813            &[section::Attachment {
11814                kind: *section::FORWARD_LINK,
11815                id: 2,
11816                flags: 0,
11817                header_bytes: 0,
11818                bytes: &[],
11819            }],
11820        )
11821        .expect("record a link that did not fit the budget");
11822
11823        let reader = Reader::open(&path).expect("reopen");
11824        let held = attached(reader.table());
11825        assert_eq!(held.len(), 1);
11826        assert_eq!(held[0].extents, 0);
11827        assert_eq!(held[0].extent_page, 0, "an entry that names no bytes points at none");
11828        assert!(reader.extents(held[0]).expect("no extent table").is_empty());
11829        assert!(reader.payload(held[0]).expect("no payload").is_empty());
11830
11831        fs::remove_file(&path).expect("clean up");
11832    }
11833
11834    #[test]
11835    fn a_payload_past_one_extent_is_split_and_joined_back() {
11836        // Issue #745's rule, exercised rather than argued. One byte past the bound is the smallest
11837        // payload that has to be two extents, and it is the case a split written for the common
11838        // size gets wrong.
11839        let path = linked_file("attach_two_extents", 8);
11840        let payload = vec![0x5a_u8; section::MAX_EXTENT as usize + 1];
11841        attach(
11842            &path,
11843            "items",
11844            &[section::Attachment {
11845                kind: *section::KEY_MAP,
11846                id: 0,
11847                flags: 0,
11848                header_bytes: 0,
11849                bytes: &payload,
11850            }],
11851        )
11852        .expect("attach a payload past the bound");
11853
11854        let reader = Reader::open(&path).expect("reopen");
11855        let held = attached(reader.table());
11856        let extents = reader.extents(held[0]).expect("extent table");
11857        assert_eq!(extents.len(), 2, "one byte past the bound is two extents");
11858        assert_eq!(extents[0].length, section::MAX_EXTENT);
11859        assert_eq!(extents[1].length, 1);
11860        assert_eq!(extents[1].first, u64::from(section::MAX_EXTENT));
11861        // And the extent the caller wants is readable on its own, which is the point of the split.
11862        assert_eq!(reader.extent(&extents[1]).expect("the last extent"), vec![0x5a]);
11863        assert_eq!(reader.payload(held[0]).expect("the whole payload").len(), payload.len());
11864
11865        fs::remove_file(&path).expect("clean up");
11866    }
11867
11868    #[test]
11869    fn a_torn_extent_is_refused_rather_than_decoded() {
11870        let path = linked_file("attach_torn", 8);
11871        let payload = a_key_map_payload();
11872        attach(
11873            &path,
11874            "items",
11875            &[section::Attachment {
11876                kind: *section::KEY_MAP,
11877                id: 0,
11878                flags: 0,
11879                header_bytes: 0,
11880                bytes: &payload,
11881            }],
11882        )
11883        .expect("attach");
11884
11885        let reader = Reader::open(&path).expect("reopen");
11886        let extent = reader.extents(&reader.table().sections()[0]).expect("extent table")[0];
11887        let file = OpenOptions::new().write(true).open(&path).expect("reopen to corrupt");
11888        write_at(&file, extent.offset + 7, &[0xff]).expect("flip a byte of the payload");
11889        drop(file);
11890
11891        let reader = Reader::open(&path).expect("the table still opens");
11892        let error = reader
11893            .payload(&reader.table().sections()[0])
11894            .expect_err("a corrupt payload is not handed out");
11895        assert!(error.to_string().contains("checksum"), "{error}");
11896        // And the table is still readable, which is section 3.1: a section that cannot be trusted
11897        // costs the query its shortcut and nothing else.
11898        assert_eq!(reader.read(0, &[0]).expect("the column is untouched").width(), 1);
11899
11900        fs::remove_file(&path).expect("clean up");
11901    }
11902
11903    #[test]
11904    fn attaching_to_a_file_of_the_previous_format_is_refused_rather_than_done() {
11905        // Readable is not writable. A format 22 directory has no section block, and adding one
11906        // without moving the number in the header would leave a file claiming a format it is not.
11907        let path = linked_file("attach_old_format", 8);
11908        let file = OpenOptions::new().write(true).open(&path).expect("reopen to patch");
11909        write_at(&file, 8, &22_u32.to_le_bytes()).expect("stamp the older format");
11910        drop(file);
11911
11912        let payload = a_key_map_payload();
11913        let error = attach(
11914            &path,
11915            "items",
11916            &[section::Attachment {
11917                kind: *section::KEY_MAP,
11918                id: 0,
11919                flags: 0,
11920                header_bytes: 0,
11921                bytes: &payload,
11922            }],
11923        )
11924        .expect_err("format 22 cannot gain a section");
11925        assert!(error.to_string().contains("format 22"), "{error}");
11926        assert!(Reader::open(&path).expect("and the file is untouched").table().rows() == 8);
11927
11928        fs::remove_file(&path).expect("clean up");
11929    }
11930
11931    #[test]
11932    fn a_section_header_longer_than_its_payload_is_refused_at_the_write() {
11933        let path = linked_file("attach_bad_header", 8);
11934        let error = attach(
11935            &path,
11936            "items",
11937            &[section::Attachment {
11938                kind: *section::KEY_MAP,
11939                id: 0,
11940                flags: 0,
11941                header_bytes: 40,
11942                bytes: &[1, 2, 3],
11943            }],
11944        )
11945        .expect_err("a writer's bug stops at the write");
11946        assert!(error.to_string().contains("header is longer"), "{error}");
11947
11948        fs::remove_file(&path).expect("clean up");
11949    }
11950
11951    #[test]
11952    fn attaching_to_a_name_the_file_does_not_hold_says_so() {
11953        let path = linked_file("attach_wrong_name", 8);
11954        let error = attach(&path, "orders", &[]).expect_err("no such table");
11955        assert!(error.to_string().contains("orders"), "{error}");
11956        fs::remove_file(&path).expect("clean up");
11957    }
11958
11959    #[test]
11960    fn the_planner_gets_an_exact_count_for_a_leading_value_of_an_incomplete_synopsis() {
11961        // The case a complete synopsis does not cover, and the one worth the most. 16,000 rows over
11962        // 601 distinct values, 10,000 of them holding a single value and the rest spread ten apiece
11963        // over six hundred more. The writer holds 512 values, so the list is a prefix and most of
11964        // the tail is outside it. The counts inside it are still exact, because the pass recounts
11965        // the candidates that survived it, so `id = 1` is ten thousand rows rather than the
11966        // twenty six a distinct count of 601 would divide its way to.
11967        let path = path("frequency_prefix_for_the_planner");
11968        let mut writer =
11969            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
11970                .expect("new file");
11971        let mut values = vec![Value::Integer(1); 10_000];
11972        for _ in 0..10 {
11973            values.extend((0..600).map(|tail| Value::Integer(1_000 + tail)));
11974        }
11975        // A vector holds 8,192 rows, so this goes in as several parts. The pass that takes the
11976        // synopsis walks the whole column rather than a part, so the counts are the same either way.
11977        for part in values.chunks(8_000) {
11978            let rows = Chunk::new(vec![
11979                Vector::from_values(LogicalType::Integer, part).expect("integers"),
11980            ])
11981            .expect("one column");
11982            writer.append(&rows).expect("a part");
11983        }
11984        writer.finish().expect("commit");
11985        let reader = Reader::open(&path).expect("reopen from disk");
11986        let prefix =
11987            reader.frequency_prefix(0).expect("a readable synopsis").expect("the column has one");
11988        // A prefix and not the whole column, and the writer said how many rows anything left out of
11989        // it can hold.
11990        assert_eq!(prefix.entries.len(), 512);
11991        assert_eq!(prefix.omitted_max, 10);
11992        let common = Common::new(reader);
11993        assert_eq!(common.rows(), 16_000);
11994        let column = common.column("id").expect("the file has that column");
11995        assert_eq!(
11996            common.rows_with(column, &Bound::Int(1)),
11997            Stat::exact(10_000, Provenance::FrequencySynopsis)
11998        );
11999        // In the prefix, because ties go to the smaller value and the prefix reaches 1,510.
12000        assert_eq!(
12001            common.rows_with(column, &Bound::Int(1_100)),
12002            Stat::exact(10, Provenance::FrequencySynopsis)
12003        );
12004        // Outside it, and a prefix says nothing about a value it does not list. Not zero, which is
12005        // what a complete list would say, and the file holds ten rows of this one.
12006        assert_eq!(common.rows_with(column, &Bound::Int(1_550)), Stat::Unknown);
12007        // Not in the file at all, and still nothing rather than a zero. A prefix cannot tell the
12008        // two apart, which is the whole of what it gives up.
12009        assert_eq!(common.rows_with(column, &Bound::Int(9_999)), Stat::Unknown);
12010        // What the prefix left out, which is what turns the unknown above into a number. The 512
12011        // entries account for 15,110 rows, so 890 are left for the 89 values the writer dropped,
12012        // and 890 over 89 is the ten rows each of them really holds.
12013        let remainder = common.remainder(column).expect("the list is a prefix");
12014        assert_eq!(remainder, Remainder { rows: 890, listed: 512, most: 10 });
12015        assert_eq!(remainder.rows / (601 - remainder.listed), 10);
12016        fs::remove_file(&path).expect("clean up");
12017    }
12018
12019    /// A file with no table in it is a file, and opening it says so rather than failing.
12020    #[test]
12021    fn a_file_holding_no_table_commits_and_opens_and_a_table_can_be_added_to_it() {
12022        let path = path("empty");
12023        Writer::empty(&path, &[]).expect("a file with nothing in it");
12024        let catalog = Catalog::open(&path).expect("the empty file opens");
12025        assert_eq!(catalog.len(), 0);
12026        assert!(catalog.is_empty());
12027        assert_eq!(catalog.names().count(), 0);
12028        // The next generation goes over the top of it the way it goes over any other, which is what
12029        // says this is a committed file and not a special case somebody has to know about.
12030        let mut writer =
12031            Writer::open(&path, "items", vec![Field::required("id", LogicalType::Integer)])
12032                .expect("a table goes into the empty file");
12033        writer.append(&sample_ids()).expect("rows");
12034        writer.finish().expect("commit");
12035        let catalog = Catalog::open(&path).expect("the file opens again");
12036        assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
12037        fs::remove_file(&path).expect("clean up");
12038    }
12039
12040    /// A committed table with no rows is a name the next generation takes over, and one with rows
12041    /// is a name it refuses.
12042    ///
12043    /// The refusal is what it always was and it is load bearing: carrying a table that holds rows
12044    /// forward means reading and rewriting its pages, and a writer that quietly wrote a second
12045    /// entry under the same name would leave a file with two tables a reader cannot tell apart. An
12046    /// empty one has no pages and no reader, so there is nothing to carry and nothing to lose, and
12047    /// taking its place is what lets a schema committed by an earlier session be loaded by a stream
12048    /// instead of through memory.
12049    #[test]
12050    fn a_committed_empty_table_gives_up_its_name_and_one_with_rows_does_not() {
12051        let path = path("empty-name");
12052        let field = || vec![Field::required("id", LogicalType::Integer)];
12053        Writer::create(&path, "items", field()).expect("new file").finish().expect("commit");
12054        let catalog = Catalog::open(&path).expect("the file opens");
12055        assert_eq!(catalog.rows().collect::<Vec<_>>(), vec![("items", 0)]);
12056
12057        let mut writer = Writer::open(&path, "items", field()).expect("the empty name is free");
12058        writer.append(&sample_ids()).expect("rows");
12059        writer.finish().expect("commit");
12060        let catalog = Catalog::open(&path).expect("the file opens again");
12061        // One entry and not two. The generation replaced the empty table rather than joining it.
12062        assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
12063        let held = catalog.rows().collect::<Vec<_>>();
12064        assert_eq!(held.len(), 1);
12065        assert!(held[0].1 > 0, "the rows that were appended are the ones the catalog counts");
12066
12067        // The same call against the same name now that it holds rows, which is still refused.
12068        let error = Writer::open(&path, "items", field()).expect_err("a name with rows is taken");
12069        assert!(error.to_string().contains("same name"), "{error}");
12070        fs::remove_file(&path).expect("clean up");
12071    }
12072
12073    /// A view, with everything about it that a reopened catalog has to be able to answer from.
12074    fn sample_view(name: &str) -> ViewEntry {
12075        ViewEntry {
12076            name: name.to_string(),
12077            sql: "SELECT id FROM items WHERE id > 0".to_string(),
12078            statement: format!("CREATE VIEW {name} AS SELECT id FROM items WHERE (id > 0);"),
12079            aliases: vec!["n".to_string()],
12080            columns: vec![Field::new("n", LogicalType::Integer)],
12081        }
12082    }
12083
12084    #[test]
12085    fn a_view_written_into_the_catalog_comes_back_whole() {
12086        let path = path("views");
12087        let mut writer =
12088            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
12089                .expect("new file");
12090        writer.append(&sample_ids()).expect("rows");
12091        writer.with_views(vec![sample_view("v")]).finish().expect("commit");
12092        let catalog = Catalog::open(&path).expect("reopen");
12093        assert_eq!(catalog.views().cloned().collect::<Vec<_>>(), vec![sample_view("v")]);
12094        // The tables are still there and are still read the same way, so the section on the end did
12095        // not move anything in front of it.
12096        assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
12097        fs::remove_file(&path).expect("clean up");
12098    }
12099
12100    /// A writer opened to append a table says nothing about views and must not lose them.
12101    #[test]
12102    fn appending_a_table_carries_the_views_forward() {
12103        let path = path("viewscarry");
12104        let mut writer =
12105            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
12106                .expect("new file");
12107        writer.append(&sample_ids()).expect("rows");
12108        writer.with_views(vec![sample_view("v")]).finish().expect("commit");
12109        let mut writer =
12110            Writer::open(&path, "other", vec![Field::required("id", LogicalType::Integer)])
12111                .expect("a second table");
12112        writer.append(&sample_ids()).expect("rows");
12113        writer.finish().expect("commit");
12114        let catalog = Catalog::open(&path).expect("reopen");
12115        assert_eq!(catalog.views().count(), 1);
12116        assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items", "other"]);
12117        fs::remove_file(&path).expect("clean up");
12118    }
12119
12120    /// The whole point of [`Writer::restate`]: the views change and the pages do not move.
12121    #[test]
12122    fn restating_the_views_leaves_every_table_where_it_was() {
12123        let path = path("restate");
12124        let mut writer =
12125            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
12126                .expect("new file");
12127        writer.append(&sample_ids()).expect("rows");
12128        writer.finish().expect("commit");
12129        let before = fs::metadata(&path).expect("the file is there").len();
12130        Writer::restate(&path, &[sample_view("v"), sample_view("w")]).expect("two views");
12131        let catalog = Catalog::open(&path).expect("reopen");
12132        assert_eq!(catalog.views().count(), 2);
12133        assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
12134        // A catalog on the end and nothing else, so what it grew by is the size of a catalog rather
12135        // than the size of the table.
12136        let after = fs::metadata(&path).expect("the file is there").len();
12137        assert!(after > before, "a generation was written");
12138        assert!(after - before < before, "the table was not written again");
12139        // The rows are still readable through the new generation, which is the part that would go
12140        // wrong if the catalog carried the wrong directory pointers forward.
12141        let reader = Catalog::open(&path).expect("reopen").table("items").expect("the table");
12142        assert_eq!(reader.table().rows, 3);
12143        // And a restate over a restate keeps working, because each one reads the slot that
12144        // checksummed rather than the highest number in the header.
12145        Writer::restate(&path, &[]).expect("no views at all");
12146        assert_eq!(Catalog::open(&path).expect("reopen").views().count(), 0);
12147        fs::remove_file(&path).expect("clean up");
12148    }
12149
12150    /// Two entries under one name is a catalog no lookup can answer, whichever two they are.
12151    #[test]
12152    fn a_view_named_after_a_table_is_refused_when_the_catalog_is_read() {
12153        let bytes = encode_catalog(
12154            &[Entry {
12155                name: "items".to_string(),
12156                fields: vec![Field::required("id", LogicalType::Integer)],
12157                rows: 1,
12158                directory: Page { offset: HEADER, length: 8, hash: 0 },
12159                nonzero: vec![None],
12160                aggregates: vec![None],
12161                distincts: vec![None],
12162                extremes: vec![None],
12163                frequencies: vec![None],
12164            }],
12165            &[sample_view("items")],
12166        )
12167        .expect("it encodes, because encoding does not look");
12168        let error = decode_catalog(&bytes, HEADER + 8).expect_err("and decoding does");
12169        assert!(error.to_string().contains("same name"), "{error}");
12170    }
12171
12172    #[test]
12173    fn committed_file_reopens_and_reads_only_requested_columns() {
12174        let path = path("reopen");
12175        let mut writer = Writer::create(
12176            &path,
12177            "items",
12178            vec![
12179                Field::required("id", LogicalType::Integer),
12180                Field::new("text", LogicalType::Varchar),
12181            ],
12182        )
12183        .expect("new file");
12184        writer.append(&sample()).expect("first part");
12185        writer.append(&sample()).expect("second part");
12186        writer.finish().expect("commit");
12187        let reader = Reader::open(&path).expect("reopen from disk");
12188        assert_eq!(reader.table().rows(), 6);
12189        // Two appends below the stripe bound are two parts of one stripe, which is the whole point
12190        // of the split: the directory describes the stripe and the scan still reads a part.
12191        assert_eq!(reader.table().stripes().len(), 1);
12192        assert_eq!(reader.parts(), 2);
12193        assert_eq!(reader.part_rows(0), 3);
12194        assert_eq!(reader.part_rows(1), 3);
12195        let text = reader.read(1, &[1]).expect("only text page");
12196        assert_eq!(text.width(), 1);
12197        assert_eq!(text.value_at(1, 0), Value::Null);
12198        assert_eq!(text.value_at(2, 0), Value::Varchar("long text after a slash".into()));
12199        let sparse = reader.read_sparse(1, &[1]).expect("one part without its whole page");
12200        assert_eq!(sparse.width(), 1);
12201        assert_eq!(sparse.value_at(1, 0), Value::Null);
12202        assert_eq!(sparse.value_at(2, 0), Value::Varchar("long text after a slash".into()));
12203        assert!(!reader.skips_codes(0, 1, &[0]).expect("alpha is in the stripe"));
12204        assert!(!reader.skips_codes(0, 1, &[2]).expect("long text is in the stripe"));
12205        assert!(reader.skips_codes(0, 1, &[3]).expect("unknown code is absent"));
12206        let count = reader.read(0, &[]).expect("no page is needed for count");
12207        assert_eq!(count.len(), 3);
12208        assert!(reader.skips(0, &[Probe { column: 0, op: Op::Greater, value: Bound::Int(100) }]));
12209        assert!(!reader.skips(0, &[Probe { column: 0, op: Op::Greater, value: Bound::Int(0) }]));
12210        let integers = reader.top_frequencies(0, 1).expect("valid integer synopsis").expect("kept");
12211        assert_eq!(
12212            integers,
12213            vec![(Value::Integer(-2), 2), (Value::Integer(4), 2), (Value::Integer(9), 2),]
12214        );
12215        let strings = reader.top_frequencies(1, 1).expect("valid string synopsis").expect("kept");
12216        assert_eq!(strings.len(), 3);
12217        assert!(strings.contains(&(Value::Null, 2)));
12218        assert!(strings.contains(&(Value::Varchar("alpha".into()), 2)));
12219        assert!(strings.contains(&(Value::Varchar("long text after a slash".into()), 2)));
12220        fs::remove_file(path).expect("remove scratch file");
12221    }
12222
12223    /// Two pipeline instances handing over whole runs, which is what makes the native sink safe to
12224    /// instance.
12225    ///
12226    /// The runs arrive in the order the instances finished reading them rather than in source
12227    /// order, and the second one to finish is the one that read the earlier rows. Each run is still
12228    /// a stripe of its own and the table still reads back in source order, which is the whole of
12229    /// what the writer promises about ordering.
12230    #[test]
12231    fn runs_handed_over_out_of_order_still_read_back_in_source_order() {
12232        let path = path("interleaved-runs");
12233        let mut writer =
12234            Writer::create(&path, "interleaved", vec![Field::new("v", LogicalType::BigInt)])
12235                .expect("new file");
12236        for morsel in [2_u64, 0, 3, 1] {
12237            let parts = (0..4_u64)
12238                .map(|chunk| {
12239                    let first = i64::try_from(morsel * 32 + chunk * 8).expect("small");
12240                    let values =
12241                        (0..8_i64).map(|row| Value::BigInt(first + row)).collect::<Vec<_>>();
12242                    let column =
12243                        Vector::from_values(LogicalType::BigInt, &values).expect("a column");
12244                    ((morsel, chunk), Chunk::new(vec![column]).expect("one column"))
12245                })
12246                .collect::<Vec<_>>();
12247            writer.append_stripe(parts).expect("a stripe");
12248        }
12249        writer.finish().expect("commit");
12250
12251        let reader = Reader::open(&path).expect("valid directory");
12252        assert_eq!(reader.table().stripes().len(), 4, "a run is a stripe of its own");
12253        assert_eq!(reader.table().rows(), 128);
12254        for part in 0..16_usize {
12255            let read = reader.read(part, &[0]).expect("a part back");
12256            for row in 0..8_usize {
12257                let want = i64::try_from(part * 8 + row).expect("small");
12258                assert_eq!(read.value_at(row, 0), Value::BigInt(want), "part {part} row {row}");
12259            }
12260        }
12261        fs::remove_file(path).expect("remove scratch file");
12262    }
12263
12264    /// Runs from different callers may interleave and may not overlap, and the commit is what
12265    /// catches an overlap.
12266    #[test]
12267    fn runs_that_overlap_each_other_are_refused_at_commit() {
12268        let path = path("overlapping-runs");
12269        let mut writer =
12270            Writer::create(&path, "overlapping", vec![Field::new("v", LogicalType::BigInt)])
12271                .expect("new file");
12272        let one = |order: (u64, u64)| {
12273            let column =
12274                Vector::from_values(LogicalType::BigInt, &[Value::BigInt(1)]).expect("a column");
12275            (order, Chunk::new(vec![column]).expect("one column"))
12276        };
12277        // The second run sits inside the first rather than after it, which is a thing no instance
12278        // holding its own contiguous run can produce and a thing the file cannot represent.
12279        writer.append_stripe(vec![one((0, 0)), one((0, 2))]).expect("a stripe");
12280        writer.append_stripe(vec![one((0, 1))]).expect("a stripe");
12281        let error = writer.finish().expect_err("the runs overlap");
12282        assert!(error.message().contains("source order"), "{error}");
12283        fs::remove_file(path).expect("remove scratch file");
12284    }
12285
12286    /// A stripe holds [`STRIPE_PARTS`] parts, so a run longer than that is a caller bug rather than
12287    /// something to split, and the writer says so at the door instead of quietly cutting it in two.
12288    #[test]
12289    fn a_run_longer_than_a_stripe_is_refused() {
12290        let path = path("overlong-run");
12291        let mut writer =
12292            Writer::create(&path, "overlong", vec![Field::new("v", LogicalType::BigInt)])
12293                .expect("new file");
12294        let parts = (0..=STRIPE_PARTS)
12295            .map(|at| {
12296                let column = Vector::from_values(LogicalType::BigInt, &[Value::BigInt(1)])
12297                    .expect("a column");
12298                let chunk = Chunk::new(vec![column]).expect("one column");
12299                ((0, u64::try_from(at).expect("small")), chunk)
12300            })
12301            .collect::<Vec<_>>();
12302        let error = writer.append_stripe(parts).expect_err("one part too many");
12303        assert!(error.message().contains("more parts than it holds"), "{error}");
12304        fs::remove_file(path).expect("remove scratch file");
12305    }
12306
12307    /// Parts past the stripe bound start a new stripe, and every part stays addressable on its own.
12308    ///
12309    /// This is the shape the format exists for, so both ends of the split are checked here. The
12310    /// directory holds three stripes rather than a hundred and thirty one, and a read of any one
12311    /// part still answers with that part's rows rather than with its whole stripe's.
12312    #[test]
12313    fn parts_past_the_stripe_bound_start_a_new_stripe() {
12314        let path = path("stripe-bound");
12315        let mut writer = Writer::create(
12316            &path,
12317            "items",
12318            vec![
12319                Field::required("id", LogicalType::Integer),
12320                Field::new("text", LogicalType::Varchar),
12321            ],
12322        )
12323        .expect("new file");
12324        let parts = STRIPE_PARTS * 2 + 3;
12325        for part in 0..parts {
12326            let id = part as i32;
12327            let chunk = Chunk::new(vec![
12328                Vector::from_values(
12329                    LogicalType::Integer,
12330                    &[Value::Integer(id), Value::Integer(-id)],
12331                )
12332                .expect("integers"),
12333                Vector::from_values(
12334                    LogicalType::Varchar,
12335                    &[Value::Varchar(format!("value {part}")), Value::Null],
12336                )
12337                .expect("strings"),
12338            ])
12339            .expect("matching rows");
12340            writer.append(&chunk).expect("one part");
12341        }
12342        writer.finish().expect("commit");
12343
12344        let reader = Reader::open(&path).expect("reopen from disk");
12345        assert_eq!(reader.parts(), parts);
12346        assert_eq!(reader.table().rows(), parts * 2);
12347        assert_eq!(reader.table().stripes().len(), parts.div_ceil(STRIPE_PARTS));
12348        assert_eq!(reader.table().stripes()[0].parts(), STRIPE_PARTS);
12349        assert_eq!(reader.table().stripes()[0].rows(), STRIPE_PARTS * 2);
12350        assert_eq!(reader.table().stripes()[2].parts(), 3);
12351        // Backwards on purpose. The reader keeps four stripes a column, so a scan that walks the
12352        // table the other way is what catches a cache that only ever holds what it just read.
12353        for part in (0..parts).rev() {
12354            let dense = reader.read(part, &[0, 1]).expect("a whole page read");
12355            let sparse = reader.read_sparse(part, &[0, 1]).expect("one part read");
12356            for chunk in [&dense, &sparse] {
12357                assert_eq!(chunk.len(), 2, "part {part} has its own row count");
12358                assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
12359                assert_eq!(chunk.value_at(1, 0), Value::Integer(-(part as i32)));
12360                assert_eq!(chunk.value_at(0, 1), Value::Varchar(format!("value {part}")));
12361                assert_eq!(chunk.value_at(1, 1), Value::Null);
12362            }
12363        }
12364        // The bounds are merged over the stripe, so they answer for the range the whole stripe
12365        // covers and not for the part that was asked about.
12366        let above = [Probe { column: 0, op: Op::Greater, value: Bound::Int(100) }];
12367        assert!(reader.skips(0, &above), "the first stripe stops at 63");
12368        assert!(!reader.skips(STRIPE_PARTS * 2, &above), "the third stripe reaches 130");
12369        fs::remove_file(path).expect("remove scratch file");
12370    }
12371
12372    /// A scattered value in the column that decides `WHERE UserID = ?`.
12373    fn scattered(n: i64) -> i64 {
12374        n.wrapping_mul(-7_046_029_254_386_353_131)
12375    }
12376
12377    /// A part whose sieve does not hold the constant is skipped, and a range would skip none of them.
12378    ///
12379    /// This is ClickBench query 19 in miniature. The values are spread over the whole of `BIGINT`, so
12380    /// every stripe's bounds cover nearly all of it and rule out nothing, and the part that really
12381    /// holds the value is the only one a scan has to read.
12382    #[test]
12383    fn a_part_is_skipped_when_its_sieve_does_not_hold_the_constant() {
12384        let path = path("sieve-skip");
12385        let mut writer =
12386            Writer::create(&path, "hits", vec![Field::required("id", LogicalType::BigInt)])
12387                .expect("new file");
12388        let parts = STRIPE_PARTS + 3;
12389        // Big enough that the filter is worth its bytes. A part of eight numbers packs to under a
12390        // hundred bytes and the smallest filter there is is sixty nine, so a filter over a part
12391        // that small costs about as much to read as the rows do and is no longer written.
12392        let per_part = 128;
12393        for part in 0..parts {
12394            let held: Vec<Value> = (0..per_part)
12395                .map(|row| Value::BigInt(scattered((part * per_part + row) as i64)))
12396                .collect();
12397            let chunk =
12398                Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
12399                    .expect("one column");
12400            writer.append(&chunk).expect("one part");
12401        }
12402        writer.finish().expect("commit");
12403
12404        let reader = Reader::open(&path).expect("reopen from disk");
12405        let probe = |value: i64| Probe {
12406            column: 0,
12407            op: Op::Equal,
12408            value: Bound::Int(i128::from(scattered(value))),
12409        };
12410        for wanted in [0_i64, (per_part + 1) as i64, (parts * per_part - 1) as i64] {
12411            let tests = [probe(wanted)];
12412            let kept: Vec<usize> = (0..parts).filter(|&part| !reader.skips(part, &tests)).collect();
12413            let home = wanted as usize / per_part;
12414            assert!(kept.contains(&home), "the part holding {wanted} is read");
12415            // A filter answers maybe, so a part it keeps need not hold the value. Sixty seven parts
12416            // of a hundred and twenty eight numbers each, at a dozen bits a value, is about one
12417            // stray part across the whole file and that is what this leaves room for.
12418            assert!(kept.len() <= 2, "{wanted} keeps {kept:?}, which is more than one stray part");
12419        }
12420        let absent = [probe((parts * per_part) as i64 + 1)];
12421        let kept = (0..parts).filter(|&part| !reader.skips(part, &absent)).count();
12422        assert!(kept <= 1, "{kept} parts of {parts} kept a value no part holds");
12423        // The same probes against the bounds alone, which is what this replaces. A column of
12424        // scattered numbers has a range per stripe that covers nearly the whole type.
12425        let tests = [probe(0)];
12426        assert!(
12427            reader.table().stripes().iter().all(|stripe| !stripe.zone.skips(&tests)),
12428            "the bounds rule out no stripe at all"
12429        );
12430        fs::remove_file(path).expect("remove scratch file");
12431    }
12432
12433    /// A part whose own bounds rule out an ordered comparison is skipped where the stripe's keep it.
12434    ///
12435    /// This is the shape of ClickBench 24. Each part covers a narrow stretch of the column and the
12436    /// stripe covers all sixty four of them at once, so a comparison that lands inside the stripe
12437    /// rules out none of it and rules out all but a few parts.
12438    #[test]
12439    fn a_part_is_skipped_when_its_own_bounds_rule_out_a_comparison_the_stripe_keeps() {
12440        let path = path("part-range-skip");
12441        let mut writer =
12442            Writer::create(&path, "hits", vec![Field::required("at", LogicalType::BigInt)])
12443                .expect("new file");
12444        let parts = STRIPE_PARTS + 3;
12445        let per_part = 128;
12446        for part in 0..parts {
12447            // Scattered inside the part's own band rather than a run, because a run of
12448            // consecutive numbers encodes to a stride of a few bytes and then the page of ranges
12449            // costs more than reading the column it indexes, which is the case the writer declines.
12450            let held: Vec<Value> = (0..per_part)
12451                .map(|row| {
12452                    Value::BigInt((part * 1_000) as i64 + (scattered(row as i64).rem_euclid(900)))
12453                })
12454                .collect();
12455            let chunk =
12456                Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
12457                    .expect("one column");
12458            writer.append(&chunk).expect("one part");
12459        }
12460        writer.finish().expect("commit");
12461
12462        let reader = Reader::open(&path).expect("reopen from disk");
12463        let under = [Probe { column: 0, op: Op::Less, value: Bound::Int(3_000) }];
12464        let kept: Vec<usize> = (0..parts).filter(|&part| !reader.skips(part, &under)).collect();
12465        assert_eq!(kept, vec![0, 1, 2], "only the three parts that start under three thousand");
12466        // The same question asked of the stripe alone, which is what this replaces.
12467        assert!(!reader.stripe_skips(0, &under), "the stripe reaches from zero and keeps itself");
12468        fs::remove_file(path).expect("remove scratch file");
12469    }
12470
12471    /// The other half of the same page. A part whose own bounds put every row of it inside the
12472    /// filter is waved through, so the comparison never runs on it, where the stripe's bounds reach
12473    /// across every part and can prove nothing.
12474    #[test]
12475    fn a_part_is_waved_through_when_its_own_bounds_pass_a_comparison_the_stripe_cannot() {
12476        let path = path("part-range-certain");
12477        let mut writer =
12478            Writer::create(&path, "hits", vec![Field::required("at", LogicalType::BigInt)])
12479                .expect("new file");
12480        let parts = STRIPE_PARTS + 3;
12481        let per_part = 128;
12482        for part in 0..parts {
12483            let held: Vec<Value> = (0..per_part)
12484                .map(|row| {
12485                    Value::BigInt((part * 1_000) as i64 + (scattered(row as i64).rem_euclid(900)))
12486                })
12487                .collect();
12488            let chunk =
12489                Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
12490                    .expect("one column");
12491            writer.append(&chunk).expect("one part");
12492        }
12493        writer.finish().expect("commit");
12494
12495        let reader = Reader::open(&path).expect("reopen from disk");
12496        let under = [Probe { column: 0, op: Op::Less, value: Bound::Int(3_000) }];
12497        let waved: Vec<usize> = (0..parts).filter(|&part| reader.certain(part, &under)).collect();
12498        assert_eq!(waved, vec![0, 1, 2], "the three parts that end under three thousand");
12499        // The first stripe reaches from zero to past sixty thousand, so it straddles three thousand
12500        // and settles nothing either way. The three yeses above are the parts' own ends talking.
12501        assert!(!reader.stripe_skips(0, &under), "the stripe straddles the comparison");
12502        fs::remove_file(path).expect("remove scratch file");
12503    }
12504
12505    /// The page is worth its bytes on a column with parts to tell apart and is not written on one
12506    /// that has a single part, where the stripe bounds already are the part's.
12507    #[test]
12508    fn a_stripe_of_one_part_writes_no_range_page_and_a_stripe_of_many_does() {
12509        for (parts, wanted) in [(1_usize, false), (STRIPE_PARTS, true)] {
12510            let path = path("part-range-page");
12511            let mut writer =
12512                Writer::create(&path, "hits", vec![Field::required("at", LogicalType::BigInt)])
12513                    .expect("new file");
12514            for part in 0..parts {
12515                let held: Vec<Value> = (0..128)
12516                    .map(|row| {
12517                        Value::BigInt((part * 1_000) as i64 + scattered(row as i64).rem_euclid(900))
12518                    })
12519                    .collect();
12520                let chunk = Chunk::new(vec![
12521                    Vector::from_values(LogicalType::BigInt, &held).expect("numbers"),
12522                ])
12523                .expect("one column");
12524                writer.append(&chunk).expect("one part");
12525            }
12526            writer.finish().expect("commit");
12527            let reader = Reader::open(&path).expect("reopen from disk");
12528            let bytes = reader.layout().columns[0].part_ranges;
12529            assert_eq!(bytes > 0, wanted, "{parts} parts wrote {bytes} bytes of ranges");
12530            fs::remove_file(path).expect("remove scratch file");
12531        }
12532    }
12533
12534    /// A cut down string end is still an end on the side it was, which is the only thing that keeps
12535    /// a shortened bound from turning a skip into a wrong answer.
12536    #[test]
12537    fn a_string_end_that_is_cut_down_still_covers_the_value_it_came_from() {
12538        let long = vec![b'a'; PART_BOUND_BYTES * 2];
12539        let low = shortened(Some(Bound::Bytes(long.clone())), false).expect("a low end");
12540        let high = shortened(Some(Bound::Bytes(long.clone())), true).expect("a high end");
12541        let Bound::Bytes(low) = low else { panic!("a string stays a string") };
12542        let Bound::Bytes(high) = high else { panic!("a string stays a string") };
12543        assert!(low.len() <= PART_BOUND_BYTES && high.len() <= PART_BOUND_BYTES);
12544        assert!(low.as_slice() <= long.as_slice(), "the low end is at or under the value");
12545        assert!(high.as_slice() >= long.as_slice(), "the high end is at or over the value");
12546    }
12547
12548    /// A string of nothing but the largest byte has no prefix that can be stepped up, so the high
12549    /// end is given up rather than claimed too small. No end keeps the part, which is always safe.
12550    #[test]
12551    fn a_string_end_with_no_room_to_step_up_gives_up_the_bound() {
12552        let long = vec![u8::MAX; PART_BOUND_BYTES * 2];
12553        assert_eq!(shortened(Some(Bound::Bytes(long.clone())), true), None);
12554        let low = shortened(Some(Bound::Bytes(long)), false).expect("a low end is still a prefix");
12555        assert_eq!(low, Bound::Bytes(vec![u8::MAX; PART_BOUND_BYTES]));
12556    }
12557
12558    /// What a column is stored as, asked of two files holding the same rows in a different order.
12559    ///
12560    /// This is the question the report exists to answer and it is the one the directory cannot. The
12561    /// two files have the same rows, the same schema and the same number of parts, and the column
12562    /// comes out four times smaller in one of them, because ascending keys delta encode to a few
12563    /// bits a row and shuffled ones do not. Nothing about the file's shape says so. The page header
12564    /// says so, and reading it is what this does.
12565    ///
12566    /// It is q18 on TPC-H in miniature: clustering lineitem by ship date leaves `l_orderkey`
12567    /// ascending inside a partition but sparse, its deltas go from six bits to twelve, and the scan
12568    /// pays for the wider ones.
12569    #[test]
12570    fn what_a_column_is_stored_as_follows_the_order_the_rows_were_written_in() {
12571        let parts = 4;
12572        let per_part = 1024;
12573        let rows = parts * per_part;
12574        let written = |name: &str, keys: &[i64]| {
12575            let path = path(name);
12576            let fields = vec![Field::required("key", LogicalType::BigInt)];
12577            let mut writer = Writer::create(&path, "keys", fields).expect("new file");
12578            for part in 0..parts {
12579                let values: Vec<Value> = keys[part * per_part..(part + 1) * per_part]
12580                    .iter()
12581                    .map(|key| Value::BigInt(*key))
12582                    .collect();
12583                let chunk = Chunk::new(vec![
12584                    Vector::from_values(LogicalType::BigInt, &values).expect("numbers"),
12585                ])
12586                .expect("one column");
12587                writer.append(&chunk).expect("one part");
12588            }
12589            writer.finish().expect("commit");
12590            path
12591        };
12592        // Ascending with a small irregular step, which is what a key column in arrival order looks
12593        // like: an order has one to seven line items, so the key repeats and then moves on by one.
12594        let climbing = |step: &dyn Fn(usize) -> i64| {
12595            let mut key = 0;
12596            (0..rows)
12597                .map(|row| {
12598                    key += step(row);
12599                    key
12600                })
12601                .collect::<Vec<i64>>()
12602        };
12603        let ascending = climbing(&|row| (row % 3) as i64);
12604        // The same rows in the same direction over a range a thousand times wider, which is what a
12605        // partition of a clustered table holds: still ascending, and far enough apart that the
12606        // deltas no longer fit in a handful of bits.
12607        let sparse = climbing(&|row| ((row * 2_654_435_761) % 4096) as i64);
12608        let near_path = written("stored-near", &ascending);
12609        let far_path = written("stored-far", &sparse);
12610
12611        let one = Reader::open(&near_path).expect("reopen from disk");
12612        let other = Reader::open(&far_path).expect("reopen from disk");
12613        let near = one.stored(0).expect("the column is stored");
12614        let far = other.stored(0).expect("the column is stored");
12615        assert_eq!(near.len(), parts, "one row per part");
12616        assert_eq!(far.len(), parts);
12617        // The bytes are the same bytes the directory totals, which is the check that this is
12618        // reading the pages the file really holds rather than some other pages.
12619        let total = |stored: &[StoredPart]| stored.iter().map(|part| part.bytes).sum::<u64>();
12620        assert_eq!(total(&near), one.layout().columns[0].pages);
12621        assert_eq!(total(&far), other.layout().columns[0].pages);
12622        assert!(
12623            total(&near) * 2 < total(&far),
12624            "the sparse keys cost more, {} against {}",
12625            total(&far),
12626            total(&near)
12627        );
12628        // Every part accounted for, in order, with the row it starts at following the one before.
12629        for (at, part) in near.iter().enumerate() {
12630            assert_eq!(part.part, at);
12631            assert_eq!(part.row, at * per_part);
12632            assert_eq!(part.rows, per_part);
12633            let held = &ascending[at * per_part..(at + 1) * per_part];
12634            assert_eq!(part.low, Some(Value::BigInt(held[0])));
12635            assert_eq!(part.high, Some(Value::BigInt(held[per_part - 1])));
12636            assert_eq!(part.nulls, Some(0));
12637        }
12638        // And the encoding is a line of text that names what the encoder chose, which is the whole
12639        // point. Both are a cascade over deltas and the widths inside them are what differ.
12640        assert!(near[0].encoding.contains("DELTA"), "{}", near[0].encoding);
12641        assert!(far[0].encoding.contains("DELTA"), "{}", far[0].encoding);
12642        assert_ne!(near[0].encoding, far[0].encoding);
12643        fs::remove_file(near_path).expect("remove scratch file");
12644        fs::remove_file(far_path).expect("remove scratch file");
12645    }
12646
12647    /// A sieve bigger than the part it indexes is not written, and one smaller than it still is.
12648    ///
12649    /// Both columns hold values spread over the whole of `BIGINT`, so neither gets a bitmap and both
12650    /// reach the filter. They differ in what the part costs to read. `spread` is a thousand distinct
12651    /// numbers and packs to eight kilobytes, so a filter of about thirteen hundred bytes is a good
12652    /// trade. `repeated` is the same thousand rows over four numbers in runs and encodes to
12653    /// almost nothing, but the filter is sized for the rows rather than the values it turns out to
12654    /// hold, so it comes out larger than the data. Reading it to decide whether to read the part spends more than
12655    /// the part, every time, and that is the case this drops.
12656    #[test]
12657    fn a_sieve_larger_than_the_part_it_indexes_is_not_written() {
12658        let path = path("sieve-pays");
12659        let fields = vec![
12660            Field::required("spread", LogicalType::BigInt),
12661            Field::required("repeated", LogicalType::BigInt),
12662        ];
12663        let mut writer = Writer::create(&path, "hits", fields).expect("new file");
12664        let parts = 3;
12665        let per_part = 1024;
12666        for part in 0..parts {
12667            let base = (part * per_part) as i64;
12668            let spread: Vec<Value> =
12669                (0..per_part).map(|row| Value::BigInt(scattered(base + row as i64))).collect();
12670            let repeated: Vec<Value> =
12671                (0..per_part).map(|row| Value::BigInt(scattered((row / 256) as i64))).collect();
12672            let chunk = Chunk::new(vec![
12673                Vector::from_values(LogicalType::BigInt, &spread).expect("numbers"),
12674                Vector::from_values(LogicalType::BigInt, &repeated).expect("numbers"),
12675            ])
12676            .expect("two columns");
12677            writer.append(&chunk).expect("one part");
12678        }
12679        writer.finish().expect("commit");
12680
12681        let reader = Reader::open(&path).expect("reopen from disk");
12682        let layout = reader.layout();
12683        let spread = &layout.columns[0];
12684        let repeated = &layout.columns[1];
12685        assert!(spread.sieves > 0, "a column whose parts are worth a filter keeps one");
12686        assert_eq!(
12687            repeated.sieves, 0,
12688            "a column whose filter costs more than its parts keeps none"
12689        );
12690        // Per part this is the rule itself, so it holds over the column as well: a part without a
12691        // sieve adds to one side of this and to nothing on the other.
12692        for column in &layout.columns {
12693            assert!(
12694                column.sieves < column.pages,
12695                "{} spends {} on sieves over {} of data",
12696                column.name,
12697                column.sieves,
12698                column.pages
12699            );
12700        }
12701        // The filter that was kept still does what it is for.
12702        let absent = [Probe {
12703            column: 0,
12704            op: Op::Equal,
12705            value: Bound::Int(i128::from(scattered((parts * per_part) as i64 + 1))),
12706        }];
12707        assert!((0..parts).all(|part| reader.skips(part, &absent)), "no part holds it");
12708        fs::remove_file(path).expect("remove scratch file");
12709    }
12710
12711    /// A damaged sieve page is a part that gets read, not a query that fails.
12712    ///
12713    /// A sieve is an index over rows that are still there and still correct, so losing one costs
12714    /// time and costs no answers. That is the opposite of the membership index beside it, which is
12715    /// the only thing standing between a string page and a wrong answer.
12716    #[test]
12717    fn a_damaged_sieve_page_is_read_through_rather_than_refused() {
12718        let path = path("sieve-damaged");
12719        let mut writer =
12720            Writer::create(&path, "hits", vec![Field::required("id", LogicalType::BigInt)])
12721                .expect("new file");
12722        let rows = 128;
12723        let held: Vec<Value> = (0..rows).map(|row| Value::BigInt(scattered(row))).collect();
12724        let chunk =
12725            Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
12726                .expect("one column");
12727        writer.append(&chunk).expect("one part");
12728        writer.finish().expect("commit");
12729
12730        let page = Reader::open(&path).expect("reopen").table.stripes[0]
12731            .sieves
12732            .get(0)
12733            .expect("a sieve page");
12734        let mut file = OpenOptions::new().write(true).open(&path).expect("open the sieve page");
12735        file.seek(SeekFrom::Start(page.offset + u64::from(page.length) - 1)).expect("seek");
12736        file.write_all(&[0xff]).expect("damage one byte");
12737        drop(file);
12738
12739        let reader = Reader::open(&path).expect("reopen the damaged file");
12740        let absent =
12741            [Probe { column: 0, op: Op::Equal, value: Bound::Int(i128::from(scattered(99))) }];
12742        assert!(!reader.skips(0, &absent), "a sieve that cannot be read skips nothing");
12743        assert_eq!(
12744            reader.read(0, &[0]).expect("the rows are untouched").len(),
12745            usize::try_from(rows).expect("a small count")
12746        );
12747        fs::remove_file(path).expect("remove scratch file");
12748    }
12749
12750    /// Eight workers over one stripe read it once between them.
12751    ///
12752    /// This is the shape a scan actually has. Parts are handed out in order, so every worker on a
12753    /// column crosses into a stripe within a few parts of the others, and before [`Reader::held`]
12754    /// started sharing the read every one of them read the whole page. On the full ClickBench file
12755    /// that was a `MIN(EventDate), MAX(EventDate)` moving 3.2 GB off the disk to look at 400 MB of
12756    /// column, which is most of what a first touch costs.
12757    ///
12758    /// The workers that lose the race still answer, out of the part reads they do instead, which is
12759    /// what the values below are checking.
12760    #[test]
12761    fn workers_that_want_the_same_stripe_read_it_once() {
12762        let path = path("single-flight");
12763        let mut writer =
12764            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
12765                .expect("new file");
12766        for part in 0..STRIPE_PARTS {
12767            let id = part as i32;
12768            let chunk = Chunk::new(vec![
12769                Vector::from_values(
12770                    LogicalType::Integer,
12771                    &[Value::Integer(id), Value::Integer(-id)],
12772                )
12773                .expect("integers"),
12774            ])
12775            .expect("matching rows");
12776            writer.append(&chunk).expect("one part");
12777        }
12778        writer.finish().expect("commit");
12779
12780        let reader = Reader::open(&path).expect("reopen from disk");
12781        assert_eq!(reader.table().stripes().len(), 1, "one stripe is the point of the test");
12782        let barrier = std::sync::Barrier::new(8);
12783        std::thread::scope(|scope| {
12784            for worker in 0..8 {
12785                let reader = &reader;
12786                let barrier = &barrier;
12787                scope.spawn(move || {
12788                    barrier.wait();
12789                    for part in (worker..STRIPE_PARTS).step_by(8) {
12790                        let chunk = reader.read(part, &[0]).expect("a whole page read");
12791                        assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
12792                        assert_eq!(chunk.value_at(1, 0), Value::Integer(-(part as i32)));
12793                    }
12794                });
12795            }
12796        });
12797        assert_eq!(reader.pages.load(Atomic::Relaxed), 1, "one stripe, one page read, whoever won");
12798        fs::remove_file(path).expect("remove scratch file");
12799    }
12800
12801    /// Opening a file reads the header and the directory, and nothing that depends on the rows.
12802    ///
12803    /// `spec/stats/04-in-memory.md` section 4.2. There are no statistics in the file yet, so this
12804    /// holds today by not having anything to load, and that is exactly why it is worth pinning now.
12805    /// The change that breaks it is the reasonable looking one: summaries are a few hundred bytes,
12806    /// the next query will want them, so read them on the way past. A process that opened the
12807    /// database to run one trivial query pays for all of it and gets nothing.
12808    ///
12809    /// Two files of the same shape and a thousand times the rows in one of them, opened, and the
12810    /// two openings cost the same. The stripe count is held equal so that the directory is the same
12811    /// size in both, which leaves the rows as the only thing that changed. Anything read out of the
12812    /// data would show up here.
12813    #[test]
12814    fn opening_costs_the_same_over_a_thousand_times_the_rows() {
12815        let opened = |label: &str, rows_per_part: i32| {
12816            let path = path(label);
12817            let mut writer =
12818                Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
12819                    .expect("new file");
12820            for part in 0..STRIPE_PARTS * 3 {
12821                // Scrambled rather than sequential, so that the fat file is actually fatter. A run
12822                // of consecutive integers encodes to almost nothing and would leave the two files
12823                // the same size, which would make this test pass for the wrong reason.
12824                let values = (0..rows_per_part)
12825                    .map(|row| {
12826                        Value::Integer((part as i32 * rows_per_part + row).wrapping_mul(2_654_435))
12827                    })
12828                    .collect::<Vec<_>>();
12829                let chunk = Chunk::new(vec![
12830                    Vector::from_values(LogicalType::Integer, &values).expect("integers"),
12831                ])
12832                .expect("matching rows");
12833                writer.append(&chunk).expect("one part");
12834            }
12835            writer.finish().expect("commit");
12836            let reader = Reader::open(&path).expect("reopen from disk");
12837            let size = fs::metadata(&path).expect("the file is there").len();
12838            let out = (reader.reads(), reader.table().stripes().len(), size);
12839            fs::remove_file(path).expect("remove scratch file");
12840            out
12841        };
12842
12843        let (thin, thin_stripes, thin_size) = opened("open-thin", 1);
12844        let (fat, fat_stripes, fat_size) = opened("open-fat", 1000);
12845        assert_eq!(
12846            thin_stripes, fat_stripes,
12847            "the same stripe count is what makes this a fair ask"
12848        );
12849        assert!(
12850            fat_size > thin_size * 50,
12851            "the fat file has to actually be larger, and it is {fat_size} against {thin_size}"
12852        );
12853
12854        assert_eq!(thin.opening.reads, fat.opening.reads, "the same reads either way");
12855        assert_eq!(thin.pages, 0, "opening read a page");
12856        assert_eq!(fat.pages, 0, "opening read a page");
12857        assert_eq!(thin.indexes, 0, "opening read an index");
12858        assert_eq!(fat.indexes, 0, "opening read an index");
12859        // Not exactly equal, because a directory holds offsets and a larger file has larger ones,
12860        // and a handful of bytes of varint is not somebody loading statistics. A factor is.
12861        assert!(
12862            fat.opening.bytes < thin.opening.bytes * 2,
12863            "opening the thin file read {} bytes and the fat one read {}",
12864            thin.opening.bytes,
12865            fat.opening.bytes
12866        );
12867    }
12868
12869    /// The reads a file costs to open are fixed by its shape and not by what ran before.
12870    ///
12871    /// `spec/stats/04-in-memory.md` section 4.3, which is the rule that keeps a plan reproducible:
12872    /// the plan is a function of the data, the generation and the settings, and never of what
12873    /// happened to be in cache. Opening the same file twice in the same process has to cost the
12874    /// same, because a second open that read less would be an open that was about to plan
12875    /// differently.
12876    #[test]
12877    fn two_opens_of_one_file_cost_the_same_and_the_second_is_not_cheaper() {
12878        let path = path("open-twice");
12879        let mut writer =
12880            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
12881                .expect("new file");
12882        for part in 0..STRIPE_PARTS * 3 {
12883            let chunk = Chunk::new(vec![
12884                Vector::from_values(LogicalType::Integer, &[Value::Integer(part as i32)])
12885                    .expect("integers"),
12886            ])
12887            .expect("matching rows");
12888            writer.append(&chunk).expect("one part");
12889        }
12890        writer.finish().expect("commit");
12891
12892        let first = Reader::open(&path).expect("open");
12893        // A whole scan in between, so the operating system's page cache is as warm as it gets and
12894        // anything that consulted it would show up in the second open.
12895        for part in 0..first.parts() {
12896            first.read(part, &[0]).expect("a part");
12897        }
12898        assert!(first.reads().pages > 0, "the scan has to have read something");
12899        let second = Reader::open(&path).expect("open again");
12900
12901        assert_eq!(first.reads().opening, second.reads().opening);
12902        assert_eq!(
12903            second.reads().pages,
12904            0,
12905            "the second open read a page off the back of the first"
12906        );
12907        assert_eq!(second.reads().indexes, 0, "the second open read an index it inherited");
12908        fs::remove_file(path).expect("remove scratch file");
12909    }
12910
12911    /// A scan reads a stripe's index once for the whole scan, not once per part that misses.
12912    ///
12913    /// The page cache holds four stripes and an index used to ride inside it, so a table with more
12914    /// stripes than that read the index again every time a stripe came back around. The index is a
12915    /// few hundred bytes and the page is a quarter of a megabyte, which is why they are now under
12916    /// different budgets. This is the test that keeps them there, since the saving is small enough
12917    /// that nothing in a benchmark would notice it going away again.
12918    #[test]
12919    fn an_index_is_read_once_per_stripe_however_often_the_page_is_evicted() {
12920        let path = path("index-cache");
12921        let mut writer =
12922            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
12923                .expect("new file");
12924        let parts = STRIPE_PARTS * (CACHED_STRIPES_PER_COLUMN + 2);
12925        for part in 0..parts {
12926            let id = part as i32;
12927            let chunk = Chunk::new(vec![
12928                Vector::from_values(LogicalType::Integer, &[Value::Integer(id)]).expect("integers"),
12929            ])
12930            .expect("matching rows");
12931            writer.append(&chunk).expect("one part");
12932        }
12933        writer.finish().expect("commit");
12934
12935        let reader = Reader::open(&path).expect("reopen from disk");
12936        let stripes = reader.table().stripes().len();
12937        assert!(stripes > CACHED_STRIPES_PER_COLUMN, "the page cache has to be too small for this");
12938        // Twice over, so that the second pass finds every page evicted and every index kept.
12939        for _ in 0..2 {
12940            for part in 0..parts {
12941                let chunk = reader.read(part, &[0]).expect("a part");
12942                assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
12943            }
12944        }
12945        assert_eq!(reader.indexes.load(Atomic::Relaxed), stripes, "one index read per stripe");
12946        assert!(
12947            reader.pages.load(Atomic::Relaxed) > stripes,
12948            "the pages are the ones that get read again, which is what makes the index count mean \
12949             something"
12950        );
12951        fs::remove_file(path).expect("remove scratch file");
12952    }
12953
12954    /// A page stays in memory from one scan to the next while the pool has room for it, and a
12955    /// table that is being read takes room from one that is not, down to the floor and no further.
12956    ///
12957    /// This is what the pool is for. Each reader lives as long as its database, so a second query
12958    /// over the same table should find every page it read the first time, and before the pool it
12959    /// found four stripes a column and read the rest off the file again.
12960    #[test]
12961    fn a_pool_keeps_pages_between_scans_and_gives_them_to_the_table_being_read() {
12962        let path = path("page-pool");
12963        let parts = STRIPE_PARTS * (CACHED_STRIPES_PER_COLUMN + 2);
12964        let fields = || vec![Field::required("id", LogicalType::Integer)];
12965        let mut writer = Writer::create(&path, "a", fields()).expect("new file");
12966        for table in ["a", "b"] {
12967            if table == "b" {
12968                writer = writer.next("b".to_string(), fields()).expect("a second table");
12969            }
12970            for part in 0..parts {
12971                let chunk = Chunk::new(vec![
12972                    Vector::from_values(LogicalType::Integer, &[Value::Integer(part as i32)])
12973                        .expect("integers"),
12974                ])
12975                .expect("matching rows");
12976                writer.append(&chunk).expect("one part");
12977            }
12978        }
12979        writer.finish().expect("commit");
12980
12981        let pool = PagePool::new(usize::MAX);
12982        let catalog = Catalog::open_in(&path, &pool).expect("the file opens");
12983        let (a, b) = (catalog.table("a").expect("a"), catalog.table("b").expect("b"));
12984        let stripes = a.table().stripes().len();
12985        assert!(stripes > CACHED_STRIPES_PER_COLUMN, "the floor has to be smaller than a table");
12986        let scan = |reader: &Reader| {
12987            for part in 0..parts {
12988                let chunk = reader.read(part, &[0]).expect("a part");
12989                assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
12990            }
12991        };
12992        scan(&a);
12993        scan(&a);
12994        assert_eq!(a.pages.load(Atomic::Relaxed), stripes, "the second scan reads nothing");
12995        let one = pool.bytes();
12996        assert!(one > 0, "the pool counts what the reader holds");
12997
12998        // Room for one table. Reading the other takes the first one's pages down to its floor.
12999        pool.budget.store(one, Atomic::Relaxed);
13000        scan(&b);
13001        assert_eq!(b.pages.load(Atomic::Relaxed), stripes, "a page is never let go while in use");
13002        assert_eq!(a.cache.held[0].load(Atomic::Relaxed), CACHED_STRIPES_PER_COLUMN);
13003        let held = a.cache.columns[0].lock().expect("the column").pages.iter().flatten().count();
13004        assert_eq!(held, CACHED_STRIPES_PER_COLUMN, "the count and the slots agree");
13005
13006        // A reader that goes takes its pages out of the count with it.
13007        drop((a, b, catalog));
13008        let c = Catalog::open_in(&path, &pool).expect("again").table("a").expect("a");
13009        scan(&c);
13010        assert!(pool.bytes() <= one, "only what the live reader holds is counted");
13011        fs::remove_file(path).expect("remove scratch file");
13012    }
13013
13014    /// A worker per stripe reads its stripe once, once the cache has been told how many there are.
13015    ///
13016    /// This is the shape a scan has when it hands out a whole stripe per morsel rather than a part.
13017    /// Nobody races for a page any more, but every worker holds a different one for the length of a
13018    /// stripe, so a cache that keeps four pages while eight workers are in eight stripes evicts
13019    /// every one of them before its owner has finished with it, and the owner reads a quarter of a
13020    /// megabyte again for the next part. The barrier is what makes that certain rather than likely:
13021    /// without it a worker can run a whole stripe before the next one starts and never collide.
13022    #[test]
13023    fn a_worker_per_stripe_reads_its_page_once_when_the_cache_was_told_to_expect_it() {
13024        let workers = CACHED_STRIPES_PER_COLUMN + 4;
13025        let path = path("stripe-per-worker");
13026        let mut writer =
13027            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
13028                .expect("new file");
13029        for part in 0..STRIPE_PARTS * workers {
13030            let chunk = Chunk::new(vec![
13031                Vector::from_values(LogicalType::Integer, &[Value::Integer(part as i32)])
13032                    .expect("integers"),
13033            ])
13034            .expect("matching rows");
13035            writer.append(&chunk).expect("one part");
13036        }
13037        writer.finish().expect("commit");
13038
13039        let read = |told: bool| {
13040            let reader = Reader::open(&path).expect("reopen from disk");
13041            assert_eq!(reader.table().stripes().len(), workers, "a stripe per worker");
13042            if told {
13043                reader.keep_stripes(workers);
13044            }
13045            let barrier = std::sync::Barrier::new(workers);
13046            std::thread::scope(|scope| {
13047                for (worker, run) in reader.stripe_parts().into_iter().enumerate() {
13048                    let reader = &reader;
13049                    let barrier = &barrier;
13050                    scope.spawn(move || {
13051                        for part in run {
13052                            barrier.wait();
13053                            let chunk = reader.read(part, &[0]).expect("a part of my own stripe");
13054                            assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
13055                        }
13056                        assert!(worker < workers);
13057                    });
13058                }
13059            });
13060            reader.pages.load(Atomic::Relaxed)
13061        };
13062
13063        assert_eq!(read(true), workers, "one page read per stripe and no more");
13064        assert!(read(false) > workers, "a cache that small is read again on every part");
13065        fs::remove_file(path).expect("remove scratch file");
13066    }
13067
13068    /// A damaged index page is caught before anything decodes a part out of it.
13069    ///
13070    /// The index is the one structure a reader trusts to find bytes with, so it carries a checksum
13071    /// per column section rather than one for the page, and this is what says that check runs.
13072    #[test]
13073    fn a_damaged_index_page_is_an_error() {
13074        let path = path("damaged-index");
13075        let mut writer =
13076            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
13077                .expect("new file");
13078        writer.append(&sample_ids()).expect("first part");
13079        writer.append(&sample_ids()).expect("second part");
13080        writer.finish().expect("commit");
13081
13082        let reader = Reader::open(&path).expect("valid directory");
13083        let index = reader.table.stripes[0].index;
13084        let mut byte = [0; 1];
13085        read_at(&reader.file, index.offset, &mut byte).expect("the first part length");
13086        let mut file = OpenOptions::new().write(true).open(&path).expect("open index page");
13087        file.seek(SeekFrom::Start(index.offset)).expect("index start");
13088        file.write_all(&[!byte[0]]).expect("damage the first part length");
13089        let error = reader.read(1, &[0]).expect_err("a damaged index must not be used");
13090        assert!(error.message().contains("index page section checksum differs"), "{error}");
13091        fs::remove_file(path).expect("remove scratch file");
13092    }
13093
13094    /// Every integer width the format knows about, written and read back.
13095    ///
13096    /// The unsigned ones are the reason ClickBench can be stored at all: `hits` types `EventDate`
13097    /// as `USMALLINT`, and one unsupported column meant the whole table was refused. The extremes
13098    /// are in here on purpose, because a width that round trips through the wrong signedness only
13099    /// goes wrong at the end of its range.
13100    #[test]
13101    fn every_integer_width_round_trips_through_a_page() {
13102        let path = path("integer-widths");
13103        let columns = [
13104            (LogicalType::TinyInt, vec![Value::TinyInt(i8::MIN), Value::TinyInt(i8::MAX)]),
13105            (LogicalType::UTinyInt, vec![Value::UTinyInt(0), Value::UTinyInt(u8::MAX)]),
13106            (LogicalType::SmallInt, vec![Value::SmallInt(i16::MIN), Value::SmallInt(i16::MAX)]),
13107            (LogicalType::USmallInt, vec![Value::USmallInt(0), Value::USmallInt(u16::MAX)]),
13108            (LogicalType::Integer, vec![Value::Integer(i32::MIN), Value::Integer(i32::MAX)]),
13109            (LogicalType::UInteger, vec![Value::UInteger(0), Value::UInteger(u32::MAX)]),
13110            (LogicalType::BigInt, vec![Value::BigInt(i64::MIN), Value::BigInt(i64::MAX)]),
13111            (LogicalType::UBigInt, vec![Value::UBigInt(0), Value::UBigInt(u64::MAX)]),
13112        ];
13113        let fields = columns
13114            .iter()
13115            .enumerate()
13116            .map(|(at, (ty, _))| Field::required(format!("c{at}"), ty.clone()))
13117            .collect::<Vec<_>>();
13118        let vectors = columns
13119            .iter()
13120            .map(|(ty, values)| Vector::from_values(ty.clone(), values).expect("a vector"))
13121            .collect::<Vec<_>>();
13122        let mut writer = Writer::create(&path, "widths", fields).expect("new file");
13123        writer.append(&Chunk::new(vectors).expect("matching rows")).expect("one stripe");
13124        writer.finish().expect("commit");
13125
13126        let reader = Reader::open(&path).expect("reopen from disk");
13127        let wanted = (0..columns.len()).collect::<Vec<_>>();
13128        let read = reader.read(0, &wanted).expect("every column");
13129        assert_eq!(read.len(), 2);
13130        // row at a time: each column has its own type and its own pair of extremes.
13131        for (at, (ty, values)) in columns.iter().enumerate() {
13132            assert_eq!(read.value_at(0, at), values[0], "the low end of {ty}");
13133            assert_eq!(read.value_at(1, at), values[1], "the high end of {ty}");
13134        }
13135        fs::remove_file(path).expect("remove scratch file");
13136    }
13137
13138    /// The rest of the fixed width types, and the byte strings, written and read back.
13139    ///
13140    /// The extremes again, and for a float that means more than the ends of the range. Negative
13141    /// zero and a NaN are the two values that go through an encoder unnoticed and come back
13142    /// different, so they are here on purpose, and the NaN is compared by its bits rather than by
13143    /// `==`, which a NaN fails against itself.
13144    ///
13145    /// A blob is here beside them because it is the same round trip asked of bytes that are not
13146    /// text. The value in it is not UTF-8, so a path that reads a payload as a string on the way
13147    /// past turns this test red rather than turning a user's column into nulls.
13148    #[test]
13149    fn every_other_type_the_format_knows_round_trips_through_a_page() {
13150        let path = path("other-types");
13151        let columns = [
13152            (LogicalType::Float, vec![Value::Float(f32::MIN), Value::Float(-0.0)]),
13153            (LogicalType::Double, vec![Value::Double(f64::MIN), Value::Double(f64::MAX)]),
13154            (LogicalType::HugeInt, vec![Value::HugeInt(i128::MIN), Value::HugeInt(i128::MAX)]),
13155            (LogicalType::UHugeInt, vec![Value::UHugeInt(0), Value::UHugeInt(u128::MAX)]),
13156            (LogicalType::Time, vec![Value::Time(0), Value::Time(86_399_999_999)]),
13157            (LogicalType::TimeTz, vec![Value::TimeTz(-50_400_000_000), Value::TimeTz(0)]),
13158            (
13159                LogicalType::TimestampTz,
13160                vec![Value::TimestampTz(i64::MIN + 1), Value::TimestampTz(i64::MAX)],
13161            ),
13162            (
13163                LogicalType::Interval,
13164                vec![
13165                    Value::Interval { months: i32::MIN, days: i32::MAX, micros: i64::MIN },
13166                    Value::Interval { months: 13, days: -1, micros: 1 },
13167                ],
13168            ),
13169            (
13170                LogicalType::Blob,
13171                vec![Value::Blob(vec![0, 0xff, 0x80, 0xfe]), Value::Blob(Vec::new())],
13172            ),
13173        ];
13174        let fields = columns
13175            .iter()
13176            .enumerate()
13177            .map(|(at, (ty, _))| Field::required(format!("c{at}"), ty.clone()))
13178            .collect::<Vec<_>>();
13179        let vectors = columns
13180            .iter()
13181            .map(|(ty, values)| Vector::from_values(ty.clone(), values).expect("a vector"))
13182            .collect::<Vec<_>>();
13183        let mut writer = Writer::create(&path, "others", fields).expect("new file");
13184        writer.append(&Chunk::new(vectors).expect("matching rows")).expect("one stripe");
13185        writer.finish().expect("commit");
13186
13187        let reader = Reader::open(&path).expect("reopen from disk");
13188        let wanted = (0..columns.len()).collect::<Vec<_>>();
13189        let read = reader.read(0, &wanted).expect("every column");
13190        assert_eq!(read.len(), 2);
13191        for (at, (ty, values)) in columns.iter().enumerate() {
13192            assert_eq!(read.value_at(0, at), values[0], "the low end of {ty}");
13193            assert_eq!(read.value_at(1, at), values[1], "the high end of {ty}");
13194        }
13195        // A float keeps its sign through a zero, which `==` says nothing about because negative
13196        // zero and zero compare equal.
13197        let Value::Float(zero) = read.value_at(1, 0) else { panic!("a float stays a float") };
13198        assert!(zero.is_sign_negative(), "a negative zero came back as {zero}");
13199
13200        fs::remove_file(path).expect("remove scratch file");
13201    }
13202
13203    /// A NaN is still a NaN after a trip through a page.
13204    ///
13205    /// Apart from the other floats because it cannot be asserted the same way. A NaN is not equal
13206    /// to itself, so a comparison against the value that was written passes for every NaN and for
13207    /// nothing else, which is the one assertion that would not catch a page that lost it.
13208    #[test]
13209    fn a_nan_survives_being_written_down() {
13210        let path = path("nan");
13211        let nan = Vector::from_values(LogicalType::Double, &[Value::Double(f64::NAN)])
13212            .expect("a NaN vector");
13213        let mut writer =
13214            Writer::create(&path, "nan", vec![Field::required("d", LogicalType::Double)])
13215                .expect("new file");
13216        writer.append(&Chunk::new(vec![nan]).expect("one column")).expect("one stripe");
13217        writer.finish().expect("commit");
13218        let read = Reader::open(&path).expect("reopen").read(0, &[0]).expect("the column");
13219        let Value::Double(back) = read.value_at(0, 0) else { panic!("a double stays a double") };
13220        assert!(back.is_nan(), "a NaN came back as {back}");
13221        fs::remove_file(path).expect("remove scratch file");
13222    }
13223
13224    /// A uuid and a bit string, which have no `Value` arm of their own and are checked as bits.
13225    ///
13226    /// A uuid is the 128 bit lane and a bit string is bytes, and neither of them reads back as
13227    /// anything in `Value` today, so asking for a value here would compare two nulls and pass
13228    /// whatever the file held. The data underneath is what the storage promise is about, so that is
13229    /// what this reads.
13230    #[test]
13231    fn a_uuid_and_a_bit_string_come_back_as_the_bits_that_went_in() {
13232        let path = path("uuid-and-bit");
13233        let uuids = vec![0_i128, i128::MIN, -1];
13234        let mut bits = StringColumn::new();
13235        for value in [&b"\x02\xff"[..], &b""[..], &b"\x00\x01\x02\x03\x04\x05"[..]] {
13236            bits.push_bytes(value);
13237        }
13238        let expected = bits.clone();
13239        let fields =
13240            vec![Field::required("u", LogicalType::Uuid), Field::required("b", LogicalType::Bit)];
13241        let vectors = vec![
13242            Vector::flat(LogicalType::Uuid, Data::Int128(uuids.clone().into())).expect("uuids"),
13243            Vector::flat(LogicalType::Bit, Data::Varlen(bits)).expect("bit strings"),
13244        ];
13245        let mut writer = Writer::create(&path, "ids", fields).expect("new file");
13246        writer.append(&Chunk::new(vectors).expect("matching rows")).expect("one stripe");
13247        writer.finish().expect("commit");
13248
13249        let reader = Reader::open(&path).expect("reopen from disk");
13250        let read = reader.read(0, &[0, 1]).expect("both columns").flatten().expect("flat");
13251        let Some(Data::Int128(back)) = read.column(0).expect("the uuids").data() else {
13252            panic!("a uuid column is the 128 bit lane")
13253        };
13254        assert_eq!(back.as_slice(), uuids.as_slice());
13255        let Some(Data::Varlen(back)) = read.column(1).expect("the bits").data() else {
13256            panic!("a bit column is bytes")
13257        };
13258        for row in 0..expected.len() {
13259            assert_eq!(back.bytes(row), expected.bytes(row), "row {row} of the bit column");
13260        }
13261        fs::remove_file(path).expect("remove scratch file");
13262    }
13263
13264    /// Counting a run at once has to leave the candidate table exactly where counting its rows one
13265    /// at a time would, including once the table is full and a run is turned away row by row.
13266    #[test]
13267    fn a_run_counted_at_once_leaves_the_candidates_a_row_at_a_time_would() {
13268        let mut rows: Vec<Option<u64>> = Vec::new();
13269        let mut state = 0x2545_f491_4f6c_dd1d_u64;
13270        for index in 0..400_000_u64 {
13271            state ^= state << 13;
13272            state ^= state >> 7;
13273            state ^= state << 17;
13274            let times = 1 + (state % 7) as usize;
13275            let bits = match state % 11 {
13276                0 => None,
13277                1..=3 => Some(state % 16),
13278                _ => Some(index.wrapping_mul(0x9e37_79b9_7f4a_7c15)),
13279            };
13280            rows.extend(std::iter::repeat_n(bits, times));
13281        }
13282        let mut by_row = Candidates::default();
13283        for &bits in &rows {
13284            by_row.add(bits, 1);
13285        }
13286        let mut by_run = Candidates::default();
13287        let mut run = Run::default();
13288        let mut runs = 0_usize;
13289        for &bits in &rows {
13290            if let Some((bits, times)) = run.push(bits) {
13291                by_run.add(bits, times);
13292                runs += 1;
13293            }
13294        }
13295        if let Some((bits, times)) = run.take() {
13296            by_run.add(bits, times);
13297        }
13298        assert!(runs < rows.len() / 2, "the rows came in runs");
13299        assert!(by_row.decrements > 0, "the table filled and turned values away");
13300        assert_eq!(by_run.counts, by_row.counts);
13301        assert_eq!(by_run.nulls, by_row.nulls);
13302        assert_eq!(by_run.decrements, by_row.decrements);
13303    }
13304
13305    #[test]
13306    fn numeric_frequency_candidates_keep_bounded_row_ordinals() {
13307        let path = path("frequency-ordinals");
13308        let mut writer =
13309            Writer::create(&path, "items", vec![Field::required("id", LogicalType::BigInt)])
13310                .expect("new file");
13311        let mut values = Vec::new();
13312        for leader in 0..10_i64 {
13313            values.extend(std::iter::repeat_n(leader, 100));
13314        }
13315        values.extend(1_000_i64..41_000);
13316        for part in values.chunks(1_024) {
13317            let vector = Vector::flat(LogicalType::BigInt, Data::Int64(part.to_vec().into()))
13318                .expect("big integers");
13319            writer.append(&Chunk::new(vec![vector]).expect("one column")).expect("one stripe");
13320        }
13321        writer.finish().expect("commit");
13322
13323        let reader = Reader::open(&path).expect("reopen from disk");
13324        let occurrences =
13325            reader.frequency_occurrences(0).expect("valid metadata").expect("bounded ordinals");
13326        assert!(occurrences.omitted_max < 100);
13327        assert!(occurrences.ordinals.len() <= FREQUENCY_ORDINALS);
13328        assert_eq!(occurrences.anchor_indices.len(), occurrences.ordinals.len());
13329        assert!(occurrences.ordinals.windows(2).all(|pair| pair[0] < pair[1]));
13330        assert_eq!(&occurrences.ordinals[..1_000], &(0_u64..1_000).collect::<Vec<_>>());
13331        assert_eq!(
13332            &occurrences.anchor_indices[..1_000]
13333                .iter()
13334                .map(|&entry| occurrences.anchors[entry as usize].clone())
13335                .collect::<Vec<_>>(),
13336            &(0_i64..10)
13337                .flat_map(|leader| std::iter::repeat_n(Value::BigInt(leader), 100))
13338                .collect::<Vec<_>>()
13339        );
13340        fs::remove_file(path).expect("remove scratch file");
13341    }
13342
13343    #[test]
13344    fn numeric_frequencies_count_nulls_and_values_past_the_top_of_bigint() {
13345        // Ten leaders, then more unique values than the candidate table holds, so the first pass
13346        // has to decrement and the counts come from the recount. The unsigned leaders sit above
13347        // `i64::MAX`, where reading the bits as signed would give a different value, and the signed
13348        // ones are negative, where reading them as unsigned would.
13349        let path = path("frequency-bits");
13350        let mut writer = Writer::create(
13351            &path,
13352            "items",
13353            vec![Field::new("u", LogicalType::UBigInt), Field::new("s", LogicalType::BigInt)],
13354        )
13355        .expect("new file");
13356        let mut rows = Vec::new();
13357        let mut leaders = Vec::new();
13358        for leader in 0..10_u64 {
13359            let count = 300 - leader * 10;
13360            let (unsigned, signed) = if leader == 0 {
13361                (Value::Null, Value::Null)
13362            } else {
13363                (Value::UBigInt(u64::MAX - leader), Value::BigInt(-(leader as i64)))
13364            };
13365            rows.extend(std::iter::repeat_n((unsigned.clone(), signed.clone()), count as usize));
13366            leaders.push(((unsigned, count), (signed, count)));
13367        }
13368        rows.extend((1_000..41_000_u64).map(|id| (Value::UBigInt(id), Value::BigInt(id as i64))));
13369        for part in rows.chunks(1_024) {
13370            let unsigned = part.iter().map(|(value, _)| value.clone()).collect::<Vec<_>>();
13371            let signed = part.iter().map(|(_, value)| value.clone()).collect::<Vec<_>>();
13372            let chunk = Chunk::new(vec![
13373                Vector::from_values(LogicalType::UBigInt, &unsigned).expect("unsigned"),
13374                Vector::from_values(LogicalType::BigInt, &signed).expect("signed"),
13375            ])
13376            .expect("matching columns");
13377            writer.append(&chunk).expect("rows");
13378        }
13379        writer.finish().expect("commit");
13380
13381        let reader = Reader::open(&path).expect("reopen from disk");
13382        for column in 0..2 {
13383            let prefix =
13384                reader.frequency_prefix(column).expect("valid metadata").expect("a synopsis");
13385            let wanted = leaders
13386                .iter()
13387                .map(|(unsigned, signed)| if column == 0 { unsigned } else { signed })
13388                .cloned()
13389                .collect::<Vec<_>>();
13390            assert_eq!(&prefix.entries[..10], &wanted[..], "column {column}");
13391            assert!(prefix.omitted_max < 210, "column {column}");
13392            assert_eq!(
13393                reader.distinct_values(column).expect("valid metadata"),
13394                Some(9 + 40_000),
13395                "column {column}"
13396            );
13397        }
13398        fs::remove_file(path).expect("remove scratch file");
13399    }
13400
13401    #[test]
13402    fn narrow_nonzero_count_matches_the_full_reader_across_stripes() {
13403        let path = path("quick-nonzero");
13404        let mut writer = Writer::create(
13405            &path,
13406            "items",
13407            vec![Field::new("label", LogicalType::Varchar), Field::new("id", LogicalType::Integer)],
13408        )
13409        .expect("create");
13410        for ids in [
13411            &[Value::Integer(0), Value::Null, Value::Integer(3)][..],
13412            &[Value::Integer(0), Value::Integer(7), Value::Null][..],
13413        ] {
13414            let labels = vec![Value::Varchar("same".into()); ids.len()];
13415            writer
13416                .append(
13417                    &Chunk::new(vec![
13418                        Vector::from_values(LogicalType::Varchar, &labels).expect("labels"),
13419                        Vector::from_values(LogicalType::Integer, ids).expect("ids"),
13420                    ])
13421                    .expect("chunk"),
13422                )
13423                .expect("append");
13424        }
13425        writer.finish().expect("finish");
13426        let catalog = Catalog::open(&path).expect("catalog");
13427        assert_eq!(catalog.entries[0].nonzero, vec![None, Some(2)]);
13428        assert_eq!(catalog.entries[0].aggregates, vec![None, Some((10, 4))]);
13429        assert_eq!(catalog.entries[0].distincts, vec![Some(1), Some(3)]);
13430        let frequencies =
13431            catalog.exact_numeric_frequencies("items", 1).expect("frequencies").expect("complete");
13432        assert_eq!(frequencies.len(), 4);
13433        for pair in [(Some(0), 2), (Some(3), 1), (Some(7), 1), (None, 2)] {
13434            assert!(frequencies.contains(&pair), "missing {pair:?}");
13435        }
13436        assert_eq!(catalog.distinct_count("items", 1).expect("distinct count"), Some(3));
13437        assert_eq!(
13438            catalog.integer_extremes("items", 1).expect("extremes"),
13439            Some(IntegerExtremes::Values { low: 0, high: 7 })
13440        );
13441        assert_eq!(
13442            catalog.aggregate_sums("items", &[1]).expect("catalog sums"),
13443            Some(CertifiedSums { columns: vec![(10, 4)], rows: 6 })
13444        );
13445        assert_eq!(catalog.nonzero_count("items", 1).expect("quick count"), Some(2));
13446        assert_eq!(
13447            reader_nonzero_counts(&catalog.table("items").expect("reader")).expect("counts"),
13448            vec![None, Some(2)]
13449        );
13450        Writer::certify_counts(&path).expect("recertify");
13451        assert_eq!(
13452            Catalog::open(&path).expect("reopen").nonzero_count("items", 1).expect("count"),
13453            Some(2)
13454        );
13455        assert_eq!(
13456            Catalog::open(&path).expect("reopen").aggregate_sums("items", &[1]).expect("sums"),
13457            Some(CertifiedSums { columns: vec![(10, 4)], rows: 6 })
13458        );
13459        assert_eq!(
13460            Catalog::open(&path).expect("reopen").distinct_count("items", 1).expect("distinct"),
13461            Some(3)
13462        );
13463        assert_eq!(
13464            Catalog::open(&path).expect("reopen").integer_extremes("items", 1).expect("ends"),
13465            Some(IntegerExtremes::Values { low: 0, high: 7 })
13466        );
13467        assert_eq!(
13468            Catalog::open(&path)
13469                .expect("reopen")
13470                .exact_numeric_frequencies("items", 1)
13471                .expect("frequencies"),
13472            Some(frequencies)
13473        );
13474        assert_eq!(catalog.table("items").expect("reader").null_count(1).expect("nulls"), 2);
13475        fs::remove_file(path).expect("remove scratch file");
13476    }
13477
13478    #[test]
13479    fn numeric_string_pair_leaders_are_certified_in_the_directory() {
13480        let path = path("pair-frequencies");
13481        let mut pairs = Vec::new();
13482        pairs.extend(std::iter::repeat_n((1_i64, "alpha".to_string()), 100));
13483        pairs.extend(std::iter::repeat_n((1_i64, "beta".to_string()), 50));
13484        pairs.extend(std::iter::repeat_n((2_i64, "gamma".to_string()), 40));
13485        pairs.extend((1_000_i64..1_600).map(|id| (id, format!("tail {id}"))));
13486        let mut writer = Writer::create(
13487            &path,
13488            "items",
13489            vec![
13490                Field::required("id", LogicalType::BigInt),
13491                Field::required("phrase", LogicalType::Varchar),
13492            ],
13493        )
13494        .expect("new file");
13495        for part in pairs.chunks(1_024) {
13496            let ids = part.iter().map(|(id, _)| Value::BigInt(*id)).collect::<Vec<_>>();
13497            let phrases =
13498                part.iter().map(|(_, phrase)| Value::Varchar(phrase.clone())).collect::<Vec<_>>();
13499            writer
13500                .append(
13501                    &Chunk::new(vec![
13502                        Vector::from_values(LogicalType::BigInt, &ids).expect("ids"),
13503                        Vector::from_values(LogicalType::Varchar, &phrases).expect("phrases"),
13504                    ])
13505                    .expect("matching columns"),
13506                )
13507                .expect("rows");
13508        }
13509        writer.finish().expect("commit");
13510
13511        let reader = Reader::open(&path).expect("reopen from disk");
13512        let leaders = reader
13513            .top_pair_frequencies(0, 1, 2)
13514            .expect("valid pair metadata")
13515            .expect("the top two beat the omitted tail");
13516        assert!(
13517            leaders.contains(&(vec![Value::BigInt(1), Value::Varchar("alpha".to_string())], 100,))
13518        );
13519        assert!(
13520            leaders.contains(&(vec![Value::BigInt(1), Value::Varchar("beta".to_string())], 50,))
13521        );
13522        fs::remove_file(path).expect("remove scratch file");
13523    }
13524
13525    /// The bug this is here for cost a 43 GB ClickBench table and an hour of reloading it. The
13526    /// format went from 11 to 12, every binary built after that said "magic or major version is
13527    /// unsupported" about the file, and there was no way to tell from the message whether the path
13528    /// was wrong, the file was truncated, or it was ours and simply older. The number this build
13529    /// wants is the whole answer and it was the one thing the message did not carry.
13530    #[test]
13531    fn a_file_from_another_format_says_which_format_it_is() {
13532        let older = path("older-format");
13533        let mut writer =
13534            Writer::create(&older, "items", vec![Field::new("id", LogicalType::Integer)])
13535                .expect("new file");
13536        let chunk = Chunk::new(vec![
13537            Vector::flat(LogicalType::Integer, Data::Int32(vec![1, 2, 3].into()))
13538                .expect("integers"),
13539        ])
13540        .expect("chunk");
13541        writer.append(&chunk).expect("page written");
13542        writer.finish().expect("commit");
13543
13544        // A format below the whole readable set, rather than `FORMAT - 1`, because the set has
13545        // more than one member now: format 22 is deliberately still readable, so the version that
13546        // has to be refused is the one under the oldest one accepted.
13547        let unreadable =
13548            READABLE.iter().copied().min().expect("at least one format is readable") - 1;
13549        let mut file = OpenOptions::new().write(true).open(&older).expect("open for the header");
13550        file.seek(SeekFrom::Start(8)).expect("the version follows the magic");
13551        file.write_all(&unreadable.to_le_bytes()).expect("write an older version");
13552        drop(file);
13553        let complaint = Reader::open(&older).expect_err("an older format is refused").to_string();
13554        assert!(complaint.contains(&format!("format {unreadable}")), "{complaint}");
13555        assert!(complaint.contains(&format!("format {FORMAT}")), "{complaint}");
13556
13557        let mut file = OpenOptions::new().write(true).open(&older).expect("open for the header");
13558        file.seek(SeekFrom::Start(0)).expect("the magic is first");
13559        file.write_all(b"NOTRUDB!").expect("write another engine's magic");
13560        drop(file);
13561        let complaint = Reader::open(&older).expect_err("a foreign file is refused").to_string();
13562        assert!(complaint.contains("magic"), "{complaint}");
13563        assert!(!complaint.contains("format"), "a version has nothing to do with it: {complaint}");
13564        fs::remove_file(older).expect("remove scratch file");
13565    }
13566
13567    #[test]
13568    fn an_unfinished_or_damaged_file_does_not_answer_with_partial_rows() {
13569        let unfinished = path("unfinished");
13570        let mut writer =
13571            Writer::create(&unfinished, "items", vec![Field::new("id", LogicalType::Integer)])
13572                .expect("new file");
13573        let chunk = Chunk::new(vec![
13574            Vector::flat(LogicalType::Integer, Data::Int32(vec![1, 2, 3].into()))
13575                .expect("integers"),
13576        ])
13577        .expect("chunk");
13578        writer.append(&chunk).expect("page written");
13579        drop(writer);
13580        assert!(Reader::open(&unfinished).is_err(), "no directory was committed");
13581        fs::remove_file(unfinished).expect("remove scratch file");
13582
13583        let damaged = path("damaged");
13584        let mut writer =
13585            Writer::create(&damaged, "items", vec![Field::new("id", LogicalType::Integer)])
13586                .expect("new file");
13587        writer.append(&chunk).expect("page written");
13588        writer.finish().expect("commit");
13589        let reader = Reader::open(&damaged).expect("valid directory");
13590        let mut file =
13591            OpenOptions::new().write(true).open(&damaged).expect("open for a damaged page");
13592        file.seek(SeekFrom::Start(HEADER + 1)).expect("inside first page");
13593        file.write_all(&[255]).expect("damage one byte");
13594        assert!(reader.read(0, &[0]).is_err(), "page checksum rejects corruption");
13595        fs::remove_file(damaged).expect("remove scratch file");
13596    }
13597
13598    #[test]
13599    fn damaged_lazy_dictionary_payload_is_an_error() {
13600        let path = path("damaged-dictionary");
13601        let mut writer = Writer::create(
13602            &path,
13603            "items",
13604            vec![
13605                Field::required("id", LogicalType::Integer),
13606                Field::new("text", LogicalType::Varchar),
13607            ],
13608        )
13609        .expect("new file");
13610        writer.append(&sample()).expect("stripe written");
13611        writer.finish().expect("commit");
13612
13613        let reader = Reader::open(&path).expect("valid directory");
13614        let dictionary = reader.table.dictionaries[1].expect("string dictionary page");
13615        // Read the count out of the page rather than writing it here, so that adding something
13616        // else to the index does not silently turn this into a test that damages the index.
13617        let mut header = [0; DICTIONARY_HEADER];
13618        read_at(&reader.file, dictionary.offset, &mut header).expect("dictionary header");
13619        // The first block's start is the first word after the offsets, since the blocks are written
13620        // during the load and are wherever the writer was when each was encoded.
13621        let count = u32::from_le_bytes(header[0..4].try_into().expect("four bytes")) as usize;
13622        let width = u32::from_le_bytes(header[12..16].try_into().expect("four bytes"));
13623        assert_ne!(width & DICTIONARY_SCATTERED, 0, "the blocks say where they are");
13624        let bits = (width & !(DICTIONARY_SCATTERED | DICTIONARY_GRAMS)) as usize;
13625        let mut start = [0; 8];
13626        let at = dictionary.offset + (DICTIONARY_HEADER + offset_bytes(count, bits)) as u64;
13627        read_at(&reader.file, at, &mut start).expect("the first block's start");
13628        let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
13629        file.seek(SeekFrom::Start(u64::from_le_bytes(start))).expect("inside dictionary payload");
13630        file.write_all(&[255]).expect("damage dictionary payload");
13631
13632        let chunk = reader.read(0, &[1]).expect("code page and dictionary index remain valid");
13633        let error =
13634            chunk.validate_external().expect_err("payload corruption must reach the caller");
13635        assert!(error.message().contains("payload checksum differs"), "{error}");
13636        fs::remove_file(path).expect("remove scratch file");
13637    }
13638
13639    /// A column whose values are all different is written without a dictionary, and one whose
13640    /// values repeat keeps it.
13641    ///
13642    /// The two columns go in the same table and hold the same number of rows, so the only thing
13643    /// separating them is how much of the first stripe was a value it had not seen before. Both have
13644    /// to read back the values that were written, because the decision is about cost and nothing
13645    /// else. The file size is the other half of it: a column written without a dictionary goes
13646    /// through the string cascade instead, so dropping the dictionary must not turn into storing the
13647    /// column raw.
13648    #[test]
13649    fn a_column_of_all_different_values_is_written_without_a_dictionary() {
13650        let path = path("dictionary-decide");
13651        let rows = 20_000;
13652        // Long enough that storing it raw would show, and different in every row.
13653        let unique =
13654            |row: usize| format!("{row:09} a value that appears exactly once in the table");
13655        // The same values in the same shape, each one used forty times over.
13656        let repeated = |row: usize| unique(row / 40);
13657        let mut writer = Writer::create(
13658            &path,
13659            "items",
13660            vec![
13661                Field::required("unique", LogicalType::Varchar),
13662                Field::required("repeated", LogicalType::Varchar),
13663            ],
13664        )
13665        .expect("new file");
13666        for part in (0..rows).step_by(1_000) {
13667            let span = part..(part + 1_000).min(rows);
13668            let left = span.clone().map(|row| Value::Varchar(unique(row))).collect::<Vec<_>>();
13669            let right = span.map(|row| Value::Varchar(repeated(row))).collect::<Vec<_>>();
13670            writer
13671                .append(
13672                    &Chunk::new(vec![
13673                        Vector::from_values(LogicalType::Varchar, &left).expect("strings"),
13674                        Vector::from_values(LogicalType::Varchar, &right).expect("strings"),
13675                    ])
13676                    .expect("two columns"),
13677                )
13678                .expect("a part");
13679        }
13680        writer.finish().expect("commit");
13681
13682        let reader = Reader::open(&path).expect("reopen from disk");
13683        assert!(
13684            reader.table.dictionaries[0].is_none(),
13685            "a column with no repeats has nothing to say twice"
13686        );
13687        assert!(
13688            reader.table.dictionaries[1].is_some(),
13689            "a column whose values come round again keeps its dictionary"
13690        );
13691        let mut first = 0;
13692        for part in 0..reader.parts() {
13693            let chunk = reader.read(part, &[0, 1]).expect("a part");
13694            for row in 0..chunk.len() {
13695                assert_eq!(chunk.value_at(row, 0), Value::Varchar(unique(first + row)));
13696                assert_eq!(chunk.value_at(row, 1), Value::Varchar(repeated(first + row)));
13697            }
13698            first += chunk.len();
13699        }
13700        assert_eq!(first, rows, "every row was read back");
13701        let raw = (0..rows).map(|row| unique(row).len()).sum::<usize>();
13702        let size = fs::metadata(&path).expect("the file is there").len() as usize;
13703        assert!(size < raw, "a column without a dictionary is still encoded: {size} against {raw}");
13704        fs::remove_file(path).expect("remove scratch file");
13705    }
13706
13707    /// A payload of many blocks reads and checks every block of it.
13708    ///
13709    /// The test above has a dictionary of three values, which is one block, so it says nothing
13710    /// about a reader finding the right block among many. This one has thirty two thousand values,
13711    /// which is thirty two blocks, and it reads a value out of the first block and a value out of
13712    /// the last and then damages the last and asks for it again.
13713    ///
13714    /// Forty thousand rows over those thirty two thousand values, because a column the writer finds
13715    /// to be all distinct does not get a dictionary at all and there would be nothing here to test.
13716    /// Four rows in five holding a value the stripe has not seen before is a column that keeps one.
13717    /// The repeats are put at the front so that the values still arrive in order after them, which
13718    /// is what keeps the last part of the table on the last block of the payload.
13719    #[test]
13720    fn a_dictionary_over_many_blocks_checks_every_block_of_it() {
13721        let path = path("dictionary-blocks");
13722        let value = |row: usize| {
13723            let row = row.saturating_sub(8_000);
13724            format!("{row:07} a value long enough to be worth a payload block")
13725        };
13726        let parts = 40;
13727        let per_part = 1000;
13728        let mut writer =
13729            Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
13730                .expect("new file");
13731        for part in 0..parts {
13732            let values = (0..per_part)
13733                .map(|row| Value::Varchar(value(part * per_part + row)))
13734                .collect::<Vec<_>>();
13735            let chunk = Chunk::new(vec![
13736                Vector::from_values(LogicalType::Varchar, &values).expect("strings"),
13737            ])
13738            .expect("matching rows");
13739            writer.append(&chunk).expect("a part");
13740        }
13741        writer.finish().expect("commit");
13742
13743        let reader = Reader::open(&path).expect("reopen from disk");
13744        let dictionary = reader.table.dictionaries[0].expect("string dictionary page");
13745        assert!(
13746            parts * per_part > TEXT_PAYLOAD_VALUES * 4,
13747            "the dictionary has to be several blocks for this to be testing anything"
13748        );
13749        for part in [0, parts - 1] {
13750            let chunk = reader.read(part, &[0]).expect("a part");
13751            chunk.validate_external().expect("every payload block checks out");
13752            assert_eq!(chunk.value_at(0, 0), Value::Varchar(value(part * per_part)));
13753        }
13754
13755        // The last block is wherever the writer was when it was encoded, which the index says.
13756        let mut header = [0; DICTIONARY_HEADER];
13757        read_at(&reader.file, dictionary.offset, &mut header).expect("dictionary header");
13758        let count = u32::from_le_bytes(header[0..4].try_into().expect("four bytes")) as usize;
13759        let blocks = u32::from_le_bytes(header[8..12].try_into().expect("four bytes")) as usize;
13760        let width = u32::from_le_bytes(header[12..16].try_into().expect("four bytes"));
13761        let bits = (width & !(DICTIONARY_SCATTERED | DICTIONARY_GRAMS)) as usize;
13762        let mut place = [0; 16];
13763        let at = DICTIONARY_HEADER + offset_bytes(count, bits) + (blocks - 1) * 16;
13764        read_at(&reader.file, dictionary.offset + at as u64, &mut place).expect("its place");
13765        let start = u64::from_le_bytes(place[..8].try_into().expect("eight bytes"));
13766        let length = u64::from_le_bytes(place[8..].try_into().expect("eight bytes"));
13767        let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
13768        file.seek(SeekFrom::Start(start + length - 4)).expect("the last bytes of the last block");
13769        file.write_all(&[255]).expect("damage the last payload block");
13770        let reader = Reader::open(&path).expect("the directory and the index are untouched");
13771        let chunk = reader.read(parts - 1, &[0]).expect("the code page remains valid");
13772        let error = chunk.validate_external().expect_err("the damage must reach the caller");
13773        assert!(error.message().contains("payload checksum differs"), "{error}");
13774        fs::remove_file(path).expect("remove scratch file");
13775    }
13776
13777    /// Values of different lengths read back where the offsets say they do.
13778    ///
13779    /// The offsets are packed at one width for the column, they are relative to the payload block a
13780    /// value lands in, and they go in runs of half a block, so there are two boundaries where the
13781    /// arithmetic could be off by one and neither shows up on values that are all the same length.
13782    /// This writes 5,000 values whose lengths cycle through a wide range and reads every one back,
13783    /// so the first value of a block, the last value of a run and the last value of a block are all
13784    /// covered several times over. An empty value is in the cycle because a zero length span is the
13785    /// case the reader short circuits.
13786    ///
13787    /// Six thousand rows over those 5,000 values, because a column the writer finds to be all
13788    /// distinct is written without a dictionary and then there are no packed offsets to be off by
13789    /// one in.
13790    #[test]
13791    fn values_of_different_lengths_read_back_out_of_packed_offsets() {
13792        let path = path("dictionary-offsets");
13793        let value = |row: usize| {
13794            let row = row % 5_000;
13795            if row % 511 == 3 { String::new() } else { "x".repeat(row % 97) + &format!("{row:05}") }
13796        };
13797        let rows = 6_000;
13798        let mut writer =
13799            Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
13800                .expect("new file");
13801        let values = (0..rows).map(|row| Value::Varchar(value(row))).collect::<Vec<_>>();
13802        for part in values.chunks(1_000) {
13803            let chunk =
13804                Chunk::new(vec![Vector::from_values(LogicalType::Varchar, part).expect("strings")])
13805                    .expect("matching rows");
13806            writer.append(&chunk).expect("a part");
13807        }
13808        writer.finish().expect("commit");
13809
13810        let reader = Reader::open(&path).expect("reopen from disk");
13811        assert!(
13812            rows > TEXT_PAYLOAD_VALUES * 4,
13813            "the dictionary has to be several blocks for this to be testing anything"
13814        );
13815        for part in 0..rows / 1_000 {
13816            let chunk = reader.read(part, &[0]).expect("a part");
13817            for row in 0..1_000 {
13818                let row = part * 1_000 + row;
13819                assert_eq!(
13820                    chunk.value_at(row % 1_000, 0),
13821                    Value::Varchar(value(row)),
13822                    "value {row}"
13823                );
13824            }
13825        }
13826        // The lengths a vector at a time, twice over, because the first pass is what makes the
13827        // table of ends worth building and the second is read out of the lengths worked out of it.
13828        for _ in 0..2 {
13829            for part in 0..rows / 1_000 {
13830                let chunk = reader.read(part, &[0]).expect("a part");
13831                let mut lens = vec![0_i64; 1_000];
13832                let column = chunk.column(0).expect("one column");
13833                assert!(column.try_bytes_lens(&mut lens).expect("lengths"), "a stored column");
13834                for (row, &len) in lens.iter().enumerate() {
13835                    let row = part * 1_000 + row;
13836                    assert_eq!(len as usize, value(row).len(), "the length of value {row}");
13837                }
13838            }
13839        }
13840        fs::remove_file(path).expect("remove scratch file");
13841    }
13842
13843    /// Lengths start again at every block, and ends that go backwards inside one give no table.
13844    #[test]
13845    fn lengths_restart_at_each_block_and_refuse_ends_that_go_backwards() {
13846        let mut ends: Vec<u32> = (1..=TEXT_PAYLOAD_VALUES as u32).map(|at| at * 2).collect();
13847        ends.extend([3, 3, 10]);
13848        let lens = lengths_of(&ends).expect("ordered ends");
13849        assert!(lens[..TEXT_PAYLOAD_VALUES].iter().all(|&len| len == 2));
13850        assert_eq!(&lens[TEXT_PAYLOAD_VALUES..], &[3, 0, 7]);
13851        ends.push(9);
13852        assert_eq!(lengths_of(&ends), None);
13853    }
13854
13855    /// Every worker of a scan wants the dictionary at the same moment and one of them fetches it.
13856    ///
13857    /// Asking a `OnceLock` whether it holds something answers the question a worker that already has
13858    /// the dictionary is asking and not the one a worker without it is asking, which is whether
13859    /// somebody is already on their way with it. Sixteen workers that all miss will all read the
13860    /// page, all verify it and all decode it, and fifteen will drop the result. Nothing about that
13861    /// is incorrect, which is why it went unnoticed, and it showed up as ClickBench 38 getting
13862    /// slower when the scan in front of it got faster and stopped staggering the arrivals.
13863    ///
13864    /// The barrier is what makes the test about that rather than about luck. Without it the first
13865    /// thread is usually finished before the last one starts and the count is one either way.
13866    #[test]
13867    fn a_global_dictionary_is_opened_once_however_many_workers_ask_at_once() {
13868        let path = path("dictionary-once");
13869        let parts = 8;
13870        let per_part = 500;
13871        let value =
13872            |row: usize| format!("{row:07} a value long enough to be worth a payload block");
13873        let mut writer =
13874            Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
13875                .expect("new file");
13876        for part in 0..parts {
13877            let values = (0..per_part)
13878                .map(|row| Value::Varchar(value(part * per_part + row)))
13879                .collect::<Vec<_>>();
13880            let chunk = Chunk::new(vec![
13881                Vector::from_values(LogicalType::Varchar, &values).expect("strings"),
13882            ])
13883            .expect("matching rows");
13884            writer.append(&chunk).expect("a part");
13885        }
13886        writer.finish().expect("commit");
13887
13888        let reader = Reader::open(&path).expect("reopen from disk");
13889        assert!(reader.table.dictionaries[0].is_some(), "the column has to have one to share");
13890        assert_eq!(reader.reads().dictionaries, 0, "opening the file does not open a dictionary");
13891
13892        let workers = 16;
13893        let gate = std::sync::Barrier::new(workers);
13894        std::thread::scope(|scope| {
13895            for worker in 0..workers {
13896                let reader = reader.clone();
13897                let gate = &gate;
13898                scope.spawn(move || {
13899                    gate.wait();
13900                    let chunk = reader.read(worker % parts, &[0]).expect("a part");
13901                    assert_eq!(
13902                        chunk.value_at(0, 0),
13903                        Value::Varchar(value((worker % parts) * per_part))
13904                    );
13905                });
13906            }
13907        });
13908
13909        assert_eq!(reader.reads().dictionaries, 1, "sixteen workers, one dictionary, one open");
13910        fs::remove_file(path).expect("remove scratch file");
13911    }
13912
13913    /// The sorted order sits outside the index the page checksum covers, because a query that
13914    /// never searches a dictionary should not read it, so it carries its own checksums and this is
13915    /// what says they are checked. A search that trusted a damaged order would give a wrong answer
13916    /// rather than a slow one.
13917    #[test]
13918    fn a_damaged_sorted_order_is_an_error() {
13919        let path = path("damaged-order");
13920        let mut writer = Writer::create(
13921            &path,
13922            "items",
13923            vec![
13924                Field::required("id", LogicalType::Integer),
13925                Field::new("text", LogicalType::Varchar),
13926            ],
13927        )
13928        .expect("new file");
13929        writer.append(&sample()).expect("stripe written");
13930        writer.finish().expect("commit");
13931
13932        let reader = Reader::open(&path).expect("valid directory");
13933        let page = reader.table.dictionaries[1].expect("string dictionary page");
13934        let mut header = [0; DICTIONARY_HEADER];
13935        read_at(&reader.file, page.offset, &mut header).expect("dictionary header");
13936        let index_len = dictionary_index_len(&header);
13937        let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
13938        file.seek(SeekFrom::Start(page.offset + index_len)).expect("the first head");
13939        file.write_all(&[255]).expect("damage the order");
13940
13941        let dictionary = reader.dictionary(1).expect("read").expect("a string column has one");
13942        let error = dictionary.compare_rank(0, b"anything").expect_err("a damaged order is caught");
13943        assert!(error.message().contains("rank checksum differs"), "{error}");
13944        fs::remove_file(path).expect("remove scratch file");
13945    }
13946
13947    /// Codes stay in first appearance order and the sorted order is written beside them, so a
13948    /// reader can put the values back in order without the writer having had to know them all
13949    /// before it handed out the first code.
13950    #[test]
13951    fn a_global_dictionary_carries_the_sorted_order_of_its_values() {
13952        // Chosen so the sort cannot be decided on the first eight bytes alone. Three values share
13953        // a nine byte prefix, one is a prefix of another, and one is empty.
13954        let spellings = ["overlong1z", "b", "", "overlong1a", "overlong", "ab", "a", "overlong1"];
13955        let path = path("dictionary-order");
13956        let mut writer =
13957            Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
13958                .expect("new file");
13959        writer
13960            .append(
13961                &Chunk::new(vec![
13962                    Vector::from_values(
13963                        LogicalType::Varchar,
13964                        &spellings.map(|text| Value::Varchar(text.into())),
13965                    )
13966                    .expect("strings"),
13967                ])
13968                .expect("one column"),
13969            )
13970            .expect("stripe written");
13971        writer.finish().expect("commit");
13972
13973        let reader = Reader::open(&path).expect("valid directory");
13974        let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
13975        let count = dictionary.ranks().expect("a v10 file stores one");
13976        assert_eq!(count, spellings.len(), "every distinct value has a rank");
13977        let order = (0..count)
13978            .map(|rank| dictionary.code_at_rank(rank).expect("a code"))
13979            .collect::<Vec<_>>();
13980        let mut seen = order.clone();
13981        seen.sort_unstable();
13982        assert_eq!(seen, (0..spellings.len() as u32).collect::<Vec<_>>(), "a permutation of codes");
13983
13984        let ranked = order
13985            .iter()
13986            .map(|&code| {
13987                dictionary.try_bytes_at(code as usize).expect("read").expect("a value").to_vec()
13988            })
13989            .collect::<Vec<_>>();
13990        let mut expected = spellings.map(|text| text.as_bytes().to_vec()).to_vec();
13991        expected.sort();
13992        assert_eq!(ranked, expected, "rank order is value order");
13993
13994        // What a search asks, on the values themselves rather than through a kernel, so that a
13995        // file whose heads disagree with its bytes is caught here rather than as a wrong answer.
13996        for (rank, value) in expected.iter().enumerate() {
13997            assert_eq!(
13998                dictionary.compare_rank(rank, value).expect("compare"),
13999                Ordering::Equal,
14000                "rank {rank} is its own value"
14001            );
14002            if rank > 0 {
14003                assert_eq!(
14004                    dictionary.compare_rank(rank - 1, value).expect("compare"),
14005                    Ordering::Less,
14006                    "rank {rank} follows the one before it"
14007                );
14008            }
14009        }
14010        fs::remove_file(path).expect("remove scratch file");
14011    }
14012
14013    /// Five text columns of different sizes close at the same time, and each comes back with its
14014    /// own values in its own order.
14015    ///
14016    /// The sizes differ so that the columns are taken in an order that is not the column order, and
14017    /// the values of each column are spelled with its number so that one column's page written in
14018    /// another's place would read back as the wrong strings rather than the right ones by chance.
14019    #[test]
14020    fn text_columns_closed_at_once_each_keep_their_own_dictionary() {
14021        let sizes = [300_usize, 5_000, 40, 2_000, 1_200];
14022        let path = path("dictionaries-at-once");
14023        let fields = (0..sizes.len())
14024            .map(|column| Field::new(format!("text{column}"), LogicalType::Varchar))
14025            .collect::<Vec<_>>();
14026        let mut writer = Writer::create(&path, "items", fields).expect("new file");
14027        let rows = 10_000_usize;
14028        for start in (0..rows).step_by(1_024) {
14029            let columns = sizes
14030                .iter()
14031                .enumerate()
14032                .map(|(column, &size)| {
14033                    let values = (start..(start + 1_024).min(rows))
14034                        .map(|row| Value::Varchar(format!("c{column}-{:05}", (row * 7919) % size)))
14035                        .collect::<Vec<_>>();
14036                    Vector::from_values(LogicalType::Varchar, &values).expect("strings")
14037                })
14038                .collect::<Vec<_>>();
14039            writer.append(&Chunk::new(columns).expect("five columns")).expect("stripe written");
14040        }
14041        writer.finish().expect("commit");
14042
14043        let reader = Reader::open(&path).expect("valid directory");
14044        for (column, &size) in sizes.iter().enumerate() {
14045            let dictionary =
14046                reader.dictionary(column).expect("read").expect("a string column has one");
14047            let count = dictionary.ranks().expect("a v10 file stores one");
14048            assert_eq!(count, size, "column {column} has its own distinct count");
14049            let ranked = (0..count)
14050                .map(|rank| {
14051                    let code = dictionary.code_at_rank(rank).expect("a code");
14052                    dictionary.try_bytes_at(code as usize).expect("read").expect("a value").to_vec()
14053                })
14054                .collect::<Vec<_>>();
14055            let expected = (0..size)
14056                .map(|value| format!("c{column}-{value:05}").into_bytes())
14057                .collect::<Vec<_>>();
14058            assert_eq!(ranked, expected, "column {column} ranks its own values in order");
14059        }
14060        fs::remove_file(path).expect("remove scratch file");
14061    }
14062
14063    /// A dictionary large enough to be decoded and sorted on several threads ranks the way one small
14064    /// enough for one thread does.
14065    ///
14066    /// Seventy thousand values over sixty nine blocks, in no order and each four times over so the
14067    /// column is worth a dictionary, written and ranked in the close.
14068    /// Some share a long prefix and some differ only in the last byte, so the buckets of the sort cut
14069    /// through runs of values that agree for a long way.
14070    #[test]
14071    fn a_large_dictionary_ranks_in_value_order() {
14072        let path = path("dictionary-large-rank");
14073        let value = |row: u64| {
14074            let mixed = row.wrapping_mul(0x9e37_79b9_7f4a_7c15) >> 40;
14075            match row % 3 {
14076                0 => format!("https://example.com/a/long/shared/path/{mixed:08}"),
14077                1 => format!("{mixed}"),
14078                _ => format!("x{}", row % 1000).repeat(1 + (row % 4) as usize) + &row.to_string(),
14079            }
14080        };
14081        let distinct = 70_000;
14082        let parts = 4 * distinct / 1000;
14083        let per_part = 1000;
14084        let mut writer =
14085            Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
14086                .expect("new file");
14087        for part in 0..parts {
14088            let values = (0..per_part)
14089                .map(|row| Value::Varchar(value((part * per_part + row) / 4)))
14090                .collect::<Vec<_>>();
14091            let chunk = Chunk::new(vec![
14092                Vector::from_values(LogicalType::Varchar, &values).expect("strings"),
14093            ])
14094            .expect("matching rows");
14095            writer.append(&chunk).expect("a part");
14096        }
14097        writer.finish().expect("commit");
14098
14099        let reader = Reader::open(&path).expect("reopen from disk");
14100        let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
14101        let count = dictionary.ranks().expect("a ranked dictionary");
14102        assert_eq!(count, distinct as usize, "every distinct value has a rank");
14103        assert!(count >= PARALLEL_SORT_MIN, "too few values to be sorted on more than one thread");
14104        let ranked = (0..count)
14105            .map(|rank| {
14106                let code = dictionary.code_at_rank(rank).expect("a code");
14107                dictionary.try_bytes_at(code as usize).expect("read").expect("a value").to_vec()
14108            })
14109            .collect::<Vec<_>>();
14110        let mut expected = (0..distinct).map(|row| value(row).into_bytes()).collect::<Vec<_>>();
14111        expected.sort();
14112        assert_eq!(ranked, expected, "rank order is value order");
14113        fs::remove_file(path).expect("remove scratch file");
14114    }
14115
14116    /// A string column's synopsis is turned into values without keeping the blocks it went through.
14117    ///
14118    /// Three thousand values, every fifth of them four times over, so the synopsis is a prefix of
14119    /// five hundred and twelve codes spread over all three payload blocks. Reading it used to leave
14120    /// all three decoded for as long as the reader lived. It leaves none of them now, and the second
14121    /// read answers out of what the first remembered.
14122    /// A directory read out of the file a window at a time is the directory read whole.
14123    ///
14124    /// The windows here are far smaller than any field is long, so every kind of field is split
14125    /// across a refill somewhere, and a bound is offered to its codec short more than once. The
14126    /// synopses are left in the file, and each one read back from where it was left is the one the
14127    /// whole read decoded.
14128    #[test]
14129    fn a_directory_read_a_window_at_a_time_is_the_directory_read_whole() {
14130        let path = path("windowed-directory");
14131        let fields = vec![
14132            Field::required("id", LogicalType::BigInt),
14133            Field::required("word", LogicalType::Varchar),
14134            Field::new("score", LogicalType::Double),
14135        ];
14136        let mut writer = Writer::create(&path, "items", fields).expect("new file");
14137        for part in 0..70_i64 {
14138            let ids = (0..100).map(|row| Value::BigInt(part * 100 + row % 7)).collect::<Vec<_>>();
14139            let words = (0..100)
14140                .map(|row| Value::Varchar(format!("word {}", row % 13)))
14141                .collect::<Vec<_>>();
14142            let scores = (0..100)
14143                .map(|row| if row % 4 == 0 { Value::Null } else { Value::Double(row as f64) })
14144                .collect::<Vec<_>>();
14145            let chunk = Chunk::new(vec![
14146                Vector::from_values(LogicalType::BigInt, &ids).expect("integers"),
14147                Vector::from_values(LogicalType::Varchar, &words).expect("strings"),
14148                Vector::from_values(LogicalType::Double, &scores).expect("doubles"),
14149            ])
14150            .expect("three columns");
14151            writer.append(&chunk).expect("a part");
14152        }
14153        writer.finish().expect("commit");
14154
14155        let catalog = Catalog::open(&path).expect("reopen");
14156        let entry = catalog.entries.first().expect("one table").directory;
14157        let (offset, length) = (entry.offset, entry.length as usize);
14158        let mut bytes = vec![0; length];
14159        read_at(&catalog.file, offset, &mut bytes).expect("the directory");
14160        assert_eq!(file_checksum(&catalog.file, offset, length).expect("checksum"), entry.hash);
14161        let whole = decode_directory(&bytes, catalog.size).expect("whole");
14162        assert!(whole.stripes.len() > 1, "the table should span stripes");
14163        for size in [1, 7, 33, 4_096] {
14164            let mut cursor = Cursor::over(&catalog.file, offset, length);
14165            cursor.window.as_mut().expect("a window").size = size;
14166            let windowed = read_directory(cursor, catalog.size, Some(offset)).expect("windowed");
14167            assert_eq!(format!("{:?}", windowed.stripes), format!("{:?}", whole.stripes));
14168            assert_eq!(format!("{:?}", windowed.fields), format!("{:?}", whole.fields));
14169            let mut stored = 0;
14170            for (column, (left, held)) in
14171                windowed.frequencies.iter().zip(&whole.frequencies).enumerate()
14172            {
14173                match (left, held) {
14174                    (None, None) => {}
14175                    (
14176                        Some(super::Frequencies::Stored { span, values }),
14177                        Some(super::Frequencies::Held(summary)),
14178                    ) => {
14179                        let mut one = vec![0; span.length as usize];
14180                        read_at(&catalog.file, span.offset, &mut one).expect("a synopsis");
14181                        let read = decode_summary(
14182                            &mut Cursor::new(&one),
14183                            &whole.fields[column],
14184                            whole.rows,
14185                            *values,
14186                        )
14187                        .expect("a valid synopsis")
14188                        .expect("one is there");
14189                        assert_eq!(format!("{read:?}"), format!("{summary:?}"));
14190                        stored += 1;
14191                    }
14192                    other => panic!("column {column} came back as {other:?}"),
14193                }
14194            }
14195            assert!(stored >= 2, "only {stored} synopses were left in the file");
14196        }
14197        let reader = catalog.table("items").expect("the table");
14198        assert!(reader.frequency_summaries[1].get().is_none());
14199        assert!(reader.top_frequencies(1, 1).expect("a readable synopsis").is_some());
14200        let first = reader.frequency_summaries[1].get().expect("decoded synopsis");
14201        let clone = reader.clone();
14202        assert!(clone.top_frequencies(1, 1).expect("cached synopsis").is_some());
14203        assert!(Arc::ptr_eq(first, clone.frequency_summaries[1].get().expect("same synopsis")));
14204        fs::remove_file(path).expect("remove scratch file");
14205    }
14206
14207    #[test]
14208    fn a_checksum_carried_across_reads_is_the_checksum_of_the_whole() {
14209        let path = path("file-checksum");
14210        let bytes = (0..200_000_u32)
14211            .map(|at| (at.wrapping_mul(2_654_435_761) >> 13) as u8)
14212            .collect::<Vec<_>>();
14213        fs::write(&path, &bytes).expect("scratch file");
14214        let file = File::open(&path).expect("open");
14215        for (offset, length) in [
14216            (0, 0),
14217            (3, 1),
14218            (5, 31),
14219            (0, 32),
14220            (9, 33),
14221            (1, 65_536),
14222            (7, 65_567),
14223            (0, 200_000),
14224            (11, 131_101),
14225        ] {
14226            let whole = checksum(&bytes[offset..offset + length]);
14227            assert_eq!(
14228                file_checksum(&file, offset as u64, length).expect("read"),
14229                whole,
14230                "{offset} {length}"
14231            );
14232        }
14233        fs::remove_file(path).expect("remove scratch file");
14234    }
14235
14236    #[test]
14237    fn a_string_synopsis_is_read_without_keeping_the_dictionary_blocks() {
14238        let path = path("synopsis-keeps-no-block");
14239        let spelled = |index: usize| Value::Varchar(format!("phrase {index:05}"));
14240        let mut values = (0..3_000).map(spelled).collect::<Vec<_>>();
14241        for _ in 0..3 {
14242            values.extend((0..3_000).step_by(5).map(spelled));
14243        }
14244        let mut writer =
14245            Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
14246                .expect("new file");
14247        for part in values.chunks(1_024) {
14248            writer
14249                .append(
14250                    &Chunk::new(vec![
14251                        Vector::from_values(LogicalType::Varchar, part).expect("strings"),
14252                    ])
14253                    .expect("one column"),
14254                )
14255                .expect("a part");
14256        }
14257        writer.finish().expect("commit");
14258
14259        let reader = Reader::open(&path).expect("reopen from disk");
14260        let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
14261        let resting = dictionary.footprint();
14262        let prefix = reader.frequency_prefix(0).expect("a readable synopsis").expect("one");
14263        assert_eq!(prefix.entries.len(), 512);
14264        for (value, count) in &prefix.entries {
14265            let Value::Varchar(text) = value else { panic!("a string column gave {value:?}") };
14266            let index = text["phrase ".len()..].parse::<usize>().expect("a spelled number");
14267            assert_eq!((index % 5, *count), (0, 4), "{text} came back with {count}");
14268        }
14269        assert_eq!(dictionary.footprint(), resting, "reading the synopsis kept a decoded block");
14270        let again = reader.frequency_prefix(0).expect("a readable synopsis").expect("one");
14271        assert_eq!(again.entries, prefix.entries);
14272        fs::remove_file(path).expect("remove scratch file");
14273    }
14274
14275    /// A sweep of the dictionary reads every value, and the second sweep keeps what it read, up to
14276    /// the budget.
14277    ///
14278    /// The point of the sweep is the resident size rather than the answer, so both are checked
14279    /// here. The first sweep keeps nothing, because a process that runs one statement never reads
14280    /// a block twice. A dictionary this small is well under [`TEXT_KEEP_BUDGET`], so the second
14281    /// sweep keeps everything and a third decodes nothing, which is what makes a session asking the
14282    /// same question again cost what it should. The ceiling is the other half of it and it has its own
14283    /// test below, because a ceiling that never binds is not a ceiling anybody checked.
14284    #[test]
14285    fn a_dictionary_sweep_reads_every_value_and_keeps_it_under_the_budget() {
14286        let path = path("dictionary-sweep");
14287        // Two thousand five hundred distinct values is two whole payload blocks and a part of a
14288        // third, so the sweep has to be called more than once and the last call has to stop short.
14289        let spellings = (0..2_500)
14290            .map(|index| Value::Varchar(format!("value {index:08} {}", "x".repeat(index % 40))))
14291            .collect::<Vec<_>>();
14292        let mut writer =
14293            Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
14294                .expect("new file");
14295        // A chunk is a part and a part is at most 1,024 rows, so the values go in three of them.
14296        // The dictionary is table wide and does not care where a value was written.
14297        for part in spellings.chunks(1_024) {
14298            writer
14299                .append(
14300                    &Chunk::new(vec![
14301                        Vector::from_values(LogicalType::Varchar, part).expect("strings"),
14302                    ])
14303                    .expect("one column"),
14304                )
14305                .expect("stripe written");
14306        }
14307        writer.finish().expect("commit");
14308
14309        let reader = Reader::open(&path).expect("valid directory");
14310        let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
14311        assert_eq!(dictionary.len(), spellings.len(), "every value is distinct");
14312        for first in [0, TEXT_PAYLOAD_VALUES, TEXT_PAYLOAD_VALUES * 2] {
14313            assert!(dictionary.text_block_might_contain(first, b"value").expect("signature"));
14314            assert!(!dictionary.text_block_might_contain(first, b"google").expect("signature"));
14315        }
14316
14317        let resting = dictionary.footprint();
14318        let sweep = || {
14319            let mut swept: Vec<Vec<u8>> = Vec::new();
14320            let mut at = 0;
14321            let mut calls = 0;
14322            while at < dictionary.len() {
14323                let stopped = dictionary
14324                    .sweep_text(at, dictionary.len(), &mut |index: usize, text: &[u8]| {
14325                        assert_eq!(index, swept.len(), "a sweep hands its values over in order");
14326                        swept.push(text.to_vec());
14327                        Ok(())
14328                    })
14329                    .expect("a sweep reads");
14330                assert!(stopped > at, "a sweep moves");
14331                at = stopped;
14332                calls += 1;
14333            }
14334            assert_eq!(calls, 3, "a sweep hands over one block at a time");
14335            swept
14336        };
14337        let swept = sweep();
14338        assert_eq!(dictionary.footprint(), resting, "a first sweep keeps nothing it decoded");
14339        assert_eq!(sweep(), swept, "a second sweep reads what the first did");
14340        let after = dictionary.footprint();
14341        assert!(after > resting, "a second sweep under the budget keeps what it decoded");
14342
14343        let read = (0..dictionary.len())
14344            .map(|code| dictionary.try_bytes_at(code).expect("read").expect("a value").to_vec())
14345            .collect::<Vec<_>>();
14346        assert_eq!(swept, read, "a sweep answers what a point read answers");
14347        // A read per value is about what makes the unpacked ends worth building, so whether they
14348        // are built here depends on how many reads the sweep made on the way. They are the one thing
14349        // allowed to grow, by four bytes a value, and nothing of the payload is.
14350        let grown = dictionary.footprint() - after;
14351        assert!(
14352            grown == 0 || grown == dictionary.len() * size_of::<u32>(),
14353            "a point read of a kept block decodes nothing, and {grown} bytes grew"
14354        );
14355        fs::remove_file(path).expect("remove scratch file");
14356    }
14357
14358    #[test]
14359    fn a_damaged_substring_signature_is_checked_only_when_used() {
14360        let path = path("damaged-substring-signature");
14361        let mut writer =
14362            Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
14363                .expect("new file");
14364        let rows = [Value::Varchar("google".into()), Value::Varchar("example".into())];
14365        writer
14366            .append(
14367                &Chunk::new(vec![
14368                    Vector::from_values(LogicalType::Varchar, &rows).expect("strings"),
14369                ])
14370                .expect("one column"),
14371            )
14372            .expect("stripe written");
14373        writer.finish().expect("commit");
14374
14375        let reader = Reader::open(&path).expect("valid directory");
14376        let page = reader.table.dictionaries[0].expect("string dictionary page");
14377        let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
14378        file.seek(SeekFrom::Start(page.offset + u64::from(page.length) - 1))
14379            .expect("last signature byte");
14380        file.write_all(&[255]).expect("damage signature");
14381        let reader = Reader::open(&path).expect("the directory is still valid");
14382        let dictionary = reader.dictionary(0).expect("index is still valid").expect("dictionary");
14383        let error = dictionary
14384            .text_block_might_contain(0, b"goog")
14385            .expect_err("a used signature checks its own checksum");
14386        assert!(error.message().contains("substring signatures checksum differs"), "{error}");
14387        fs::remove_file(path).expect("remove scratch file");
14388    }
14389
14390    /// A sweep over a block whose second run of offsets is short reads the same values as a point
14391    /// read does.
14392    ///
14393    /// The sweep decodes the offsets of a whole run at a time rather than a value at a time, and a
14394    /// run holds half a block, so the count it asks for is the run length everywhere but at the end
14395    /// of the dictionary. Two thousand five hundred values, which is what the test above writes,
14396    /// never puts a short run second in its block: the last block there begins on a run boundary and
14397    /// holds one run. Two thousand eight hundred does, so the last block is a whole run of five
14398    /// hundred and twelve followed by two hundred and forty, and an off by one in either the count
14399    /// asked for or the slice taken out of the answer shows up as a wrong value or a refusal.
14400    #[test]
14401    fn a_sweep_over_a_block_with_a_short_second_run_reads_what_a_point_read_reads() {
14402        let path = path("dictionary-sweep-short-run");
14403        let spellings = (0..2_800)
14404            .map(|index| Value::Varchar(format!("value {index:08} {}", "x".repeat(index % 40))))
14405            .collect::<Vec<_>>();
14406        let mut writer =
14407            Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
14408                .expect("new file");
14409        for part in spellings.chunks(1_024) {
14410            writer
14411                .append(
14412                    &Chunk::new(vec![
14413                        Vector::from_values(LogicalType::Varchar, part).expect("strings"),
14414                    ])
14415                    .expect("one column"),
14416                )
14417                .expect("stripe written");
14418        }
14419        writer.finish().expect("commit");
14420
14421        let reader = Reader::open(&path).expect("valid directory");
14422        let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
14423        assert_eq!(dictionary.len(), spellings.len(), "every value is distinct");
14424        let last = dictionary.len() % TEXT_PAYLOAD_VALUES;
14425        assert!(last > TEXT_OFFSET_RUN, "the last block has to reach into a second run of offsets");
14426        assert!(last < TEXT_PAYLOAD_VALUES, "and that second run has to be short of a whole one");
14427
14428        let mut swept: Vec<Vec<u8>> = Vec::new();
14429        let mut at = 0;
14430        while at < dictionary.len() {
14431            let stopped = dictionary
14432                .sweep_text(at, dictionary.len(), &mut |index: usize, text: &[u8]| {
14433                    assert_eq!(index, swept.len(), "a sweep hands its values over in order");
14434                    swept.push(text.to_vec());
14435                    Ok(())
14436                })
14437                .expect("a sweep reads");
14438            assert!(stopped > at, "a sweep moves");
14439            at = stopped;
14440        }
14441        let read = (0..dictionary.len())
14442            .map(|code| dictionary.try_bytes_at(code).expect("read").expect("a value").to_vec())
14443            .collect::<Vec<_>>();
14444        assert_eq!(swept, read, "a sweep answers what a point read answers");
14445        fs::remove_file(path).expect("remove scratch file");
14446    }
14447
14448    /// The unpacked ends answer what the packed ends answer, on both sides of the switch.
14449    ///
14450    /// A column asked for one offset at a time reads them out of the packed form until the reads
14451    /// are worth a table and out of the table after that, so every value here is read twice and the
14452    /// two passes are compared against the spellings and against each other. Two thousand eight
14453    /// hundred values is two payload blocks and a bit, which puts the switch in the middle of the
14454    /// first pass and means the pass straddles a block boundary, where the start of a value is zero
14455    /// rather than the end of the value before it.
14456    #[test]
14457    fn the_unpacked_ends_answer_what_the_packed_ends_answer() {
14458        let path = path("dictionary-unpacked-ends");
14459        let spellings = (0..2_800)
14460            .map(|index| Value::Varchar(format!("value {index:08} {}", "x".repeat(index % 40))))
14461            .collect::<Vec<_>>();
14462        let mut writer =
14463            Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
14464                .expect("new file");
14465        for part in spellings.chunks(1_024) {
14466            writer
14467                .append(
14468                    &Chunk::new(vec![
14469                        Vector::from_values(LogicalType::Varchar, part).expect("strings"),
14470                    ])
14471                    .expect("one column"),
14472                )
14473                .expect("stripe written");
14474        }
14475        writer.finish().expect("commit");
14476
14477        let reader = Reader::open(&path).expect("valid directory");
14478        let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
14479        assert_eq!(dictionary.len(), spellings.len(), "every value is distinct");
14480        let wanted = (0..spellings.len())
14481            .map(|index| format!("value {index:08} {}", "x".repeat(index % 40)).into_bytes())
14482            .collect::<Vec<_>>();
14483
14484        let pass = |what: &str| {
14485            for (index, value) in wanted.iter().enumerate() {
14486                let len = dictionary.try_bytes_len_at(index).expect("read").expect("a value");
14487                assert_eq!(len, value.len(), "{what} has the wrong length at {index}");
14488                let bytes = dictionary.try_bytes_at(index).expect("read").expect("a value");
14489                assert_eq!(bytes, value.as_slice(), "{what} has the wrong value at {index}");
14490            }
14491        };
14492        pass("the first pass");
14493        pass("the second pass");
14494
14495        // The whole vector in one call, over the text and through codes into it, which is how a
14496        // scan of a stored column hands it out. The codes run backwards and repeat so that they are
14497        // neither the positions nor in order.
14498        let lens = wanted.iter().map(|value| value.len() as i64).collect::<Vec<_>>();
14499        let mut whole = vec![0i64; wanted.len()];
14500        assert!(dictionary.try_bytes_lens(&mut whole).expect("read"), "the text answers whole");
14501        assert_eq!(whole, lens, "a vector of lengths answers what a length at a time answers");
14502        let codes = (0..4_000_u32).map(|row| (7 * (4_000 - row)) % 2_800).collect::<Vec<_>>();
14503        let coded = Vector::dictionary_over(codes.clone(), dictionary).expect("codes in range");
14504        let mut through = vec![0i64; codes.len()];
14505        assert!(coded.try_bytes_lens(&mut through).expect("read"), "the codes answer whole");
14506        for (row, &code) in codes.iter().enumerate() {
14507            assert_eq!(through[row], lens[code as usize], "row {row} reads code {code}");
14508            let one = coded.try_bytes_len_at(row).expect("read").expect("a value");
14509            assert_eq!(through[row], one as i64, "row {row} a row at a time");
14510        }
14511
14512        // A handful of codes over a column nobody has read yet is short of the table, so the same
14513        // call answers out of the packed ends instead, and has to answer the same.
14514        let fresh = Reader::open(&path).expect("valid directory");
14515        let untouched = fresh.dictionary(0).expect("read").expect("a string column has one");
14516        let few = vec![2_799_u32, 0, 1_024, 1_023, 511, 512];
14517        let coded = Vector::dictionary_over(few.clone(), untouched).expect("in range");
14518        let mut short = vec![0i64; few.len()];
14519        assert!(coded.try_bytes_lens(&mut short).expect("read"), "the codes answer whole");
14520        let expected = few.iter().map(|&code| lens[code as usize]).collect::<Vec<_>>();
14521        assert_eq!(short, expected, "the packed ends answer what the table answers");
14522        fs::remove_file(path).expect("remove scratch file");
14523    }
14524
14525    /// Narrowing a page takes what fits and refuses the page for anything that does not.
14526    ///
14527    /// The edges of the range on both sides and one step past each of them, for every type, because
14528    /// checking a page separately from converting it is only right if the check refuses exactly what
14529    /// `TryFrom` would have refused, and off by one there is a file that reads back a different
14530    /// number than it was given. The check is a bit pattern rather than a comparison, so it is not
14531    /// the shape a reader would guess from the bounds, which is why all six are here. The empty page
14532    /// is here because a check written the obvious way starts with the extremes the wrong way round
14533    /// and refuses it.
14534    #[test]
14535    fn narrowing_a_page_takes_what_fits_and_refuses_what_does_not() {
14536        assert_eq!(fit::<i8>(&[]).expect("an empty page fits anything"), Vec::<i8>::new());
14537        assert_eq!(fit::<i8>(&[-128, 0, 127]).expect("the edges fit"), vec![-128_i8, 0, 127]);
14538        fit::<i8>(&[128]).expect_err("one past the top does not fit");
14539        fit::<i8>(&[-129]).expect_err("one past the bottom does not fit");
14540        assert_eq!(fit::<u8>(&[0, 255]).expect("the edges fit"), vec![0_u8, 255]);
14541        fit::<u8>(&[256]).expect_err("one past the top does not fit");
14542        fit::<u8>(&[-1]).expect_err("a negative does not fit an unsigned page");
14543        assert_eq!(
14544            fit::<i16>(&[-32_768, 0, 32_767]).expect("the edges fit"),
14545            vec![-32_768_i16, 0, 32_767]
14546        );
14547        fit::<i16>(&[32_768]).expect_err("one past the top does not fit");
14548        fit::<i16>(&[-32_769]).expect_err("one past the bottom does not fit");
14549        assert_eq!(fit::<u16>(&[0, 65_535]).expect("the edges fit"), vec![0_u16, 65_535]);
14550        fit::<u16>(&[65_536]).expect_err("one past the top does not fit");
14551        fit::<u16>(&[-1]).expect_err("a negative does not fit an unsigned page");
14552        assert_eq!(
14553            fit::<i32>(&[i64::from(i32::MIN), 0, i64::from(i32::MAX)]).expect("the edges fit"),
14554            vec![i32::MIN, 0, i32::MAX]
14555        );
14556        fit::<i32>(&[i64::from(i32::MAX) + 1]).expect_err("one past the top does not fit");
14557        fit::<i32>(&[i64::from(i32::MIN) - 1]).expect_err("one past the bottom does not fit");
14558        assert_eq!(
14559            fit::<u32>(&[0, 4_294_967_295]).expect("the edges fit"),
14560            vec![0_u32, 4_294_967_295]
14561        );
14562        fit::<u32>(&[4_294_967_296]).expect_err("one past the top does not fit");
14563        fit::<u32>(&[-1]).expect_err("a negative does not fit an unsigned page");
14564
14565        // One value in a page that fits is still a page that does not, which is the thing an or
14566        // into an accumulator could get wrong in a way a page of one value would never show.
14567        fit::<i8>(&[0, 1, 2, 128, 3]).expect_err("one bad value spoils the page");
14568    }
14569
14570    /// The residue says yes to exactly what `TryFrom` says yes to.
14571    ///
14572    /// The edges above are the cases anyone would think to write down. This is the argument that
14573    /// there are no others, made by asking both questions about every value either narrow type could
14574    /// have an opinion about, and then about the values around the wide edges and the ends of an
14575    /// `i64`, which a range that size cannot reach.
14576    #[test]
14577    fn the_residue_agrees_with_a_checked_conversion_everywhere() {
14578        for value in -70_000_i64..70_000 {
14579            assert_eq!(fit::<i8>(&[value]).is_ok(), i8::try_from(value).is_ok(), "{value} as i8");
14580            assert_eq!(fit::<u8>(&[value]).is_ok(), u8::try_from(value).is_ok(), "{value} as u8");
14581            assert_eq!(fit::<i16>(&[value]).is_ok(), i16::try_from(value).is_ok(), "{value} i16");
14582            assert_eq!(fit::<u16>(&[value]).is_ok(), u16::try_from(value).is_ok(), "{value} u16");
14583        }
14584        let wide = [i64::MIN, i64::MIN + 1, i64::from(i32::MIN), 0, i64::from(u32::MAX), i64::MAX];
14585        for edge in wide {
14586            for step in -2_i64..=2 {
14587                let value = edge.saturating_add(step);
14588                assert_eq!(
14589                    fit::<i32>(&[value]).is_ok(),
14590                    i32::try_from(value).is_ok(),
14591                    "{value} as i32"
14592                );
14593                assert_eq!(
14594                    fit::<u32>(&[value]).is_ok(),
14595                    u32::try_from(value).is_ok(),
14596                    "{value} as u32"
14597                );
14598            }
14599        }
14600    }
14601
14602    /// All three block layouts come back as the same values in the same order.
14603    ///
14604    /// Blocks outside the page are what every file this build writes holds. Blocks that say where
14605    /// they are but sit inside the page behind the order are format 26, and blocks behind one
14606    /// another with only their ends recorded are older still. Nothing in the writer produces the
14607    /// last two any more, so the only way to find out whether the reader still understands those
14608    /// files is to write them here. The
14609    /// bytes go straight into a file with no directory around them, because what is under test is
14610    /// [`open_global_dictionary`], which is handed a page and a file and asks the directory for
14611    /// nothing.
14612    ///
14613    /// Three thousand values so that there are three payload blocks and a partial fourth, which is
14614    /// what makes the last block the one place where a length and an end disagree about what they
14615    /// are counting.
14616    #[test]
14617    fn a_dictionary_reads_the_same_whether_its_blocks_say_where_they_are() {
14618        let spellings = (0..3_000)
14619            .map(|index| format!("value {index:08} {}", "y".repeat(index % 40)))
14620            .collect::<Vec<_>>();
14621        let mut read = Vec::new();
14622        for layout in ["outside", "inside", "behind"] {
14623            let mut dictionary = GlobalDictionary::new();
14624            for text in &spellings {
14625                dictionary.code(text).expect("a code for every spelling");
14626            }
14627            dictionary.finish_blocks().expect("the last block encodes");
14628            let order = dictionary.ranked(None).expect("a sorted order");
14629            // Where the blocks go if they start at `from` and follow one another.
14630            let laid = |from: u64| {
14631                let mut at = from;
14632                dictionary
14633                    .blocks
14634                    .iter()
14635                    .map(|block| {
14636                        let place =
14637                            Placed { start: at, length: block.len() as u64, hash: checksum(block) };
14638                        at += block.len() as u64;
14639                        place
14640                    })
14641                    .collect::<Vec<_>>()
14642            };
14643            let payload = dictionary.blocks.concat();
14644            let scattered = layout != "behind";
14645            let (bytes, encoded, offset, length) = if layout == "outside" {
14646                let mut bytes = vec![0; HEADER as usize];
14647                bytes.extend_from_slice(&payload);
14648                let encoded = encode_global_dictionary(&dictionary, &order, &laid(HEADER), true)
14649                    .expect("an encoding");
14650                let offset = bytes.len() as u64;
14651                bytes.extend_from_slice(&encoded.index);
14652                bytes.extend_from_slice(&encoded.ranks);
14653                bytes.extend_from_slice(&encoded.grams);
14654                let length = encoded.index.len() + encoded.ranks.len() + encoded.grams.len();
14655                (bytes, encoded, offset, length)
14656            } else {
14657                // The index is the same length wherever the blocks are, so a first pass says where
14658                // the page ends and the second writes the places that follow it.
14659                let first = encode_global_dictionary(&dictionary, &order, &laid(0), scattered)
14660                    .expect("an encoding");
14661                let body = (first.index.len() + first.ranks.len() + first.grams.len()) as u64;
14662                let encoded = encode_global_dictionary(&dictionary, &order, &laid(body), scattered)
14663                    .expect("an encoding");
14664                let mut bytes = encoded.index.clone();
14665                bytes.extend_from_slice(&encoded.ranks);
14666                bytes.extend_from_slice(&encoded.grams);
14667                bytes.extend_from_slice(&payload);
14668                let length = bytes.len();
14669                (bytes, encoded, 0, length)
14670            };
14671            let path = path(&format!("blocks-{layout}"));
14672            fs::write(&path, &bytes).expect("the dictionary is written on its own");
14673            let file = Arc::new(File::open(&path).expect("it opens again"));
14674            let page = Page {
14675                offset,
14676                length: u32::try_from(length).expect("a test dictionary is small"),
14677                hash: checksum(&encoded.index),
14678            };
14679            let opened =
14680                open_global_dictionary(file, page, &LogicalType::Varchar, TEXT_KEEP_BUDGET)
14681                    .expect("a dictionary laid out either way opens");
14682            let mut swept: Vec<Vec<u8>> = Vec::new();
14683            let mut at = 0;
14684            while at < opened.len() {
14685                at = opened
14686                    .sweep_text(at, opened.len(), &mut |_index: usize, text: &[u8]| {
14687                        swept.push(text.to_vec());
14688                        Ok(())
14689                    })
14690                    .expect("a sweep reads");
14691            }
14692            fs::remove_file(&path).expect("clean up");
14693            read.push(swept);
14694        }
14695        let wanted =
14696            spellings.iter().map(|text| text.as_bytes().to_vec()).collect::<Vec<Vec<u8>>>();
14697        assert_eq!(read[0], wanted, "the blocks outside the page hold the values");
14698        assert_eq!(read[1], read[0], "the blocks inside the page hold the same values");
14699        assert_eq!(read[2], read[0], "the blocks behind one another hold the same values");
14700    }
14701
14702    /// A dictionary at its budget sweeps without keeping, and still answers what it answered.
14703    ///
14704    /// The budget is a quarter of a gigabyte in a running database, which is a fine size for a real
14705    /// column and no size at all for a test, so this opens the same dictionary a second time with a
14706    /// budget of zero. That is the shape of the hundred million row case: `URL` fills the budget
14707    /// somewhere in the middle of itself and everything past that point is read and dropped, which
14708    /// costs the decode again and holds none of it.
14709    #[test]
14710    fn a_dictionary_at_its_budget_sweeps_without_keeping() {
14711        let path = path("dictionary-budget");
14712        let spellings = (0..2_500)
14713            .map(|index| Value::Varchar(format!("value {index:08} {}", "y".repeat(index % 40))))
14714            .collect::<Vec<_>>();
14715        let mut writer =
14716            Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
14717                .expect("new file");
14718        for part in spellings.chunks(1_024) {
14719            writer
14720                .append(
14721                    &Chunk::new(vec![
14722                        Vector::from_values(LogicalType::Varchar, part).expect("strings"),
14723                    ])
14724                    .expect("one column"),
14725                )
14726                .expect("stripe written");
14727        }
14728        writer.finish().expect("commit");
14729
14730        let reader = Reader::open(&path).expect("valid directory");
14731        let page = reader.table.dictionaries[0].expect("a string column has one");
14732        let file = Arc::clone(&reader.file);
14733        let starved = open_global_dictionary(file, page, &LogicalType::Varchar, 0)
14734            .expect("a dictionary opens whatever it may keep");
14735
14736        let resting = starved.footprint();
14737        let mut swept: Vec<Vec<u8>> = Vec::new();
14738        let mut at = 0;
14739        while at < starved.len() {
14740            at = starved
14741                .sweep_text(at, starved.len(), &mut |_index: usize, text: &[u8]| {
14742                    swept.push(text.to_vec());
14743                    Ok(())
14744                })
14745                .expect("a sweep reads");
14746        }
14747        assert_eq!(swept.len(), spellings.len(), "a starved sweep still reads every value");
14748        assert_eq!(starved.footprint(), resting, "and keeps no block it decoded");
14749
14750        let generous = reader.dictionary(0).expect("read").expect("a string column has one");
14751        let read = (0..generous.len())
14752            .map(|code| generous.try_bytes_at(code).expect("read").expect("a value").to_vec())
14753            .collect::<Vec<_>>();
14754        assert_eq!(swept, read, "a starved sweep answers what a point read answers");
14755        fs::remove_file(path).expect("remove scratch file");
14756    }
14757
14758    #[test]
14759    fn damaged_membership_cannot_skip_a_string_page() {
14760        let path = path("damaged-membership");
14761        let mut writer = Writer::create(
14762            &path,
14763            "items",
14764            vec![
14765                Field::required("id", LogicalType::Integer),
14766                Field::new("text", LogicalType::Varchar),
14767            ],
14768        )
14769        .expect("new file");
14770        writer.append(&sample()).expect("stripe written");
14771        writer.finish().expect("commit");
14772
14773        let reader = Reader::open(&path).expect("valid directory");
14774        let membership = reader.table.stripes[0].memberships.get(1).expect("string membership");
14775        let mut file = OpenOptions::new().write(true).open(&path).expect("open membership page");
14776        file.seek(SeekFrom::Start(membership.offset)).expect("membership start");
14777        file.write_all(&[255]).expect("damage membership");
14778        let error = reader.skips_codes(0, 1, &[3]).expect_err("corruption must not skip rows");
14779        assert!(error.message().contains("membership page checksum differs"), "{error}");
14780        fs::remove_file(path).expect("remove scratch file");
14781    }
14782
14783    #[test]
14784    fn membership_delta_stream_is_sorted_exact_and_bounded() {
14785        let unique = unique_codes(&[900, 4, 4, 72, 9, u32::MAX]);
14786        assert_eq!(unique, [4, 9, 72, 900, u32::MAX]);
14787        let encoded = encode_membership(&unique);
14788        assert_eq!(
14789            decode_membership(&encoded).expect("valid membership"),
14790            [4, 9, 72, 900, u32::MAX]
14791        );
14792        // A stripe's index is the union of its parts', so a code in two of them is in it once and
14793        // the result is still one ascending run of deltas.
14794        let merged = merged_codes(vec![vec![4, 900], vec![9, 900, u32::MAX], vec![72]]);
14795        assert_eq!(merged, [4, 9, 72, 900, u32::MAX]);
14796        assert_eq!(
14797            decode_membership(&encode_membership(&merged)).expect("valid membership"),
14798            unique
14799        );
14800        assert!(decode_membership(&[1, 0x80]).is_err(), "a truncated varint is invalid");
14801        assert!(
14802            decode_membership(&[1, 0xff, 0xff, 0xff, 0xff, 0x10]).is_err(),
14803            "a value past u32 is invalid"
14804        );
14805    }
14806
14807    #[test]
14808    fn a_global_dictionary_may_be_larger_than_one_column_page() {
14809        let dictionary = Page {
14810            offset: HEADER,
14811            length: u32::try_from(MAX_PAGE + 1).expect("the page bound fits on disk"),
14812            hash: 0,
14813        };
14814        let table = Table {
14815            name: "items".to_owned(),
14816            fields: vec![Field::new("text", LogicalType::Varchar)],
14817            stripes: Vec::new(),
14818            rows: 0,
14819            dictionaries: vec![Some(dictionary)],
14820            dictionary_payloads: Vec::new(),
14821            distincts: vec![None],
14822            frequencies: vec![None],
14823            pair_frequencies: Vec::new(),
14824            frequency_texts: Vec::new(),
14825            host_groups: None,
14826            clustering: None,
14827            generation: 1,
14828            sections: Vec::new(),
14829        };
14830        let directory = encode_directory(&table).expect("directory");
14831        let file_size = dictionary.offset + u64::from(dictionary.length) + 1;
14832
14833        let decoded = decode_directory(&directory, file_size).expect("large lazy dictionary");
14834        assert_eq!(decoded.dictionaries[0].expect("dictionary").length, dictionary.length);
14835    }
14836
14837    #[test]
14838    fn a_column_with_one_value_everywhere_costs_almost_nothing_a_row() {
14839        let path = path("constant-codes");
14840        let mut writer =
14841            Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
14842                .expect("new file");
14843        let empty = vec![Value::Varchar(String::new()); 1024];
14844        for _ in 0..4 {
14845            let column = Vector::from_values(LogicalType::Varchar, &empty).expect("strings");
14846            writer.append(&Chunk::new(vec![column]).expect("one column")).expect("a part");
14847        }
14848        writer.finish().expect("commit");
14849
14850        let reader = Reader::open(&path).expect("valid directory");
14851        let pages = reader.layout().columns.first().expect("one column").pages;
14852        // This column used to cost four bytes a row, 16,384 of them, the same as a column of four
14853        // thousand distinct URLs would. The cascade calls each part a constant, so what is left is
14854        // a tag, a count and the value, and the row count stops being what drives the number.
14855        assert!(pages < 256, "{pages} bytes of pages for 4,096 rows of one value");
14856        let read = reader.read(3, &[0]).expect("the last part back");
14857        assert_eq!(read.value_at(0, 0), Value::Varchar(String::new()));
14858        assert_eq!(read.value_at(1023, 0), Value::Varchar(String::new()));
14859        fs::remove_file(path).expect("remove scratch file");
14860    }
14861
14862    #[test]
14863    fn a_cascade_value_too_wide_for_its_column_is_refused_rather_than_cut() {
14864        // What a damaged page looks like from here: the cascade decoded, so the bytes are not
14865        // truncated, but the values do not belong to the column the directory says they do.
14866        let over = vec![i64::from(i32::MAX) + 1];
14867        let error = narrowed(&LogicalType::Integer, over).expect_err("a page that disagrees");
14868        assert!(format!("{error}").contains("not of its type"), "{error}");
14869        assert!(narrowed(&LogicalType::BigInt, vec![i64::MIN]).is_ok(), "bigint holds all of i64");
14870        assert!(narrowed(&LogicalType::Varchar, vec![0]).is_err(), "strings are not integers");
14871    }
14872
14873    #[test]
14874    fn a_code_stream_the_cascade_cannot_shrink_is_left_alone() {
14875        // A shift register rather than a run, because an arithmetic run is the one wide shape the
14876        // cascade does shrink. This is what a column with tens of millions of distinct values hands
14877        // over: full width codes with no order to them.
14878        let mut state: u32 = 0x9e37_79b9;
14879        let spread: Vec<u32> = (0..1024)
14880            .map(|_| {
14881                state ^= state << 13;
14882                state ^= state >> 17;
14883                state ^= state << 5;
14884                state
14885            })
14886            .collect();
14887        assert_eq!(encoded_codes(&spread).expect("no failure"), None);
14888        let near: Vec<u32> = (0..1024).collect();
14889        let coded = encoded_codes(&near).expect("no failure").expect("counting up is packable");
14890        assert!(coded.len() < near.len() * 4, "{} bytes for a run of 1,024", coded.len());
14891    }
14892
14893    /// The columns of a stripe are encoded on whichever thread got to them, so the one thing that
14894    /// must not depend on which thread that was is the file. Two writes of the same rows are
14895    /// compared byte for byte rather than value for value, because a dictionary that two columns
14896    /// somehow shared would still read back correctly and would hand out its codes in the order the
14897    /// threads happened to run in, which is exactly what this is here to catch.
14898    #[test]
14899    fn two_writes_of_the_same_rows_give_the_same_bytes() {
14900        fn written(path: &PathBuf) {
14901            let fields = (0..40)
14902                .map(|column| {
14903                    let ty =
14904                        if column % 4 == 0 { LogicalType::Varchar } else { LogicalType::BigInt };
14905                    Field::new(format!("c{column}"), ty)
14906                })
14907                .collect::<Vec<_>>();
14908            let mut writer = Writer::create(path, "wide", fields).expect("new file");
14909            for part in 0..70_u64 {
14910                let columns = (0..40)
14911                    .map(|column| {
14912                        let values = (0..64_u64)
14913                            .map(|row| {
14914                                let seed = part.wrapping_mul(31).wrapping_add(row);
14915                                if column % 4 == 0 {
14916                                    Value::Varchar(format!("v{}", seed % 17))
14917                                } else {
14918                                    Value::BigInt(i64::try_from(seed % 97).expect("small"))
14919                                }
14920                            })
14921                            .collect::<Vec<_>>();
14922                        let ty = if column % 4 == 0 {
14923                            LogicalType::Varchar
14924                        } else {
14925                            LogicalType::BigInt
14926                        };
14927                        Vector::from_values(ty, &values).expect("a column")
14928                    })
14929                    .collect::<Vec<_>>();
14930                writer.append(&Chunk::new(columns).expect("forty columns")).expect("a part");
14931            }
14932            writer.finish().expect("commit");
14933        }
14934
14935        let first = path("repeatable-one");
14936        let second = path("repeatable-two");
14937        written(&first);
14938        written(&second);
14939        let left = fs::read(&first).expect("the first file");
14940        let right = fs::read(&second).expect("the second file");
14941        assert_eq!(left.len(), right.len(), "two writes of the same rows differ in length");
14942        assert!(left == right, "two writes of the same rows differ in their bytes");
14943
14944        // And the rows are still there, since a pair of identically wrong files would pass the
14945        // comparison above on its own.
14946        let reader = Reader::open(&first).expect("valid directory");
14947        assert_eq!(reader.table().rows(), 70 * 64);
14948        let read = reader.read(0, &[0, 1]).expect("the first part back");
14949        assert_eq!(read.value_at(0, 0), Value::Varchar("v0".to_owned()));
14950        assert_eq!(read.value_at(0, 1), Value::BigInt(0));
14951        fs::remove_file(first).expect("remove scratch file");
14952        fs::remove_file(second).expect("remove scratch file");
14953    }
14954
14955    /// Three tables of different shapes in one file, read back by name.
14956    fn three_tables(path: &PathBuf) {
14957        let writer = Writer::create(
14958            path,
14959            "region",
14960            vec![
14961                Field::new("r_key", LogicalType::Integer),
14962                Field::new("r_name", LogicalType::Varchar),
14963            ],
14964        )
14965        .expect("new file");
14966        let mut writer = writer;
14967        writer
14968            .append(
14969                &Chunk::new(vec![
14970                    Vector::from_values(
14971                        LogicalType::Integer,
14972                        &[Value::Integer(0), Value::Integer(1)],
14973                    )
14974                    .expect("keys"),
14975                    Vector::from_values(
14976                        LogicalType::Varchar,
14977                        &[Value::Varchar("AFRICA".to_owned()), Value::Varchar("ASIA".to_owned())],
14978                    )
14979                    .expect("names"),
14980                ])
14981                .expect("two columns"),
14982            )
14983            .expect("a part");
14984        let mut writer = writer
14985            .next("empty", vec![Field::new("nothing", LogicalType::BigInt)])
14986            .expect("a second table");
14987        writer
14988            .append(
14989                &Chunk::new(vec![
14990                    Vector::from_values(LogicalType::BigInt, &[Value::BigInt(7)]).expect("a row"),
14991                ])
14992                .expect("one column"),
14993            )
14994            .expect("a part");
14995        let mut writer =
14996            writer.next("wide", vec![Field::new("n", LogicalType::BigInt)]).expect("a third table");
14997        for part in 0..70_i64 {
14998            let values = (0..64).map(|row| Value::BigInt(part * 64 + row)).collect::<Vec<_>>();
14999            writer
15000                .append(
15001                    &Chunk::new(vec![
15002                        Vector::from_values(LogicalType::BigInt, &values).expect("a column"),
15003                    ])
15004                    .expect("one column"),
15005                )
15006                .expect("a part");
15007        }
15008        writer.finish().expect("commit");
15009    }
15010
15011    #[test]
15012    fn three_tables_in_one_file_read_back_by_name() {
15013        let file = path("three-tables");
15014        three_tables(&file);
15015        let catalog = Catalog::open(&file).expect("a committed catalog");
15016        assert_eq!(catalog.names().collect::<Vec<_>>(), ["region", "empty", "wide"]);
15017
15018        let region = catalog.table("region").expect("the first table");
15019        assert_eq!(region.table().rows(), 2);
15020        assert_eq!(
15021            region.read(0, &[1]).expect("names").value_at(1, 0),
15022            Value::Varchar("ASIA".to_owned())
15023        );
15024
15025        let wide = catalog.table("wide").expect("the third table");
15026        assert_eq!(wide.table().rows(), 70 * 64);
15027        assert_eq!(wide.read(0, &[0]).expect("the first part").value_at(0, 0), Value::BigInt(0));
15028
15029        // The middle table is reached without the one after it having been touched, which is what
15030        // a directory per table buys over one directory of everything.
15031        let empty = catalog.table("empty").expect("the second table");
15032        assert_eq!(empty.table().rows(), 1);
15033        assert_eq!(empty.read(0, &[0]).expect("the row").value_at(0, 0), Value::BigInt(7));
15034
15035        fs::remove_file(file).expect("remove scratch file");
15036    }
15037
15038    #[test]
15039    fn a_name_the_file_does_not_hold_is_an_error_rather_than_the_first_table() {
15040        let file = path("three-tables-missing");
15041        three_tables(&file);
15042        let catalog = Catalog::open(&file).expect("a committed catalog");
15043        let error = catalog.table("nation").expect_err("no such table");
15044        assert!(error.message().contains("nation"), "{}", error.message());
15045        fs::remove_file(file).expect("remove scratch file");
15046    }
15047
15048    #[test]
15049    fn a_file_of_three_tables_will_not_open_as_one() {
15050        let file = path("three-tables-unnamed");
15051        three_tables(&file);
15052        let error = Reader::open(&file).expect_err("more than one table");
15053        assert!(error.message().contains("more than one table"), "{}", error.message());
15054        fs::remove_file(file).expect("remove scratch file");
15055    }
15056
15057    /// One column per storage width, because the width is what decides how many bytes a row costs.
15058    #[test]
15059    fn decimals_of_every_storage_width_round_trip() {
15060        let file = path("decimals");
15061        let widths = [(4_u8, 2_u8), (9, 2), (18, 4), (38, 6)];
15062        let fields = widths
15063            .iter()
15064            .enumerate()
15065            .map(|(index, (width, scale))| {
15066                Field::new(
15067                    format!("d{index}"),
15068                    LogicalType::decimal(*width, *scale).expect("a decimal type"),
15069                )
15070            })
15071            .collect::<Vec<_>>();
15072        let mut writer = Writer::create(&file, "money", fields).expect("new file");
15073        let rows: [i128; 3] = [-1234, 0, 999];
15074        let columns = widths
15075            .iter()
15076            .map(|(width, scale)| {
15077                let values = rows
15078                    .iter()
15079                    .map(|unscaled| Value::Decimal {
15080                        unscaled: *unscaled,
15081                        width: *width,
15082                        scale: *scale,
15083                    })
15084                    .collect::<Vec<_>>();
15085                Vector::from_values(
15086                    LogicalType::decimal(*width, *scale).expect("a decimal type"),
15087                    &values,
15088                )
15089                .expect("a decimal column")
15090            })
15091            .collect::<Vec<_>>();
15092        writer.append(&Chunk::new(columns).expect("four columns")).expect("a part");
15093        writer.finish().expect("commit");
15094
15095        let reader = Reader::open(&file).expect("a committed file");
15096        for (index, (width, scale)) in widths.iter().enumerate() {
15097            assert_eq!(
15098                reader.table().fields()[index].ty,
15099                LogicalType::decimal(*width, *scale).expect("a decimal type"),
15100                "column {index} came back as another type"
15101            );
15102            let column = reader.read(0, &[index]).expect("the column");
15103            for (row, unscaled) in rows.iter().enumerate() {
15104                assert_eq!(
15105                    column.value_at(row, 0),
15106                    Value::Decimal { unscaled: *unscaled, width: *width, scale: *scale },
15107                    "column {index} row {row}"
15108                );
15109            }
15110        }
15111        fs::remove_file(file).expect("remove scratch file");
15112    }
15113
15114    #[test]
15115    fn two_tables_of_one_name_are_refused_before_anything_is_committed() {
15116        let file = path("two-of-a-name");
15117        let writer = Writer::create(&file, "t", vec![Field::new("a", LogicalType::BigInt)])
15118            .expect("new file");
15119        let error = writer
15120            .next("t", vec![Field::new("a", LogicalType::BigInt)])
15121            .expect_err("the same name twice");
15122        assert!(error.message().contains("same name"), "{}", error.message());
15123        fs::remove_file(file).expect("remove scratch file");
15124    }
15125
15126    #[test]
15127    fn opening_the_catalog_reads_no_table_directory() {
15128        let file = path("catalog-only");
15129        three_tables(&file);
15130        let catalog = Catalog::open(&file).expect("a committed catalog");
15131        // The header and one slot, and nothing under it. The third table's directory covers seventy
15132        // stripes and reading it here would be the whole point of the two levels thrown away.
15133        assert_eq!(catalog.opening.reads, 2, "opening the catalog read more than the slot");
15134        assert_eq!(catalog.names().len(), 3);
15135        fs::remove_file(file).expect("remove scratch file");
15136    }
15137
15138    /// The checksum answers what it has always answered, at every length its branches split on.
15139    ///
15140    /// This is a compatibility test rather than a correctness one. Nothing about the hash has to be
15141    /// any particular function, but a file already on disk carries the answers the version that
15142    /// wrote it gave, so a change here is a change that makes every stored file fail to verify. The
15143    /// lengths are the ones the code makes decisions about: nothing, under a block, a block exactly,
15144    /// a block and a word, a word and a half word, and a half word and a byte.
15145    ///
15146    /// The empty answer is the published xxHash64 vector for an empty input at seed zero, which is
15147    /// also a check that this is the function it says it is.
15148    #[test]
15149    fn the_checksum_answers_what_it_has_always_answered() {
15150        let bytes: Vec<u8> =
15151            (0..1000_u32).map(|at| (at.wrapping_mul(31).wrapping_add(7) % 251) as u8).collect();
15152        for (length, expected) in [
15153            (0, 0xef46_db37_51d8_e999),
15154            (1, 0xa96c_7f0c_e858_bbb7),
15155            (3, 0x56e6_9576_32a4_87f9),
15156            (4, 0xc60d_15b1_e3ff_8f04),
15157            (5, 0x8088_1585_8624_dd4e),
15158            (7, 0xafbe_fc3d_6c6f_9a8e),
15159            (8, 0x3da5_c7aa_2696_83e0),
15160            (9, 0x465e_c429_b13c_3892),
15161            (15, 0xdee8_9d8a_065a_6233),
15162            (16, 0x1330_489a_7767_9c80),
15163            (31, 0x3391_303d_485e_846e),
15164            (32, 0x40b7_aff7_5d45_bbc8),
15165            (33, 0x4997_cae4_951c_17a5),
15166            (39, 0x5807_28fd_5c14_5739),
15167            (40, 0xf95c_f6f5_c08a_3d3b),
15168            (63, 0x2944_b4da_fc69_b206),
15169            (64, 0xbb76_f6ef_19bd_5a1b),
15170            (65, 0x814e_0c65_4a9f_d640),
15171            (127, 0x00de_aab1_31cf_f89b),
15172            (1000, 0x9e33_00c1_cde3_c58d),
15173        ] {
15174            assert_eq!(checksum(&bytes[..length]), expected, "the checksum of {length} bytes");
15175        }
15176        assert_eq!(checksum(b"the quick brown fox jumps over the lazy dog"), 0xed71_4233_c5a9_a792);
15177    }
15178    /// A declared order survives the file, and a table that declared none stays as it was.
15179    ///
15180    /// The second half is the one worth a test. The clustering section is written only when there
15181    /// is a declaration, so a file of two tables where one is clustered exercises both the present
15182    /// and the absent branch of the decoder in one directory, which is where a length bug would
15183    /// show up as one table reading the other's bytes.
15184    #[test]
15185    fn a_declared_order_comes_back_out_of_the_file() {
15186        let path = path("clustered");
15187        let shipped = vec![
15188            Field::new("key", LogicalType::BigInt),
15189            Field::new("line", LogicalType::Integer),
15190            Field::new("shipdate", LogicalType::Date),
15191        ];
15192        let plain = vec![Field::new("a", LogicalType::Integer)];
15193        let stage_zero = Clustering::new(vec![2, 0, 1], Width::Month, &shipped).expect("valid");
15194
15195        let mut writer = Writer::create(&path, "lineitem", shipped)
15196            .expect("new file")
15197            .declare(stage_zero.clone())
15198            .expect("the columns are the table's");
15199        let column = |ty: LogicalType, values: &[Value]| {
15200            Vector::from_values(ty, values).expect("the values match the type")
15201        };
15202        writer
15203            .append(
15204                &Chunk::new(vec![
15205                    column(
15206                        LogicalType::BigInt,
15207                        &[Value::BigInt(0), Value::BigInt(1), Value::BigInt(2), Value::BigInt(3)],
15208                    ),
15209                    column(
15210                        LogicalType::Integer,
15211                        &[
15212                            Value::Integer(1),
15213                            Value::Integer(1),
15214                            Value::Integer(1),
15215                            Value::Integer(1),
15216                        ],
15217                    ),
15218                    column(
15219                        LogicalType::Date,
15220                        &[Value::Date(0), Value::Date(1), Value::Date(2), Value::Date(3)],
15221                    ),
15222                ])
15223                .expect("three columns"),
15224            )
15225            .expect("four rows");
15226        let mut writer = writer.next("nation", plain).expect("a second table");
15227        writer
15228            .append(
15229                &Chunk::new(vec![column(LogicalType::Integer, &[Value::Integer(7)])])
15230                    .expect("one column"),
15231            )
15232            .expect("one row");
15233        writer.finish().expect("commit");
15234
15235        let catalog = Catalog::open(&path).expect("reopen");
15236        let lineitem = catalog.table("lineitem").expect("the clustered table");
15237        assert_eq!(lineitem.table().clustering(), Some(&stage_zero));
15238        let nation = catalog.table("nation").expect("the plain table");
15239        assert_eq!(nation.table().clustering(), None, "nobody declared one here");
15240
15241        // And the rows are still the rows, because the section goes on the end of the directory
15242        // and the easy way to break that is to leave the cursor somewhere the next read trusts.
15243        assert_eq!(lineitem.table().rows(), 4);
15244        assert_eq!(nation.table().rows(), 1);
15245        fs::remove_file(&path).ok();
15246    }
15247
15248    /// A declaration naming a column the table does not have is refused where it is made.
15249    #[test]
15250    fn a_declaration_off_the_end_of_the_table_never_reaches_the_file() {
15251        let path = path("clustered-bad");
15252        let writer = Writer::create(&path, "items", vec![Field::new("a", LogicalType::Integer)])
15253            .expect("new file");
15254        let four =
15255            (0..4).map(|at| Field::new(format!("c{at}"), LogicalType::Integer)).collect::<Vec<_>>();
15256        let wrong = Clustering::new(vec![3], Width::Exact, &four).expect("valid against four");
15257        assert!(writer.declare(wrong).is_err(), "the table has one column, not four");
15258        fs::remove_file(&path).ok();
15259    }
15260
15261    /// The sorted order is the byte order, whatever the values do before they differ.
15262    ///
15263    /// A block handed out of the writer's lock to be encoded, and given back in whatever order the
15264    /// stripes happen to finish in, is the same block with the same signature as one encoded in
15265    /// place, and lands in the same position.
15266    #[test]
15267    fn blocks_handed_out_and_given_back_out_of_order_are_the_blocks_encoded_in_place() {
15268        let values = (0..PAYLOAD_SAMPLE_BLOCKS * TEXT_PAYLOAD_VALUES * 2 + 100)
15269            .map(|at| format!("http://example{}.test/page/{at:06}", at % 7))
15270            .collect::<Vec<_>>();
15271        let filled = || {
15272            let mut dictionary = GlobalDictionary::new();
15273            for value in &values {
15274                dictionary.code(value).expect("a code for every value");
15275            }
15276            dictionary.settle().expect("a shape");
15277            dictionary
15278        };
15279        let mut in_place = filled();
15280        in_place.finish_blocks().expect("every block encodes");
15281
15282        let mut handed = filled();
15283        let out = handed.hand_out(3);
15284        assert_eq!(out.len(), PAYLOAD_SAMPLE_BLOCKS * 2, "every sealed block goes out");
15285        assert!(handed.waiting.is_empty(), "and none is left to be encoded under the lock");
15286        for job in out.iter().rev() {
15287            assert_eq!(job.place().0, 3, "a block goes back to the column it came from");
15288            handed.take_back(job.place().1, job.encode().expect("encodes")).expect("taken back");
15289        }
15290        assert!(handed.early.is_empty(), "nothing is waiting on a gap");
15291        handed.finish_blocks().expect("the last block encodes");
15292
15293        assert_eq!(handed.blocks, in_place.blocks, "the same blocks in the same order");
15294        assert_eq!(handed.grams, in_place.grams, "with the same signatures");
15295    }
15296
15297    /// A block given back twice is a bug in whoever gave it, and is said rather than written twice.
15298    #[test]
15299    fn a_block_given_back_twice_is_refused() {
15300        let mut dictionary = GlobalDictionary::new();
15301        for at in 0..PAYLOAD_SAMPLE_BLOCKS * TEXT_PAYLOAD_VALUES {
15302            dictionary.code(&format!("value {at}")).expect("a code");
15303        }
15304        dictionary.settle().expect("a shape");
15305        let out = dictionary.hand_out(0);
15306        let last = out.last().expect("blocks went out");
15307        let at = last.place().1;
15308        dictionary.take_back(at, last.encode().expect("encodes")).expect("taken back once");
15309        assert!(dictionary.take_back(at, last.encode().expect("encodes")).is_err());
15310    }
15311
15312    /// The values here are the shape the sort is built for and the shape a comparison sort is worst
15313    /// at: a common scheme, a handful of hosts, and a path that only decides the pair thirty bytes
15314    /// in. They also cover what the bucketing has to get right at the edges, which is a value that
15315    /// has run out where another carries on, the empty value, and enough entries to take the range
15316    /// down through several passes and out the bottom into the comparison that finishes it.
15317    #[test]
15318    fn the_dictionary_order_is_the_byte_order_however_deep_the_values_agree() {
15319        let mut values = vec![String::new(), "http://".to_owned()];
15320        for host in 0..7 {
15321            for path in 0..30 {
15322                values.push(format!("http://example{host}.test/page/{path:04}/index.html"));
15323                values.push(format!("http://example{host}.test/page/{path:04}"));
15324            }
15325        }
15326        values.push("http://example0.test/page/0000/index.htmlx".to_owned());
15327
15328        let mut dictionary = GlobalDictionary::new();
15329        for value in &values {
15330            dictionary.code(value).expect("a code for every value");
15331        }
15332        dictionary.finish_blocks().expect("the last block encodes");
15333        let ranked = dictionary.ranked(None).expect("a sorted order");
15334        assert_eq!(ranked.len(), values.len(), "one entry a distinct value");
15335
15336        let spellings = dictionary_values(&dictionary);
15337        let seen = ranked
15338            .iter()
15339            .map(|&(_, code)| {
15340                String::from_utf8(spellings[code as usize].clone()).expect("text in, text out")
15341            })
15342            .collect::<Vec<_>>();
15343        let mut wanted = values.clone();
15344        wanted.sort_unstable();
15345        assert_eq!(seen, wanted, "the order is the order the bytes give");
15346
15347        for &(carried, code) in &ranked {
15348            let value = &spellings[code as usize];
15349            assert_eq!(carried, head(value), "the head belongs to the value it is filed with");
15350        }
15351    }
15352
15353    /// Picking the commonest entries leaves exactly what sorting all of them and cutting left.
15354    ///
15355    /// The counts here are deliberately full of ties, including a tie that straddles the cut, which
15356    /// is where a partition and a sort can disagree if the comparison they are given is not total.
15357    #[test]
15358    fn the_commonest_entries_are_the_ones_a_full_sort_would_have_kept() {
15359        let entry =
15360            |value: u32, count: u64| FrequencyEntry { value: FrequencyValue::Code(value), count };
15361        let mut all = (0..FREQUENCY_ENTRIES as u32 * 3)
15362            .map(|code| entry(code, u64::from(code % 7) + 1))
15363            .collect::<Vec<_>>();
15364        all.push(FrequencyEntry { value: FrequencyValue::Null, count: 4 });
15365
15366        let mut sorted = all.clone();
15367        sorted.sort_unstable_by(|left, right| {
15368            right.count.cmp(&left.count).then_with(|| frequency_order(left.value, right.value))
15369        });
15370        let wanted_omitted = sorted[FREQUENCY_ENTRIES].count;
15371        sorted.truncate(FREQUENCY_ENTRIES);
15372
15373        let mut picked = all.clone();
15374        let omitted = keep_most_frequent(&mut picked);
15375        assert_eq!(omitted, wanted_omitted, "the largest count that did not make the cut");
15376        assert_eq!(picked.len(), FREQUENCY_ENTRIES, "the cut is where it says it is");
15377        assert!(
15378            picked
15379                .iter()
15380                .zip(&sorted)
15381                .all(|(one, two)| one.value == two.value && one.count == two.count),
15382            "the same entries in the same order"
15383        );
15384
15385        let mut short = all[..FREQUENCY_ENTRIES - 1].to_vec();
15386        let omitted = keep_most_frequent(&mut short);
15387        assert_eq!(omitted, 0, "nothing is omitted when everything fits");
15388        assert!(short.windows(2).all(|pair| pair[0].count >= pair[1].count), "still in order");
15389    }
15390
15391    /// A dictionary too small to bucket, and one with nothing in it, come back in order too.
15392    #[test]
15393    fn a_short_dictionary_sorts_without_a_bucketing_pass() {
15394        let empty = GlobalDictionary::new();
15395        assert!(empty.ranked(None).expect("an empty order").is_empty(), "nothing in, nothing out");
15396
15397        let mut dictionary = GlobalDictionary::new();
15398        for value in ["pear", "apple", "", "apples", "app"] {
15399            dictionary.code(value).expect("a code for every value");
15400        }
15401        dictionary.finish_blocks().expect("the one block encodes");
15402        let spellings = dictionary_values(&dictionary);
15403        let seen = dictionary
15404            .ranked(None)
15405            .expect("a sorted order")
15406            .iter()
15407            .map(|&(_, code)| spellings[code as usize].clone())
15408            .collect::<Vec<_>>();
15409        let wanted: Vec<Vec<u8>> =
15410            [&b""[..], b"app", b"apple", b"apples", b"pear"].iter().map(|v| v.to_vec()).collect();
15411        assert_eq!(seen, wanted, "shorter first where one runs out inside another");
15412    }
15413}