Skip to main content

rudb_native/
lib.rs

1//! Rudb's single-file columnar snapshot format.
2//!
3//! A committed directory names independently readable column pages. It has two levels: a catalog
4//! directory naming every table in the file, which is what a footer slot points at and what opening
5//! a database reads, and one directory per table under it holding that table's stripes, pages and
6//! statistics. One slot write publishes all of them, so a commit is atomic across tables.
7//!
8//! This version handles scalar columns; the file header has two generation slots so an unfinished
9//! replacement directory cannot hide the last complete one. See
10//! `spec/storage-v3/12-many-tables-in-one-file.md`.
11//!
12//! # Parts and stripes
13//!
14//! A part is one appended chunk, which is a thousand rows, and it is the unit a scan decodes and
15//! hands to the pipeline. A stripe is sixty four parts, and it is the unit the directory describes
16//! and the unit the file is laid out in: one page per column per stripe, holding that column's
17//! sixty four part payloads end to end.
18//!
19//! The two are separate because they are sized by different pressures. A part wants to be small
20//! because it is a vector and vectors live in cache. A stripe wants to be large because everything
21//! the directory holds is per stripe and the directory is one buffer that has to be read and
22//! decoded before a single row can be answered. A hundred million rows of the hundred and five
23//! column ClickBench table is ninety seven thousand parts, and a directory with a page entry and a
24//! pair of bounds per part per column is several hundred megabytes, which is what made that load
25//! fail before this split existed. Sixty four parts to a stripe divides that by sixty four.
26//!
27//! Where the parts of a page start is not in the directory either, for the same reason. Each
28//! stripe writes one index page holding a length and a checksum per part per column, and a reader
29//! preads the sixty four entries belonging to the column it wants. A scan reads the whole column
30//! page once and slices it; a sparse row fetch reads the index entries and then only the part it
31//! needs.
32
33#![forbid(unsafe_code)]
34
35use std::borrow::Cow;
36use std::cmp::Ordering;
37use std::collections::{HashMap, VecDeque};
38use std::fs::{File, OpenOptions};
39use std::io::{Read, Seek, SeekFrom};
40use std::mem::{size_of, size_of_val};
41use std::path::Path;
42use std::slice;
43use std::sync::atomic::{AtomicBool, AtomicUsize, Ordering as Atomic};
44use std::sync::{Arc, Mutex, OnceLock, Weak};
45
46use rudb_common::bounds::{self, Bound, Op, scaled_as};
47use rudb_common::{Clustering, Error, Field, LogicalType, PhysicalType, Result, Value, Width};
48use rudb_encoding::{bitpack, chooser, integer, string};
49use rudb_metrics::{LoadProfile, Stage};
50use rudb_storage::sieve::Sieve;
51use rudb_storage::{Probe, Range, Zone};
52use rudb_vector::string::StringColumn;
53use rudb_vector::validity::Validity;
54use rudb_vector::{Buffer, Chunk, Data, Packed, TextSource, Vector, search_below};
55
56mod distinct;
57pub mod graph;
58pub mod host;
59mod prepare;
60pub mod section;
61pub mod stats;
62mod zones;
63
64pub use prepare::{Merged, Paged, Prepared, Preparer};
65pub use section::Section;
66pub use zones::{Common, Stripes, ascending, distincts};
67
68const MAGIC: &[u8; 8] = b"RUDBNV10";
69const DIRECTORY: &[u8; 8] = b"RUDBDI10";
70const CATALOG: &[u8; 8] = b"RUDBCA10";
71const NONZERO_COUNTS: &[u8; 8] = b"RUDBNZ10";
72const AGGREGATE_SUMS: &[u8; 8] = b"RUDBAG10";
73const FORMAT: u32 = 28;
74
75/// Formats this build can open.
76///
77/// More than one, for the first time, and the reason is spec/graph/10-milestones.md's G1 exit
78/// criterion: a build with the section table in it has to open a file written before the section
79/// table existed, unchanged and without a rewrite. Formats 22 and 23 are those files, and both read
80/// as a table with an empty section table, which is exactly what section 3.1 says a table with no
81/// graph sections is.
82///
83/// All three of the older ones are readable for the same reason. What took the format from 22 to 23
84/// was tags for fourteen more column types, and a file written before that has none of them in it,
85/// so nothing in an older file is a tag this build cannot read. What took it from 23 to 24 is the
86/// section table, which a file written before it simply does not have. What takes it from 24 to 25
87/// is the view section on the end of the catalog, which an older file does not have either, and a
88/// catalog that ends where the tables end reads as a catalog with no views in it. What takes it
89/// from 25 to 26 is that a global dictionary's payload blocks now say where they are, and a file
90/// written before that has them behind one another, which [`open_global_dictionary`] reads by
91/// turning the ends it finds into the same places the newer files name outright. What takes it
92/// from 26 to 27 is that those blocks are written into the file as the load goes, between the
93/// stripes, rather than behind the dictionary's index at the end, so the dictionary's page is the
94/// index and the sorted order and nothing else. A format 26 file has its blocks inside the page,
95/// and the reader tells the two apart by whether the page has room left over for them.
96///
97/// Format 28 adds per-payload-block substring signatures to global string dictionaries. Older
98/// files have no signatures and use the ordinary exact string filter.
99///
100/// This is not a general compatibility promise. Seven formats are readable because there was a
101/// specific reason for each, and the list shrinks again the moment the older ones stop being worth
102/// carrying.
103const READABLE: &[u32] = &[22, 23, 24, 25, 26, 27, FORMAT];
104
105const HEADER: u64 = 80;
106const SLOT_BYTES: usize = 28;
107const MAX_PAGE: usize = 256 * 1024 * 1024;
108const MAX_DIRECTORY: usize = 128 * 1024 * 1024;
109const FREQUENCIES_V2: &[u8; 8] = b"RUDBFQ2\0";
110const FREQUENCIES: &[u8; 8] = b"RUDBFQ3\0";
111/// Inline spellings for string entries in the bounded frequency synopsis.
112///
113/// A planner usually asks about one literal such as the empty string. Without this block it opens
114/// a multi-million-value global dictionary and visits the payload blocks of every retained entry
115/// merely to compare that literal with at most 512 heavy hitters. The spellings are already in
116/// memory while the writer sorts the dictionary, so storing this bounded copy makes planning a
117/// directory read and leaves the dictionary unopened.
118const FREQUENCY_TEXTS: &[u8; 8] = b"RUDBFT1\0";
119/// Certified host aggregate state for the version-one anchored replacement expression.
120const HOST_GROUPS: &[u8; 8] = b"RUDBHG1\0";
121/// Exact leading counts for a bounded pair of dictionary-backed grouping keys.
122///
123/// This is a separate optional directory block rather than another frequency format. Readers that
124/// predate it still understand every earlier directory, and a table without a pair worth keeping
125/// writes no block at all.
126const PAIR_FREQUENCIES: &[u8; 8] = b"RUDBPF1\0";
127/// The clustering declaration, written after the frequencies and only when there is one.
128///
129/// No format bump for this, which is the convention the frequency section set in #728: a new
130/// optional trailing section with its own magic leaves every file that does not use it byte for
131/// byte what it was, and the version is bumped for a change to a layout that already exists, as
132/// #1029 did. A file with no declaration is the same bytes this build wrote yesterday.
133///
134/// The width byte in this block gained a fifth value for #1285, for a declaration that leaves the
135/// bucket to the row count, and that did not bump the format either. It is the one case where the
136/// reasoning needs saying out loud, because it is a new value in a layout that already exists
137/// rather than a new section. A build without it reading one of these says `clustering width
138/// tag differs` and refuses the table, which is what that message was written for. Bumping the
139/// format instead would have made every file this build writes unreadable to an older one, whether
140/// it has a declaration in it or not, to warn about a case that only arises when it does.
141const CLUSTERING: &[u8; 8] = b"RUDBCL1\0";
142/// The graph section table, written after the clustering declaration and written even when empty.
143///
144/// Same convention and the same reason as the block above it, with one difference: this one is
145/// always there, so a file written by this build says which sections it has rather than leaving a
146/// reader to infer it from where the bytes ran out. Section 3.1 of the graph spec is what makes
147/// that safe to add without a format bump, because a table with no sections answers every query
148/// the way it did before, only without the graph path.
149const SECTIONS: &[u8; 8] = b"RUDBSE1\0";
150/// How many bytes of each column's global dictionary live outside its page, written only when any do.
151///
152/// From format 27 a dictionary's payload blocks are written into the file while the load runs, so
153/// they sit between the stripes and the dictionary's page covers only its index and sorted order.
154/// Nothing needs the total to read the file, because the index names every block. It is here for
155/// what a file costs a column, which [`Reader::layout`] and the statistics budget both report, and
156/// which would otherwise lose most of the bytes of every large string column.
157const DICTIONARY_PAYLOADS: &[u8; 8] = b"RUDBDP1\0";
158
159/// The most sections one table's directory may name.
160///
161/// A relationship contributes at most three sections, so this bounds a table at a few thousand
162/// relationships, which is far past anything a schema has. The bound is here so that a torn
163/// directory naming four billion of them is refused at decode rather than turned into an
164/// allocation, the same reason the extent count has one.
165const MAX_SECTIONS: usize = 4096;
166const FREQUENCY_CANDIDATES: usize = 32_768;
167const FREQUENCY_ENTRIES: usize = 512;
168const FREQUENCY_BUILD_RANK: usize = 10;
169const FREQUENCY_ORDINALS: usize = 131_072;
170const MAX_PAIR_FREQUENCIES: usize = 1024;
171/// The most exact heavy-hitter text one column may copy into the directory.
172///
173/// A column with unusually large leading values keeps the old code-only synopsis instead. The
174/// optimization must never turn a valid load into a directory-size failure.
175const FREQUENCY_TEXT_BUDGET: usize = 1024 * 1024;
176/// The most threads the two per column passes at the end of a commit are spread over.
177///
178/// A table like `hits` has ninety numeric columns, so on a machine with more cores than this the
179/// cap is what decides how long the frequencies take rather than the columns are. It is here at all
180/// because each worker holds a candidate table and a decoded part, and a hundred of those at once
181/// on a narrow machine would be worse than waiting.
182const MAX_FREQUENCY_WORKERS: usize = 32;
183
184/// How many threads the passes at the end of a commit are spread over on this machine.
185fn close_workers() -> usize {
186    std::thread::available_parallelism().map_or(1, usize::from).min(MAX_FREQUENCY_WORKERS)
187}
188
189/// The most threads one stripe's encode is spread over.
190///
191/// Higher than the frequency cap because this is the load itself rather than a pass at the end of
192/// it, and the work is one column of sixty four parts, which is large enough that a thread that
193/// takes one is not a thread that was started for nothing. A machine with more cores than this has
194/// the rest of them on the Parquet read, which is still one thread and is the other half of #808.
195const MAX_ENCODE_WORKERS: usize = 32;
196
197/// The most bytes one column of one part may spend on a membership sieve.
198///
199/// A part is a thousand rows, so a filter sized for every one of them being distinct is about
200/// thirteen hundred bytes and this never binds in practice. It is here so that a part that somehow
201/// arrives much wider than a vector cannot put an unbounded index in the file. What does bind is the
202/// rule in `Writer::encode_pages` that a sieve may not be as large as the part it indexes, which is a cap
203/// per column rather than one number for the whole file.
204const SIEVE_BUDGET: usize = 8 * 1024;
205
206/// The most bytes one end of a per part range may spend on a string.
207///
208/// A bound is allowed to be wider than the truth and never narrower, so a long string is cut down to
209/// this many bytes for the low end and cut down and then stepped up for the high end. The reason for
210/// a cap at all is that there are nine hundred and seventy four parts of a hundred and five columns
211/// in a million rows of ClickBench and `URL` runs to hundreds of bytes, so keeping every end whole
212/// would put more in the directory than the skipping is worth. Twenty four bytes is past the point
213/// where two URLs of the same site still look alike.
214const PART_BOUND_BYTES: usize = 24;
215
216fn io(error: std::io::Error) -> Error {
217    Error::io(error.to_string())
218}
219
220fn invalid(message: &str) -> Error {
221    Error::invalid_input(format!("invalid rudb native file: {message}"))
222}
223
224/// Adds a sequence of byte counts without an overflow the caller has to think about.
225fn sum(counts: impl Iterator<Item = u64>) -> u64 {
226    counts.fold(0, u64::saturating_add)
227}
228
229/// One column's span out of a per column list, or zero when the list is shorter than the column.
230fn span_bytes(spans: &[Span], at: usize) -> u64 {
231    spans.get(at).map_or(0, |span| u64::from(span.length))
232}
233
234/// One column's page out of a per column list, or zero when that column has no page at all.
235fn page_bytes(pages: &[Option<Page>], at: usize) -> u64 {
236    pages.get(at).and_then(Option::as_ref).map_or(0, Page::bytes)
237}
238
239/// Everything one column's global dictionary costs the file, its page and the blocks outside it.
240fn dictionary_bytes(table: &Table, at: usize) -> u64 {
241    page_bytes(&table.dictionaries, at)
242        .saturating_add(table.dictionary_payloads.get(at).copied().unwrap_or(0))
243}
244
245/// The xxHash64 of `bytes`, which is what every span this format stores is checked against.
246///
247/// It walks the input as chunks rather than as offsets into it, and that is the only thing about it
248/// worth a comment. The offset form reads `bytes[at..at + 8]`, and neither the slicing nor the
249/// `try_into` behind it can be proved in range by a compiler that does not know where `at` stopped,
250/// so each of the four lanes paid for a bounds check and a length check on every thirty two bytes.
251/// A chunk carries its own length, so both fold away and the loop is the multiplies and rotates it
252/// was meant to be. That loop runs over every byte of every span a query reads, which on ClickBench
253/// 8 is about five percent of the query.
254fn checksum(bytes: &[u8]) -> u64 {
255    seeded_checksum(bytes, 0)
256}
257
258/// A hundred and twenty eight bit name for `bytes`, as two xxHash64 walks under different seeds,
259/// with the format this build writes folded in so that a name made by one format is never taken
260/// for the name of a file in another.
261///
262/// For a caller outside this crate that has to name a file by what went into it, which is what a
263/// Parquet mirror's key is. See the global dictionary's use of the same pair for the arithmetic.
264#[must_use]
265pub fn content_name(bytes: &[u8]) -> u128 {
266    let seed = u64::from(FORMAT);
267    u128::from(seeded_checksum(bytes, seed)) << 64 | u128::from(seeded_checksum(bytes, !seed))
268}
269
270/// [`content_name`] of bytes that arrive in pieces, which gives the same name as the pieces joined.
271///
272/// A Parquet mirror is named for the file's footer, and that is 930 KB on the ten million row
273/// ClickBench file. Read whole to be hashed it is a freed megabyte in every process that opens the
274/// mirror, which the allocator keeps. Read a window at a time it is a window.
275#[derive(Debug, Clone)]
276pub struct ContentNamer {
277    seeds: [u64; 2],
278    lanes: [[u64; 4]; 2],
279    held: [u8; 32],
280    filled: usize,
281    length: u64,
282}
283
284impl Default for ContentNamer {
285    fn default() -> Self {
286        let seed = u64::from(FORMAT);
287        let seeds = [seed, !seed];
288        let lanes = seeds.map(|seed| {
289            [
290                seed.wrapping_add(XXH_P1).wrapping_add(XXH_P2),
291                seed.wrapping_add(XXH_P2),
292                seed,
293                seed.wrapping_sub(XXH_P1),
294            ]
295        });
296        Self { seeds, lanes, held: [0; 32], filled: 0, length: 0 }
297    }
298}
299
300impl ContentNamer {
301    /// Takes the next piece.
302    pub fn update(&mut self, mut bytes: &[u8]) {
303        self.length += bytes.len() as u64;
304        if self.filled > 0 {
305            let take = (32 - self.filled).min(bytes.len());
306            self.held[self.filled..self.filled + take].copy_from_slice(&bytes[..take]);
307            self.filled += take;
308            bytes = &bytes[take..];
309            if self.filled < 32 {
310                return;
311            }
312            let block = self.held;
313            self.lanes.iter_mut().for_each(|lanes| checksum_block(lanes, &block));
314            self.filled = 0;
315        }
316        let mut blocks = bytes.chunks_exact(32);
317        for block in blocks.by_ref() {
318            self.lanes.iter_mut().for_each(|lanes| checksum_block(lanes, block));
319        }
320        let rest = blocks.remainder();
321        self.held[..rest.len()].copy_from_slice(rest);
322        self.filled = rest.len();
323    }
324
325    /// The name of everything taken so far.
326    #[must_use]
327    pub fn finish(&self) -> u128 {
328        let rest = &self.held[..self.filled];
329        let [first, second] = [0, 1].map(|at| {
330            if self.length < 32 {
331                checksum_tail(self.seeds[at].wrapping_add(XXH_P5).wrapping_add(self.length), rest)
332            } else {
333                finish_checksum(self.lanes[at], rest, self.length)
334            }
335        });
336        u128::from(first) << 64 | u128::from(second)
337    }
338}
339
340/// The xxHash64 of `bytes` started from `seed`, which is the same walk with a different beginning.
341///
342/// A seed is here for one caller: a global dictionary decides whether two values are the same by
343/// their hashes rather than by their bytes, and one sixty four bit hash is not enough to do that
344/// with. Twenty million distinct values collide on sixty four bits about once in a hundred thousand
345/// loads, which for a wrong answer is far too often. Two hashes of the same value under different
346/// seeds are independent, so the pair is a hundred and twenty eight bits and the same arithmetic
347/// puts that at around one in 1e24.
348fn seeded_checksum(bytes: &[u8], seed: u64) -> u64 {
349    // Asked for before the loop rather than after it, because a `ChunksExact` settles what it
350    // cannot divide when it is built and hands back the same tail whether it has been walked or not.
351    let mut blocks = bytes.chunks_exact(32);
352    let rest = blocks.remainder();
353    if bytes.len() < 32 {
354        return checksum_tail(seed.wrapping_add(XXH_P5).wrapping_add(bytes.len() as u64), rest);
355    }
356    let mut lanes = [
357        seed.wrapping_add(XXH_P1).wrapping_add(XXH_P2),
358        seed.wrapping_add(XXH_P2),
359        seed,
360        seed.wrapping_sub(XXH_P1),
361    ];
362    for block in blocks.by_ref() {
363        checksum_block(&mut lanes, block);
364    }
365    finish_checksum(lanes, rest, bytes.len() as u64)
366}
367
368const XXH_P1: u64 = 11_400_714_785_074_694_791;
369const XXH_P2: u64 = 14_029_467_366_897_019_727;
370const XXH_P3: u64 = 1_609_587_929_392_839_161;
371const XXH_P4: u64 = 9_650_029_242_287_828_579;
372const XXH_P5: u64 = 2_870_177_450_012_600_261;
373
374fn checksum_round(state: u64, word: u64) -> u64 {
375    state.wrapping_add(word.wrapping_mul(XXH_P2)).rotate_left(31).wrapping_mul(XXH_P1)
376}
377
378fn checksum_word(chunk: &[u8]) -> u64 {
379    u64::from_le_bytes(chunk.try_into().expect("eight checksum bytes"))
380}
381
382/// One thirty two byte block into the four lanes.
383fn checksum_block(lanes: &mut [u64; 4], block: &[u8]) {
384    for (lane, chunk) in lanes.iter_mut().zip(block.chunks_exact(8)) {
385        *lane = checksum_round(*lane, checksum_word(chunk));
386    }
387}
388
389/// The lanes after every whole block, folded together with what was left over and the length.
390fn finish_checksum(lanes: [u64; 4], rest: &[u8], length: u64) -> u64 {
391    let merge = |state: u64, lane: u64| {
392        (state ^ checksum_round(0, lane)).wrapping_mul(XXH_P1).wrapping_add(XXH_P4)
393    };
394    let [one, two, three, four] = lanes;
395    let combined = one
396        .rotate_left(1)
397        .wrapping_add(two.rotate_left(7))
398        .wrapping_add(three.rotate_left(12))
399        .wrapping_add(four.rotate_left(18));
400    let hash = merge(merge(merge(merge(combined, one), two), three), four);
401    checksum_tail(hash.wrapping_add(length), rest)
402}
403
404/// The fewer than thirty two bytes after the last whole block, and the final mix.
405fn checksum_tail(mut hash: u64, mut rest: &[u8]) -> u64 {
406    let mut words = rest.chunks_exact(8);
407    for chunk in words.by_ref() {
408        hash ^= checksum_round(0, checksum_word(chunk));
409        hash = hash.rotate_left(27).wrapping_mul(XXH_P1).wrapping_add(XXH_P4);
410    }
411    rest = words.remainder();
412    if rest.len() >= 4 {
413        let (head, tail) = rest.split_at(4);
414        let quarter = u32::from_le_bytes(head.try_into().expect("four checksum bytes"));
415        hash ^= u64::from(quarter).wrapping_mul(XXH_P1);
416        hash = hash.rotate_left(23).wrapping_mul(XXH_P2).wrapping_add(XXH_P3);
417        rest = tail;
418    }
419    for &byte in rest {
420        hash ^= u64::from(byte).wrapping_mul(XXH_P5);
421        hash = hash.rotate_left(11).wrapping_mul(XXH_P1);
422    }
423    hash ^= hash >> 33;
424    hash = hash.wrapping_mul(XXH_P2);
425    hash ^= hash >> 29;
426    hash = hash.wrapping_mul(XXH_P3);
427    hash ^ (hash >> 32)
428}
429
430/// The checksum of `length` bytes of `file` from `offset`, read [`DIRECTORY_WINDOW`] at a time.
431///
432/// The same xxHash64 as [`checksum`], carried across reads rather than over one buffer, so that a
433/// directory can be checked without all of it being in memory at once. The four lanes take whole
434/// thirty two byte blocks, and a read that ends partway through one keeps the tail for the next.
435fn file_checksum(file: &File, offset: u64, length: usize) -> Result<u64> {
436    if length < 32 {
437        let mut bytes = vec![0; length];
438        read_at(file, offset, &mut bytes)?;
439        return Ok(checksum(&bytes));
440    }
441    let mut lanes = [XXH_P1.wrapping_add(XXH_P2), XXH_P2, 0, 0_u64.wrapping_sub(XXH_P1)];
442    let mut buffer = vec![0; DIRECTORY_WINDOW.min(length)];
443    let mut kept = 0;
444    let mut read = 0;
445    while read < length {
446        let want = (buffer.len() - kept).min(length - read);
447        read_at(file, offset + read as u64, &mut buffer[kept..kept + want])?;
448        read += want;
449        let filled = kept + want;
450        let whole = filled / 32 * 32;
451        for block in buffer[..whole].chunks_exact(32) {
452            checksum_block(&mut lanes, block);
453        }
454        buffer.copy_within(whole..filled, 0);
455        kept = filled - whole;
456    }
457    Ok(finish_checksum(lanes, &buffer[..kept], length as u64))
458}
459
460#[derive(Debug, Clone, Copy)]
461struct Slot {
462    offset: u64,
463    length: u32,
464    generation: u64,
465    hash: u64,
466}
467
468impl Slot {
469    fn bytes(self) -> [u8; SLOT_BYTES] {
470        let mut result = [0; SLOT_BYTES];
471        result[..8].copy_from_slice(&self.offset.to_le_bytes());
472        result[8..12].copy_from_slice(&self.length.to_le_bytes());
473        result[12..20].copy_from_slice(&self.generation.to_le_bytes());
474        result[20..28].copy_from_slice(&self.hash.to_le_bytes());
475        result
476    }
477
478    fn read(bytes: &[u8]) -> Self {
479        Self {
480            offset: u64::from_le_bytes(bytes[..8].try_into().expect("eight bytes")),
481            length: u32::from_le_bytes(bytes[8..12].try_into().expect("four bytes")),
482            generation: u64::from_le_bytes(bytes[12..20].try_into().expect("eight bytes")),
483            hash: u64::from_le_bytes(bytes[20..28].try_into().expect("eight bytes")),
484        }
485    }
486}
487
488#[derive(Debug, Clone, Copy)]
489struct Page {
490    offset: u64,
491    length: u32,
492    hash: u64,
493}
494
495impl Page {
496    /// How much of the file this page takes, for [`Reader::layout`].
497    fn bytes(&self) -> u64 {
498        u64::from(self.length)
499    }
500}
501
502#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
503enum FrequencyValue {
504    Null,
505    Integer(i128),
506    Code(u32),
507}
508
509/// A table keyed by the sixty four bits of the values the numeric frequency pass counts.
510///
511/// Every integer of every numeric column goes through one of these at least once when a table
512/// closes, and with the standard hasher that was a fifth of the close on its own, all of it SipHash
513/// guarding against an attacker who would have to choose the rows of the file being written.
514type FrequencyMap<V> = HashMap<u64, V, Spread>;
515
516/// Builds the hasher for [`FrequencyMap`].
517#[derive(Debug, Default, Clone, Copy)]
518struct Spread;
519
520impl std::hash::BuildHasher for Spread {
521    type Hasher = SpreadHasher;
522
523    fn build_hasher(&self) -> SpreadHasher {
524        SpreadHasher(0)
525    }
526}
527
528/// Folds each word in with a full width multiply whose two halves are xored together.
529///
530/// A plain multiply leaves the low bits of the hash as poor as the low bits of the key, and the
531/// table picks its bucket from the low bits, so a timestamp column, whose values are all multiples
532/// of a million microseconds, would pile into a sixty fourth of the buckets. Folding the high half
533/// of the product back in is what gives the low bits the whole word.
534#[derive(Debug)]
535struct SpreadHasher(u64);
536
537impl SpreadHasher {
538    fn mix(&mut self, word: u64) {
539        let product = u128::from(self.0 ^ word) * 0x9E37_79B9_7F4A_7C15_u128;
540        self.0 = (product as u64) ^ ((product >> 64) as u64);
541    }
542}
543
544impl std::hash::Hasher for SpreadHasher {
545    fn write(&mut self, bytes: &[u8]) {
546        for part in bytes.chunks(8) {
547            let mut word = [0; 8];
548            word[..part.len()].copy_from_slice(part);
549            self.mix(u64::from_le_bytes(word));
550        }
551    }
552
553    fn write_u32(&mut self, value: u32) {
554        self.mix(u64::from(value));
555    }
556
557    fn write_u64(&mut self, value: u64) {
558        self.mix(value);
559    }
560
561    fn write_i128(&mut self, value: i128) {
562        self.mix(value as u64);
563        self.mix((value >> 64) as u64);
564    }
565
566    fn write_isize(&mut self, value: isize) {
567        self.mix(value as u64);
568    }
569
570    fn finish(&self) -> u64 {
571        self.0
572    }
573}
574
575#[derive(Debug, Clone)]
576struct FrequencyEntry {
577    value: FrequencyValue,
578    count: u64,
579}
580
581/// Exact leading frequencies for one column.
582///
583/// Values outside `entries` occur at most `omitted_max` times. This lets a count-descending TopN
584/// use the synopsis only when its last winner is strictly above every omitted value.
585#[derive(Debug, Clone)]
586struct FrequencySummary {
587    entries: Vec<FrequencyEntry>,
588    omitted_max: u64,
589    ordinals: Vec<u64>,
590    ordinal_entries: Vec<u16>,
591}
592
593#[derive(Debug, Clone)]
594struct PairFrequencyEntry {
595    first_entry: u16,
596    second: Option<u32>,
597    count: u64,
598}
599
600/// Exact leading counts for one numeric frequency anchor and one stable string code space.
601///
602/// `omitted_max` covers both first-key values outside the numeric synopsis and pairs below the
603/// retained prefix. A TopN may therefore use the entries only when its boundary strictly exceeds
604/// this number.
605#[derive(Debug, Clone)]
606struct PairFrequencySummary {
607    first: u16,
608    second: u16,
609    entries: Vec<PairFrequencyEntry>,
610    omitted_max: u64,
611}
612
613/// One column's frequency synopsis, in memory or left where it is in the file.
614///
615/// A writer holds what it counted. A reader leaves every synopsis in the file and reads one back
616/// when a query asks about its column, because they are the largest thing in a directory once they
617/// are decoded, forty eight bytes an entry and nearly twenty thousand entries over `hits`, and
618/// most queries ask about none of them. Where one sits is found at open, by reading it through and
619/// checking it, so a torn synopsis is still refused when the table is opened.
620#[derive(Debug, Clone)]
621enum Frequencies {
622    Held(FrequencySummary),
623    /// Where the synopsis sits, and whether it was written with the value of each ordinal, which
624    /// is what the directory's frequency magic says and the synopsis itself does not.
625    Stored {
626        span: Span,
627        values: bool,
628    },
629}
630
631/// The values one column's frequency synopsis lists, with a bound on everything it left out.
632///
633/// What [`Reader::frequency_prefix`] answers. The counts are exact, and `omitted_max` is how many
634/// rows any value not in the list can hold, which is zero when nothing was left out at all.
635#[derive(Debug, Clone)]
636pub struct FrequencyPrefix {
637    /// Every value the synopsis lists, with the number of rows holding it, count descending.
638    pub entries: Vec<(Value, u64)>,
639    /// How many rows the most common value outside the list holds, and zero for a complete list.
640    pub omitted_max: u64,
641}
642
643/// Sparse row ordinals covered by a numeric frequency candidate set.
644#[derive(Debug, Clone, PartialEq)]
645pub struct FrequencyOccurrences {
646    /// Upper bound for the frequency of every value absent from the fetched rows.
647    pub omitted_max: u64,
648    /// Table-wide row ordinals in ascending order.
649    pub ordinals: Vec<u64>,
650    /// The retained heavy-hitter values named by `anchor_indices`.
651    pub anchors: Vec<Value>,
652    /// The index in `anchors` at each ordinal, or empty for a legacy FQ2 directory.
653    pub anchor_indices: Vec<u16>,
654}
655
656/// Exact grouped counts for a pair of values, in descending count order.
657pub type PairFrequencyCounts = Vec<(Vec<Value>, u64)>;
658
659/// Where one column's page for one stripe sits in the file.
660///
661/// A column page has no checksum of its own because every part inside it carries one, and the
662/// stripe's index page holds those. Checking a part on the way out of the page covers exactly the
663/// bytes a reader is about to decode, and covers them once whether the reader took the whole page
664/// or pulled one part out of the middle of it.
665#[derive(Debug, Clone, Copy, Default)]
666struct Span {
667    offset: u64,
668    length: u32,
669}
670
671/// One optional page for each column of a stripe, holding only the pages that are there.
672///
673/// A stripe has three of these, the membership, sieve and part range pages. As a
674/// `Vec<Option<Page>>` each was thirty two bytes a column whether the page was there or not, and
675/// over the ten million rows of `hits` that is half a megabyte at open for 7171 pages out of 16380
676/// slots. Kept sparse and packed, a page that is there is twenty four bytes and one that is not is
677/// nothing.
678#[derive(Debug, Clone, Default)]
679struct Pages {
680    columns: usize,
681    held: Box<[StripePage]>,
682}
683
684/// A page and the column it is for, packed so that the column sits where the padding was.
685#[derive(Debug, Clone, Copy)]
686struct StripePage {
687    offset: u64,
688    hash: u64,
689    length: u32,
690    column: u32,
691}
692
693impl Pages {
694    /// The pages of `columns` columns, one slot each in column order.
695    fn from_slots(slots: Vec<Option<Page>>) -> Result<Self> {
696        let mut held = Vec::with_capacity(slots.iter().flatten().count());
697        for (column, page) in slots.iter().enumerate() {
698            if let Some(page) = page {
699                let column =
700                    u32::try_from(column).map_err(|_| invalid("too many columns for a page"))?;
701                held.push(StripePage {
702                    offset: page.offset,
703                    hash: page.hash,
704                    length: page.length,
705                    column,
706                });
707            }
708        }
709        Ok(Self { columns: slots.len(), held: held.into_boxed_slice() })
710    }
711
712    /// The page of one column, if it has one.
713    fn get(&self, column: usize) -> Option<Page> {
714        let at = self.held.binary_search_by_key(&column, |placed| placed.column as usize).ok()?;
715        let placed = self.held[at];
716        Some(Page { offset: placed.offset, length: placed.length, hash: placed.hash })
717    }
718
719    /// One slot per column, in column order, the way the directory writes them.
720    fn slots(&self) -> impl Iterator<Item = Option<Page>> + '_ {
721        (0..self.columns).map(|column| self.get(column))
722    }
723
724    /// How much of the file one column's page takes, or zero when it has none.
725    fn bytes(&self, column: usize) -> u64 {
726        self.get(column).map_or(0, |page| page.bytes())
727    }
728}
729
730/// One independently readable stripe of a table.
731#[derive(Debug, Clone)]
732pub struct Stripe {
733    rows: usize,
734    /// Rows in each part, in source order. Kept in the directory so that mapping a row ordinal to a
735    /// part, which every sparse fetch does, never reads the file.
736    parts: Vec<u32>,
737    /// The index page: one section per column, holding a length and a checksum for every part and
738    /// then a checksum of the section itself, so that a reader can pread one column's section and
739    /// still know it is intact.
740    index: Span,
741    pages: Vec<Span>,
742    memberships: Pages,
743    /// One page per column holding the membership sieve of every part of the stripe, for the
744    /// columns that have one. A column whose parts all declined a sieve has no page at all.
745    sieves: Pages,
746    /// One page per column holding the two ends and the null count of every part of the stripe.
747    ///
748    /// The stripe's own `zone` below covers sixty four times as many rows, and on a column that is
749    /// not the one the rows are ordered by that is the difference between skipping half the file and
750    /// skipping all but three percent of it. On ClickBench 24 the cutoff the answer settles at
751    /// leaves eight stripes of sixteen alive and thirty parts of nine hundred and seventy four.
752    ///
753    /// A page per column rather than one page for the stripe, so that a query that compares one
754    /// column reads the ends of that column and not of the hundred and four beside it. Read lazily
755    /// for the same reason, like the sieves.
756    part_ranges: Pages,
757    zone: Zone,
758}
759
760impl Stripe {
761    /// Number of rows in this stripe.
762    #[must_use]
763    pub fn rows(&self) -> usize {
764        self.rows
765    }
766
767    /// Number of parts in this stripe.
768    #[must_use]
769    pub fn parts(&self) -> usize {
770        self.parts.len()
771    }
772
773    /// The two ends and the null count of every column over the whole stripe.
774    ///
775    /// In the directory and so in memory, which is what makes it the one a planner can ask. The
776    /// finer ones are a page per column per stripe in the file, read by [`Reader::skips`] when a
777    /// scan wants to know which parts to open.
778    #[must_use]
779    pub fn zone(&self) -> &Zone {
780        &self.zone
781    }
782}
783
784/// The committed table directory.
785#[derive(Debug, Clone)]
786pub struct Table {
787    name: String,
788    fields: Vec<Field>,
789    stripes: Vec<Stripe>,
790    rows: usize,
791    dictionaries: Vec<Option<Page>>,
792    /// Bytes of each column's dictionary payload that are outside its page, which is all of them
793    /// from format 27 and none of them before. See [`DICTIONARY_PAYLOADS`].
794    ///
795    /// Empty rather than a row of zeros on a table that has none, and read with `get` for that
796    /// reason, so that a table built by hand in a test does not have to know about it.
797    dictionary_payloads: Vec<u64>,
798    frequencies: Vec<Option<Frequencies>>,
799    pair_frequencies: Vec<PairFrequencySummary>,
800    /// String spellings aligned with each column's frequency entries.
801    ///
802    /// Empty for files written before `RUDBFT1`. A `None` entry is the null frequency entry; every
803    /// code entry in a column named by the block has its exact bytes here.
804    frequency_texts: Vec<Vec<Option<Vec<u8>>>>,
805    /// Exact candidate host aggregates and an upper bound for every omitted host.
806    host_groups: Option<host::HostSummary>,
807    /// How many distinct values each column holds, for the columns that know.
808    ///
809    /// A dictionary entry is made the first time a value is seen and nothing ever removes one, so
810    /// the size of the dictionary is the number of distinct values in the column. That is the whole
811    /// story for a column with no null in it, and the wrong number by one for a column with a null
812    /// in it, because a null row is written as the code for the empty string and makes an entry the
813    /// dictionary would not otherwise have. The writer knows which case it is, since it counts the
814    /// non-null rows that use each code while it builds the frequency summary, and the reader cannot
815    /// work it out from the dictionary alone. So the writer settles it here.
816    distincts: Vec<Option<u64>>,
817    /// The order the rows of this table are meant to be stored in, if anybody declared one.
818    ///
819    /// A declaration and not a measurement. Nothing here checks that the stripes actually arrived
820    /// in this order, and the reason it is worth storing anyway is that the order is the only thing
821    /// about a table that a rewrite destroys without anybody noticing. The fragment ranges prune on
822    /// whatever order the rows came in, so a table loaded sorted prunes and the same table after a
823    /// checkpoint that did not know to keep the order quietly stops pruning and nothing says why.
824    clustering: Option<Clustering>,
825    /// The file generation of the commit that last wrote this table's column pages.
826    ///
827    /// This is what spec/graph/03-the-file-format.md section 3.2 calls the table generation, and
828    /// the definition is deliberately about the pages rather than about the directory. A graph
829    /// section is a restatement of a column in terms of row ids, so what invalidates one is the
830    /// rows being renumbered, and nothing else. Adding a second table to the file, or attaching a
831    /// section to this one, commits a new file generation without touching a single row of this
832    /// table, and a definition that moved with those would declare every section in the file stale
833    /// for no reason.
834    ///
835    /// Zero on a table written before format 23, where nothing recorded it. Real generations start
836    /// at one, so zero can never match a section's stamp, and a table from format 22 has no
837    /// sections for it to match anyway.
838    generation: u64,
839    /// The graph sections this table carries, per spec/graph/03-the-file-format.md section 3.2.
840    ///
841    /// Empty for every table written before the section table existed, and empty is not a
842    /// degraded state: section 3.1 says deleting every graph section from a file changes no answer,
843    /// only the time, so a table with none here answers every query the same way and slower. That
844    /// is what lets this field arrive without a migration.
845    sections: Vec<Section>,
846}
847
848impl Table {
849    /// The SQL table name held by this snapshot.
850    #[must_use]
851    pub fn name(&self) -> &str {
852        &self.name
853    }
854
855    /// Columns in their SQL order.
856    #[must_use]
857    pub fn fields(&self) -> &[Field] {
858        &self.fields
859    }
860
861    /// Committed row count.
862    #[must_use]
863    pub fn rows(&self) -> usize {
864        self.rows
865    }
866
867    /// Independently readable stripes.
868    #[must_use]
869    pub fn stripes(&self) -> &[Stripe] {
870        &self.stripes
871    }
872
873    /// The order the rows are meant to be stored in, if this table was declared with one.
874    #[must_use]
875    pub fn clustering(&self) -> Option<&Clustering> {
876        self.clustering.as_ref()
877    }
878
879    /// The generation every section of this table is judged against.
880    ///
881    /// See the field. A caller deciding whether to read a section asks [`Section::usable`] with
882    /// this.
883    #[must_use]
884    pub fn generation(&self) -> u64 {
885        self.generation
886    }
887
888    /// Every graph section this table names, including the kinds this build does not know.
889    ///
890    /// Including them is the point. A caller that wants only the ones it can use asks
891    /// [`Section::usable`], and a caller rewriting the directory carries the rest through, so a
892    /// file opened by an older build and written again does not silently lose a section that build
893    /// had no name for.
894    #[must_use]
895    pub fn sections(&self) -> &[Section] {
896        &self.sections
897    }
898}
899
900/// One table's line in the catalog directory.
901///
902/// The small level of the two. It holds what opening a database needs and nothing else: the name to
903/// bind, the shape to plan against, the row count, and where the table's own directory sits. A file
904/// of eight tables is eight of these, and reading them costs the same whether the tables hold a
905/// thousand rows or a billion.
906///
907/// The name, the fields and the row count are repeated here rather than pointed at inside the table
908/// directory, which is the entire point of having two levels. A catalog that pointed at them would
909/// have to read every table directory at open to answer what tables there are, which is the cost
910/// this level exists to avoid.
911#[derive(Debug, Clone)]
912struct Entry {
913    name: String,
914    fields: Vec<Field>,
915    rows: usize,
916    /// Where this table's own directory sits, with the checksum it was committed under.
917    directory: Page,
918    /// Exact non-null, nonzero integer counts certified by the catalog checksum.
919    nonzero: Vec<Option<u64>>,
920    /// Exact sum and non-null count for signed integer columns.
921    aggregates: Vec<Option<(i128, u64)>>,
922}
923
924/// One view's line in the catalog directory.
925///
926/// A view has no pages, so unlike a table it is entirely here and there is no second level under it.
927/// What it is made of is text: the body the binder binds again at every reference, and the whole
928/// statement written back out, which is what `duckdb_views()` reports and nothing else reads.
929///
930/// The columns are a cache and they are written down anyway, which is worth saying out loud because
931/// a cache in a file looks like a mistake. It is what the pin does. Create a view on a file, open
932/// the file again in another process, and `duckdb_views()` answers `column_count` and `is_bound`
933/// true without anything having bound the body, so the list survived the write. Not writing it
934/// would answer null and false there, and the only way back would be to bind every view at open,
935/// which is the thing the cache exists to avoid.
936#[derive(Debug, Clone, PartialEq, Eq)]
937pub struct ViewEntry {
938    /// The view's own name, without the schema, the way a table entry holds its name.
939    pub name: String,
940    /// The query the view stands for, as the text that was written.
941    pub sql: String,
942    /// The whole `CREATE VIEW` written back out.
943    pub statement: String,
944    /// The column names the statement gave, which rename a prefix of what the body produces.
945    pub aliases: Vec<String>,
946    /// The columns the last bind of the body produced.
947    pub columns: Vec<Field>,
948}
949
950/// Where one column's bytes went, taken from the directory rather than by reading pages.
951#[derive(Debug, Clone)]
952pub struct ColumnLayout {
953    /// The column's name, so a report does not have to carry the field list beside this.
954    pub name: String,
955    /// The type, spelled the way the catalog spells it.
956    pub kind: String,
957    /// Every stripe's page of this column added up, which is the encoded data itself.
958    pub pages: u64,
959    /// Every stripe's exact code membership page for this column.
960    pub memberships: u64,
961    /// Every stripe's membership sieve page for this column.
962    pub sieves: u64,
963    /// Every stripe's per part range page for this column.
964    pub part_ranges: u64,
965    /// The table wide dictionary of this column, if it has one.
966    pub dictionary: u64,
967}
968
969impl ColumnLayout {
970    /// Everything this column costs, which is what the file would lose if the column went.
971    #[must_use]
972    pub fn total(&self) -> u64 {
973        self.pages
974            .saturating_add(self.memberships)
975            .saturating_add(self.sieves)
976            .saturating_add(self.part_ranges)
977            .saturating_add(self.dictionary)
978    }
979}
980
981/// Where a whole file's bytes went.
982///
983/// Every number here comes out of the committed directory, so taking it costs one directory read
984/// however large the file is. That is the point: a 45 GB table has to be able to say where it went
985/// without being read, or nobody will ask.
986///
987/// The parts that are not a column are kept apart rather than shared out over the columns. The
988/// stripe index page holds a section per column and could be split, and the directory and the
989/// header cannot be, so splitting one of the three and not the others would read as if the columns
990/// accounted for everything. They do not, and the gap is the thing worth looking at.
991#[derive(Debug, Clone)]
992pub struct Layout {
993    /// The size of the file on disk.
994    pub file: u64,
995    /// Committed rows.
996    pub rows: usize,
997    /// Committed stripes.
998    pub stripes: usize,
999    /// Committed parts, which is how many chunks a scan reads.
1000    pub parts: usize,
1001    /// One entry per column, in the table's column order.
1002    pub columns: Vec<ColumnLayout>,
1003    /// Every stripe's index page, which carries a length and a checksum for every part of every
1004    /// column and is charged per stripe rather than per column.
1005    pub indexes: u64,
1006    /// The committed directory itself, the one that was read to build this.
1007    pub directory: u64,
1008    /// The fixed header, which holds the magic, the format and the two directory slots.
1009    pub header: u64,
1010}
1011
1012impl Layout {
1013    /// Everything the columns cost together.
1014    #[must_use]
1015    pub fn columns_total(&self) -> u64 {
1016        self.columns.iter().map(ColumnLayout::total).fold(0, u64::saturating_add)
1017    }
1018
1019    /// What the file holds that this does not account for.
1020    ///
1021    /// A committed file is written once and never rewritten in place, so an earlier directory and
1022    /// the pages of an earlier snapshot are still in it. That is the honest place for them: they
1023    /// are bytes on disk that no column owns.
1024    #[must_use]
1025    pub fn unaccounted(&self) -> u64 {
1026        self.file
1027            .saturating_sub(self.columns_total())
1028            .saturating_sub(self.indexes)
1029            .saturating_sub(self.directory)
1030            .saturating_sub(self.header)
1031    }
1032}
1033
1034/// How one part of one column is stored, which is one row of `pragma_storage_info`.
1035///
1036/// Everything here is read off the file rather than worked out from the schema, because the whole
1037/// question this answers is what the encoder chose, and the encoder chooses per part. Two files
1038/// holding the same rows in a different order give different answers and that difference is the
1039/// reason to ask.
1040///
1041/// The encoding costs a read of the column's page, so this is not free the way [`Layout`] is. It is
1042/// one read per column per stripe rather than one per part, because a part is a few kilobytes out
1043/// of a page that is a quarter of a megabyte.
1044#[derive(Debug, Clone)]
1045pub struct StoredPart {
1046    /// Which stripe the part belongs to.
1047    pub stripe: usize,
1048    /// Which part of that stripe it is, counting from zero inside the stripe.
1049    pub part: usize,
1050    /// The table wide row number the part starts at.
1051    pub row: usize,
1052    /// How many rows it holds.
1053    pub rows: usize,
1054    /// What the encoder made of it, as a line of text like `DICT(PACKED, PACKED)`.
1055    pub encoding: String,
1056    /// The stored bytes of the part, which is what it costs in the file.
1057    pub bytes: u64,
1058    /// Where in the file the column page holding this part starts.
1059    pub page: u64,
1060    /// Where in that page the part starts.
1061    pub offset: u64,
1062    /// The smallest value the part holds, when the stored ranges say.
1063    pub low: Option<Value>,
1064    /// The largest, same.
1065    pub high: Option<Value>,
1066    /// How many of its rows are null, when the stored ranges say.
1067    pub nulls: Option<usize>,
1068}
1069
1070/// Seeds the second hash a global dictionary tells its values apart by.
1071///
1072/// Any value that is not zero does, since zero is the seed [`checksum`] already uses and the point
1073/// is only that the two hashes of one value are not the same number. This one is the fractional part
1074/// of the golden ratio in sixty four bits, which is the constant everything else here is built out
1075/// of and is as good a nothing-up-my-sleeve number as any.
1076const DICTIONARY_CHECK_SEED: u64 = 11_400_714_819_323_198_485;
1077
1078/// One column's table wide dictionary while the load is running.
1079///
1080/// The thing to understand about this is what it does not hold. A dictionary of `URL` at a hundred
1081/// million ClickBench rows has about eighteen million distinct values and 1.3 GB of bytes in them,
1082/// and five columns like it are twelve of the seventeen gigabytes a load of `hits` peaks at. So the
1083/// bytes are not kept. A value's bytes go into [`GlobalDictionary::filling`], and when that reaches
1084/// [`TEXT_PAYLOAD_VALUES`] values the block is sealed, handed to [`encode_ready`] at the end of the
1085/// stripe and never seen in that form again. What is left is the encoded block, which is two to
1086/// three times smaller, and that is the same bytes the file is going to hold anyway.
1087///
1088/// Two things needed the raw bytes and neither needs them now. Deciding whether a value has been
1089/// seen before was a hash lookup and then a comparison of the bytes, and is now a hash lookup and a
1090/// comparison of a second hash under a different seed, which is [`DICTIONARY_CHECK_SEED`] and the
1091/// argument for why that is sound. Sorting the values at the end needed all of them at once, and
1092/// now reads the blocks back through [`GlobalDictionary::decoded`] one column at a time, which is
1093/// one column's bytes rather than every column's.
1094///
1095/// The offsets going block relative comes free with it, and takes the four gigabyte wall with it.
1096/// They were `u32` into a per column payload, so a column could not hold more than four gigabytes of
1097/// values however much memory the machine had, and `URL` and `Referer` are within a small factor of
1098/// that at a hundred million rows. A `u32` into a block of 1,024 values is not a bound anything real
1099/// reaches. The stored form is unchanged, because [`encode_offsets`] was already subtracting a per
1100/// block base before writing.
1101#[derive(Debug)]
1102struct GlobalDictionary {
1103    primary: HashMap<u64, u32>,
1104    collisions: HashMap<u64, Vec<u32>>,
1105    /// Every value's hash under [`DICTIONARY_CHECK_SEED`], in code order.
1106    checks: Vec<u64>,
1107    /// Where every value ends inside the payload block it is in, in code order.
1108    ends: Vec<u32>,
1109    counts: Vec<u64>,
1110    nulls: u64,
1111    /// The values of the block being filled, back to back.
1112    filling: Vec<u8>,
1113    /// One conservative four-byte substring signature per sealed payload block.
1114    grams: Vec<[u8; TEXT_GRAM_BYTES]>,
1115    /// Blocks that have filled and not been encoded yet, each with its block number.
1116    ///
1117    /// Empty except between a block filling and the end of the stripe that filled it, and while the
1118    /// column is still too small to settle a shape on.
1119    waiting: Vec<(usize, Vec<u8>)>,
1120    /// Blocks kept raw to settle a shape on, spread across the column, each with its number.
1121    ///
1122    /// At most [`PAYLOAD_SAMPLE_BLOCKS`] of them and so at most a few megabytes. Spread rather than
1123    /// taken off the front for the reason [`settle_shape`] gives, and kept rather than read back
1124    /// because reading back is a decode and this is a sample of a column that is still growing.
1125    sample: Vec<(usize, Vec<u8>)>,
1126    /// How far apart the blocks in `sample` are, which doubles every time there are too many.
1127    stride: usize,
1128    /// What the blocks encoded so far were encoded with, once the column is big enough to settle it.
1129    shape: Option<chooser::Settled>,
1130    /// How many blocks had filled when that shape was settled.
1131    settled: usize,
1132    /// The blocks that are encoded and not yet in the file, in block order, following `placed`.
1133    ///
1134    /// Empty between stripes, because [`Writer::place_blocks`] writes them the moment
1135    /// [`encode_ready`] hands them back. Only a dictionary that never meets a writer, which is a
1136    /// test's, keeps them here.
1137    blocks: Vec<Vec<u8>>,
1138    /// Where every block already written to the file is, in block order.
1139    placed: Vec<Placed>,
1140}
1141
1142/// Where one payload block of a global dictionary is in the file, and its checksum.
1143#[derive(Debug, Clone, Copy)]
1144struct Placed {
1145    start: u64,
1146    length: u64,
1147    hash: u64,
1148}
1149
1150/// Sorted `(head, code)` entries and the decoded bytes and block bases they were sorted over.
1151type RankedDictionary = (Vec<(u64, u32)>, Vec<u8>, Vec<u64>);
1152
1153impl GlobalDictionary {
1154    fn new() -> Self {
1155        Self {
1156            primary: HashMap::new(),
1157            collisions: HashMap::new(),
1158            checks: Vec::new(),
1159            ends: Vec::new(),
1160            counts: Vec::new(),
1161            nulls: 0,
1162            filling: Vec::new(),
1163            grams: Vec::new(),
1164            waiting: Vec::new(),
1165            sample: Vec::new(),
1166            stride: 1,
1167            shape: None,
1168            settled: 0,
1169            blocks: Vec::new(),
1170            placed: Vec::new(),
1171        }
1172    }
1173
1174    /// How many distinct values this dictionary holds, which is one past its largest code.
1175    fn values(&self) -> usize {
1176        self.ends.len()
1177    }
1178
1179    /// How many blocks are encoded, written or not, which is the number the next one has to have.
1180    fn encoded(&self) -> usize {
1181        self.placed.len() + self.blocks.len()
1182    }
1183
1184    #[cfg(test)]
1185    fn code(&mut self, text: &str) -> Result<u32> {
1186        let bytes = text.as_bytes();
1187        self.code_hashed(bytes, checksum(bytes), seeded_checksum(bytes, DICTIONARY_CHECK_SEED))
1188    }
1189
1190    /// The code for a value whose two hashes the caller already has.
1191    ///
1192    /// A stripe prepared outside the writer's lock hashed every value it holds while it was coding
1193    /// them, and merging it into this dictionary is one of these a distinct value rather than two
1194    /// hashes of every row. See [`prepare`].
1195    fn code_hashed(&mut self, text: &[u8], hash: u64, check: u64) -> Result<u32> {
1196        if let Some(&code) = self.primary.get(&hash) {
1197            if self.checks.get(code as usize) == Some(&check) {
1198                return Ok(code);
1199            }
1200            if let Some(codes) = self.collisions.get(&hash) {
1201                if let Some(code) =
1202                    codes.iter().copied().find(|&code| self.checks[code as usize] == check)
1203                {
1204                    return Ok(code);
1205                }
1206            }
1207            let code = self.insert(text, check)?;
1208            self.collisions.entry(hash).or_default().push(code);
1209            return Ok(code);
1210        }
1211        let code = self.insert(text, check)?;
1212        self.primary.insert(hash, code);
1213        Ok(code)
1214    }
1215
1216    fn insert(&mut self, text: &[u8], check: u64) -> Result<u32> {
1217        let code = u32::try_from(self.ends.len())
1218            .map_err(|_| invalid("global dictionary has too many values"))?;
1219        self.filling.extend_from_slice(text);
1220        self.ends.push(
1221            u32::try_from(self.filling.len())
1222                .map_err(|_| invalid("a global dictionary value exceeds 4 GiB"))?,
1223        );
1224        self.checks.push(check);
1225        self.counts.push(0);
1226        if self.ends.len() % TEXT_PAYLOAD_VALUES == 0 {
1227            self.seal();
1228        }
1229        Ok(code)
1230    }
1231
1232    /// Closes the block being filled and puts it in the queue to be encoded.
1233    ///
1234    /// Also keeps a copy of it if it lands on the sample's stride, and halves the sample when that
1235    /// has left too many, which is what keeps the kept blocks spread evenly over however much of the
1236    /// column exists rather than bunched at whichever end was cheap to remember.
1237    fn seal(&mut self) {
1238        let at = self.ends.len().div_ceil(TEXT_PAYLOAD_VALUES) - 1;
1239        let bytes = std::mem::take(&mut self.filling);
1240        let mut grams = [0_u8; TEXT_GRAM_BYTES];
1241        for value in self.slices(at, &bytes) {
1242            for gram in value.windows(4) {
1243                for bit in gram_bits(gram) {
1244                    grams[bit / 8] |= 1 << (bit % 8);
1245                }
1246            }
1247        }
1248        self.grams.push(grams);
1249        if at % self.stride == 0 {
1250            self.sample.push((at, bytes.clone()));
1251            if self.sample.len() > PAYLOAD_SAMPLE_BLOCKS {
1252                self.stride *= 2;
1253                let stride = self.stride;
1254                self.sample.retain(|(at, _)| at % stride == 0);
1255            }
1256        }
1257        self.waiting.push((at, bytes));
1258    }
1259
1260    /// The values of one block, as slices into the bytes the block was filled with.
1261    fn slices<'a>(&self, at: usize, bytes: &'a [u8]) -> Vec<&'a [u8]> {
1262        let first = at * TEXT_PAYLOAD_VALUES;
1263        let last = (first + TEXT_PAYLOAD_VALUES).min(self.ends.len());
1264        let mut out = Vec::with_capacity(last.saturating_sub(first));
1265        let mut from = 0;
1266        for value in first..last {
1267            let to = self.ends[value] as usize;
1268            out.push(&bytes[from..to]);
1269            from = to;
1270        }
1271        out
1272    }
1273
1274    /// Settles the shape the waiting blocks are about to be encoded with, if there is enough column
1275    /// to settle one on.
1276    ///
1277    /// Settled again once the column has grown fourfold, because the sample it was settled on then
1278    /// covered a quarter of what exists now and a dictionary in first seen order does not look the
1279    /// same at both ends. Blocks already encoded keep the shape they were encoded with. They can,
1280    /// because a block says what it is: nothing reading one asks the column what shape to expect.
1281    fn settle(&mut self) -> Result<()> {
1282        if self.sample.len() < PAYLOAD_SAMPLE_BLOCKS {
1283            return Ok(());
1284        }
1285        let complete = self.ends.len() / TEXT_PAYLOAD_VALUES;
1286        if self.shape.is_some() && complete < self.settled.saturating_mul(4) {
1287            return Ok(());
1288        }
1289        let sample =
1290            self.sample.iter().map(|(at, bytes)| self.slices(*at, bytes)).collect::<Vec<_>>();
1291        self.shape = Some(settle_shape(&sample)?);
1292        self.settled = complete;
1293        Ok(())
1294    }
1295
1296    /// Seals the part block at the end of the load, if there is one.
1297    fn seal_rest(&mut self) {
1298        // Asked of the values rather than of the bytes, because a block of empty strings has values
1299        // in it and no bytes, and a column of nulls is exactly that.
1300        if self.ends.len() % TEXT_PAYLOAD_VALUES != 0 {
1301            self.seal();
1302        }
1303    }
1304
1305    /// Encodes the waiting block at `at`, with the settled shape when there is one and by trying
1306    /// everything when the column was too small to settle one.
1307    fn encode_waiting(&self, at: usize) -> Result<Vec<u8>> {
1308        let (block, bytes) = &self.waiting[at];
1309        let values = self.slices(*block, bytes);
1310        match &self.shape {
1311            Some(shape) => string::encode_with(&values, shape),
1312            None => string::encode(&values),
1313        }
1314    }
1315
1316    /// [`finish_dictionaries`] for one dictionary on this thread, for the tests that hold one.
1317    #[cfg(test)]
1318    fn finish_blocks(&mut self) -> Result<()> {
1319        self.seal_rest();
1320        let made = (0..self.waiting.len())
1321            .map(|at| self.encode_waiting(at))
1322            .collect::<Result<Vec<_>>>()?;
1323        for ((at, _), bytes) in std::mem::take(&mut self.waiting).into_iter().zip(made) {
1324            if self.encoded() != at {
1325                return Err(Error::internal("a dictionary block was encoded out of order"));
1326            }
1327            self.blocks.push(bytes);
1328        }
1329        Ok(())
1330    }
1331
1332    /// Every value of this dictionary read back out of its encoded blocks, as the bytes back to back
1333    /// and where each block starts in them.
1334    ///
1335    /// This is the one place the whole column is in memory at once and the reason [`Writer::close`]
1336    /// takes the columns one at a time rather than across threads. One column's values is 1.3 GB on
1337    /// the worst ClickBench column, and five columns of that at once is the peak this was all meant
1338    /// to remove.
1339    ///
1340    /// The blocks are spread over threads instead. Each block's decoded length is already known from
1341    /// the ends of its values, so the answer is laid out before anything is decoded and every thread
1342    /// decodes its own run of blocks straight into its own part of it. On the 10m ClickBench sample
1343    /// this was a second of the close for `URL` alone, on one core of thirty two, and the close is
1344    /// what a load waits on once its stripes are written.
1345    ///
1346    /// The blocks already written are read back out of `file`, so what a thread holds beyond the
1347    /// answer is one encoded block. They were written moments or minutes ago and are almost always
1348    /// still in the page cache, so this is a copy rather than a read of the disk.
1349    fn decoded(&self, file: Option<&File>) -> Result<(Vec<u8>, Vec<u64>)> {
1350        let count = self.placed.len() + self.blocks.len();
1351        if count != self.values().div_ceil(TEXT_PAYLOAD_VALUES) {
1352            return Err(invalid("global dictionary blocks do not cover its values"));
1353        }
1354        let mut bases = Vec::with_capacity(count);
1355        let mut total = 0_usize;
1356        for block in 0..count {
1357            bases.push(total as u64);
1358            let last = ((block + 1) * TEXT_PAYLOAD_VALUES).min(self.values()) - 1;
1359            total = total
1360                .checked_add(self.ends[last] as usize)
1361                .ok_or_else(|| invalid("global dictionary does not fit in memory"))?;
1362        }
1363        let mut flat = vec![0_u8; total];
1364        let mut outs = Vec::with_capacity(count);
1365        let mut rest = flat.as_mut_slice();
1366        for block in 0..count {
1367            let end = bases.get(block + 1).map_or(total, |&base| base as usize);
1368            let (out, after) = rest.split_at_mut(end - bases[block] as usize);
1369            outs.push((block, out));
1370            rest = after;
1371        }
1372        let one = |run: &mut [(usize, &mut [u8])]| -> Result<()> {
1373            let mut stored = Vec::new();
1374            for (block, out) in run {
1375                let encoded = match self.placed.get(*block) {
1376                    Some(place) => {
1377                        let file = file.ok_or_else(|| {
1378                            Error::internal("a written dictionary block has no file")
1379                        })?;
1380                        let length = usize::try_from(place.length).map_err(|_| {
1381                            invalid("global dictionary block does not fit in memory")
1382                        })?;
1383                        stored.resize(length, 0);
1384                        read_at(file, place.start, &mut stored)?;
1385                        if checksum(&stored) != place.hash {
1386                            return Err(invalid(
1387                                "a global dictionary block did not read back as written",
1388                            ));
1389                        }
1390                        stored.as_slice()
1391                    }
1392                    None => &self.blocks[*block - self.placed.len()],
1393                };
1394                let decoded = string::decode_flat(encoded)?;
1395                if decoded.bytes().len() != out.len() {
1396                    return Err(invalid(
1397                        "a global dictionary block is not the length its ends say",
1398                    ));
1399                }
1400                out.copy_from_slice(decoded.bytes());
1401            }
1402            Ok(())
1403        };
1404        // Sixteen blocks a thread at the least, because a thread costs about what decoding a few
1405        // blocks does and most columns have one or two.
1406        let workers = close_workers().min(count / 16).max(1);
1407        if workers <= 1 {
1408            one(&mut outs)?;
1409        } else {
1410            let per = count.div_ceil(workers);
1411            std::thread::scope(|scope| {
1412                outs.chunks_mut(per)
1413                    .map(|run| scope.spawn(|| one(run)))
1414                    .collect::<Vec<_>>()
1415                    .into_iter()
1416                    .try_for_each(|handle| {
1417                        handle.join().map_err(|_| {
1418                            Error::internal("a global dictionary decode worker panicked")
1419                        })?
1420                    })
1421            })?;
1422        }
1423        drop(outs);
1424        Ok((flat, bases))
1425    }
1426
1427    /// Where the value at `code` sits in the bytes [`GlobalDictionary::decoded`] handed back.
1428    ///
1429    /// A block's first value starts at the block, and every other value starts where the one before
1430    /// it ended, which is what makes 1,024 values 1,024 numbers rather than 1,025.
1431    fn value_span(ends: &[u32], bases: &[u64], code: usize) -> (usize, usize) {
1432        let Some(&base) = bases.get(code / TEXT_PAYLOAD_VALUES) else { return (0, 0) };
1433        let Some(&end) = ends.get(code) else { return (0, 0) };
1434        let base = base as usize;
1435        let from = if code % TEXT_PAYLOAD_VALUES == 0 { 0 } else { ends[code - 1] as usize };
1436        (base + from, base + end as usize)
1437    }
1438
1439    /// This dictionary's values in sorted order, each as the first eight bytes of the value and the
1440    /// code that holds it, so entry `rank` describes the value that sits at `rank` when the values
1441    /// are sorted by their bytes.
1442    ///
1443    /// Codes themselves stay in first appearance order, which is what lets the writer hand one out
1444    /// the moment it sees a value rather than waiting for the last stripe, and which also keeps a
1445    /// stripe's codes close together because the data is clustered. This is what puts the values
1446    /// back in order for anything that needs it, and it is separate from the codes so that getting
1447    /// it costs a sort of the distinct values at the end rather than a rewrite of every code page.
1448    ///
1449    /// The order is the byte order of the values and nothing else. The heads are attached after the
1450    /// sort rather than sorted on, because padding with zero on the right is order preserving for
1451    /// byte strings and so sorting by head and then by bytes lands in the same place as sorting by
1452    /// bytes: a shorter value differs from a longer one that starts the same way at a position
1453    /// where the shorter one has run out, and zero is below every byte that could be there.
1454    ///
1455    /// The heads are kept because a reader searching this order wants a comparison it can make out
1456    /// of the index alone. What they buy there depends entirely on the column and is much less than
1457    /// it looks on the columns that cost the most, which [`sort_by_value`] measures.
1458    fn ranked_with_values(&self, file: Option<&File>) -> Result<RankedDictionary> {
1459        let (flat, bases) = self.decoded(file)?;
1460        let value = |code: u32| {
1461            let (from, to) = Self::value_span(&self.ends, &bases, code as usize);
1462            flat.get(from..to).unwrap_or_default()
1463        };
1464        let mut codes = (0..self.values() as u32).collect::<Vec<_>>();
1465        sort_by_value_across(&mut codes, value, close_workers());
1466        let order = codes.into_iter().map(|code| (head(value(code)), code)).collect();
1467        Ok((order, flat, bases))
1468    }
1469
1470    #[cfg(test)]
1471    fn ranked(&self, file: Option<&File>) -> Result<Vec<(u64, u32)>> {
1472        self.ranked_with_values(file).map(|(order, _, _)| order)
1473    }
1474}
1475
1476/// Appends pages and commits a new directory.
1477///
1478/// One writer covers a whole file rather than one table. [`Writer::next`] closes the table it is on
1479/// and opens another over the same file, and [`Writer::finish`] commits every table it has closed in
1480/// one generation. That is what makes a checkpoint atomic across tables: there is one slot write at
1481/// the end of it and a reader sees every table at the generation before it or every table at the
1482/// generation after it.
1483#[derive(Debug)]
1484pub struct Writer {
1485    file: File,
1486    /// Where the next write goes, counted here rather than asked of the file.
1487    ///
1488    /// The file's own cursor is not ours. Building the numeric frequencies reads pages back through
1489    /// [`read_at`], and a positional read is only positional about where it reads from: `pread`
1490    /// leaves the cursor alone, and the call Windows has for it moves the cursor to the end of what
1491    /// it read. A writer that asked the file where it was would then write the directory over a
1492    /// page it had already written, which is what it did.
1493    at: u64,
1494    table: Table,
1495    generation: u64,
1496    /// The first and the last source position in every stripe, in the order the stripes were
1497    /// written.
1498    order: Vec<((u64, u64), (u64, u64))>,
1499    next_order: u64,
1500    dictionaries: Vec<Option<GlobalDictionary>>,
1501    /// Which columns still have a global dictionary, shared with every [`Preparer`] this writer
1502    /// hands out so that a stripe prepared after a column lost its dictionary is not coded for it.
1503    coded: Arc<[AtomicBool]>,
1504    /// One per column, folding the rows into a summary and a sketch as they go past.
1505    ///
1506    /// `None` for a column with no hash rule, which is the interval and the nested types. See
1507    /// [`stats::Gather`] for why the statistics are built here rather than by reading the file back
1508    /// once it is committed.
1509    gathers: Vec<Option<stats::Gather>>,
1510    pending: Vec<PendingChunk>,
1511    /// The tables already closed in this generation, in the order they were written.
1512    closed: Vec<Entry>,
1513    /// The views the next commit writes down, which [`Writer::with_views`] sets.
1514    ///
1515    /// Carried forward from the committed generation by [`Writer::open`], so a writer that was only
1516    /// opened to append a table does not have to know about views to avoid dropping them.
1517    views: Vec<ViewEntry>,
1518    /// Where the stages this writer runs are charged, which [`Writer::with_profile`] sets.
1519    ///
1520    /// The writer runs the page builder, the dictionary blocks, the writes and the publish, and it
1521    /// charges them once per stripe and once per worker, never per chunk. See
1522    /// `rudb_metrics::LoadProfile` for why that is the grain.
1523    profile: Option<Arc<LoadProfile>>,
1524}
1525
1526/// A chunk that has arrived and is waiting for the rest of its stripe.
1527///
1528/// The rows are kept rather than the pages they encode to, which is the whole of #808's first half.
1529/// Encoding on arrival put every column of every part on the thread that called `append_at`, and
1530/// that thread is the only one the load has. Encoding at the flush instead means a stripe's worth
1531/// of work is on the table at once, and a stripe splits by column into a hundred and five pieces
1532/// that share nothing.
1533#[derive(Debug)]
1534struct PendingChunk {
1535    order: (u64, u64),
1536    chunk: Chunk,
1537}
1538
1539/// What the writer still needs of a part once its columns are encoded: where in the source it came
1540/// from, how many rows it has and how large those rows were.
1541///
1542/// A stripe waiting for the writer's lock carries these rather than its chunks, so its rows are
1543/// freed as soon as they are encoded and not after the stripe is written. See [`prepare`].
1544#[derive(Debug, Clone, Copy)]
1545struct Part {
1546    order: (u64, u64),
1547    rows: usize,
1548    footprint: usize,
1549}
1550
1551impl Part {
1552    fn of(pending: &PendingChunk) -> Self {
1553        Self {
1554            order: pending.order,
1555            rows: pending.chunk.len(),
1556            footprint: pending.chunk.footprint(),
1557        }
1558    }
1559}
1560
1561/// One column's share of a stripe, which is what one encode worker produces.
1562///
1563/// Indexed by part, so a stripe is a column of these and the write loop reads down one of them.
1564/// That is also the order the loop wanted: `flush_pending` walks a column at a time and lays its
1565/// parts next to each other, and it used to reach across a row of parts to do it.
1566#[derive(Debug)]
1567struct ColumnStripe {
1568    pages: Vec<Vec<u8>>,
1569    codes: Vec<Option<Vec<u32>>>,
1570    sieves: Vec<Option<Sieve>>,
1571    ranges: Vec<Range>,
1572}
1573
1574/// Roughly what encoding a column of this type costs, for ordering the encode queue.
1575///
1576/// Only the order matters and only roughly. A string column hashes and copies every value into a
1577/// dictionary and is in a different class from everything else, and among the fixed widths the wide
1578/// ones carry more bytes through the cascade than the narrow ones. Anything finer than that would
1579/// be a cost model, and the queue already absorbs a wrong guess: it only has to avoid finishing on
1580/// a column nobody else can help with.
1581fn weight(ty: &LogicalType) -> usize {
1582    match ty {
1583        LogicalType::Varchar | LogicalType::Blob | LogicalType::Bit => 64,
1584        LogicalType::HugeInt
1585        | LogicalType::UHugeInt
1586        | LogicalType::Uuid
1587        | LogicalType::Interval => 16,
1588        LogicalType::BigInt
1589        | LogicalType::UBigInt
1590        | LogicalType::Timestamp
1591        | LogicalType::Time
1592        | LogicalType::TimeTz
1593        | LogicalType::TimestampTz
1594        | LogicalType::TimestampS
1595        | LogicalType::TimestampMs
1596        | LogicalType::TimestampNs
1597        | LogicalType::Double
1598        | LogicalType::Decimal { .. } => 8,
1599        LogicalType::Integer | LogicalType::UInteger | LogicalType::Date | LogicalType::Float => 4,
1600        LogicalType::SmallInt | LogicalType::USmallInt => 2,
1601        _ => 1,
1602    }
1603}
1604
1605/// Parts in one stripe.
1606///
1607/// Sixty four thousand rows is the smallest stripe that keeps the ClickBench directory in single
1608/// digit megabytes at a hundred million rows, and it puts a four byte column's page at a quarter of
1609/// a megabyte, which is the size a sequential read wants. Larger stripes buy a smaller directory
1610/// and cost a sparse fetch, which has to read a page index before it can reach one part.
1611pub const STRIPE_PARTS: usize = 64;
1612
1613/// How many rows the writer wants to see before it decides whether a varchar column gets to keep
1614/// its global dictionary.
1615///
1616/// See [`prepare::drops_dictionary`]. A stripe is up to [`STRIPE_PARTS`] parts, so most tables give it
1617/// far more than this and it binds only on a table that is smaller than one stripe. A handful of
1618/// rows says nothing about whether a column repeats itself, and the answer that costs nothing when
1619/// the sample is that small is the one the writer has always given, which is to keep the dictionary.
1620const DICTIONARY_DECIDE_ROWS: usize = 4_096;
1621
1622/// Out of ten. A varchar column loses its dictionary when more than this many rows in ten of the
1623/// first stripe held a value that stripe had not seen before.
1624///
1625/// See [`prepare::drops_dictionary`]. Nine and not five, because the properties a dictionary buys are
1626/// worth keeping everywhere they are real. On ClickBench the widest string column is `Referer` at
1627/// 0.131 of its first stripe and every other one is below that, so nothing there is near this and
1628/// every one of them keeps its dictionary, which is what a group by on codes wants. On TPC-H
1629/// `o_comment` and `c_comment` are at 0.97 and are what this catches.
1630///
1631/// `l_comment` sits at 0.883 and so keeps its dictionary. Eight was built and measured rather than
1632/// argued about, and it is not a clear win: it takes `select l_comment from lineitem` from 4.335 G
1633/// instructions to 3.473 G and the file from 280.2 MB to 260.4 MB, and it takes a `like` over the
1634/// same column from 3.29 G to 4.27 G, because a dictionary runs the predicate once a distinct value
1635/// and there are 3.6 M of those to 6.0 M rows. The 22 query suite came out 6.91 s against 7.07 s in
1636/// favour of nine. So nine stays until there is a reason to prefer one of those shapes. See #1137.
1637const DICTIONARY_DISTINCT_IN_TEN: usize = 9;
1638
1639/// Bytes one part takes in a stripe's index page: four for the length, eight for the checksum.
1640const INDEX_ENTRY: usize = size_of::<u32>() + size_of::<u64>();
1641
1642/// Bytes one column's section of a stripe's index page takes, including its own trailing checksum.
1643fn index_section(parts: usize) -> Result<usize> {
1644    parts
1645        .checked_mul(INDEX_ENTRY)
1646        .and_then(|bytes| bytes.checked_add(size_of::<u64>()))
1647        .ok_or_else(|| invalid("index page length overflow"))
1648}
1649
1650impl Writer {
1651    /// Opens a committed file and starts a table in the generation after the one it holds.
1652    ///
1653    /// The tables already in the file are carried forward by name and by directory pointer, and
1654    /// their pages are not read. Nothing in the file is overwritten: the new table's pages and the
1655    /// new catalog go on the end, past the catalog the committed generation points at, and the one
1656    /// write that is not an append is the slot in the header that [`Writer::finish`] does last.
1657    ///
1658    /// That slot is the other one. A file committed at generation 1 is named by the slot at 16 and
1659    /// generation 2 writes the one at 44, so until the last four bytes of the commit land the file
1660    /// still reads as the generation before it, and a slot torn across a write fails its checksum
1661    /// and the reader falls back to the one beside it. This is what the second slot has always been
1662    /// for.
1663    ///
1664    /// # Errors
1665    ///
1666    /// If the file has no valid committed directory, is not this build's format, repeats the name
1667    /// of a table already in it that holds rows, has a field with no scalar encoding, or cannot be
1668    /// written.
1669    pub fn open(
1670        path: impl AsRef<Path>,
1671        name: impl Into<String>,
1672        fields: Vec<Field>,
1673    ) -> Result<Self> {
1674        for field in &fields {
1675            type_tag(&field.ty)?;
1676        }
1677        let name = name.into();
1678        let path = path.as_ref();
1679        let (_, size, slot, bytes, _) = slot_bytes(path)?;
1680        let (mut closed, views) = decode_catalog(&bytes, size)?;
1681        // A table already in the file under this name is only in the way if it holds rows. One that
1682        // holds none has no pages for this generation to carry and no reader that could lose
1683        // anything, so the table being started here takes its place in the catalog rather than
1684        // colliding with it, and `finish` writes the new entry where the old one was.
1685        //
1686        // That is not a corner. It is the shape every loading script writes: the schema goes in one
1687        // statement and the rows go in the next, and a checkpoint between them commits the empty
1688        // table. Before this, the second statement had to build the whole table in memory because
1689        // the first had already put the name in the file, which is how a load of a table larger
1690        // than memory became a load that needed memory the size of the table.
1691        if let Some(at) = closed.iter().position(|held| held.name == name) {
1692            if closed[at].rows > 0 {
1693                return Err(invalid("two tables in one native file have the same name"));
1694            }
1695            closed.remove(at);
1696        }
1697        // The generation of the slot whose bytes checksummed, and not the highest number in the
1698        // header. A slot torn across a write can hold any number at all, and taking that one would
1699        // be choosing which slot to overwrite from a value nothing has vouched for, which is how a
1700        // half written commit gets to destroy the one good copy beside it.
1701        let generation = slot
1702            .generation
1703            .checked_add(1)
1704            .ok_or_else(|| invalid("native file generation overflow"))?;
1705        let file = OpenOptions::new().write(true).read(true).open(path).map_err(io)?;
1706        Ok(Self {
1707            file,
1708            // The end of the file, so that the committed generation's catalog stays where its slot
1709            // says it is and keeps naming a file a reader can still open.
1710            at: size,
1711            dictionaries: fields
1712                .iter()
1713                .map(|field| (field.ty == LogicalType::Varchar).then(GlobalDictionary::new))
1714                .collect(),
1715            coded: fields
1716                .iter()
1717                .map(|field| AtomicBool::new(field.ty == LogicalType::Varchar))
1718                .collect(),
1719            gathers: fields.iter().map(|field| stats::Gather::new(&field.ty, generation)).collect(),
1720            table: Table {
1721                name,
1722                dictionaries: vec![None; fields.len()],
1723                dictionary_payloads: Vec::new(),
1724                distincts: vec![None; fields.len()],
1725                fields,
1726                stripes: Vec::new(),
1727                rows: 0,
1728                frequencies: Vec::new(),
1729                pair_frequencies: Vec::new(),
1730                frequency_texts: Vec::new(),
1731                host_groups: None,
1732                clustering: None,
1733                generation,
1734                sections: Vec::new(),
1735            },
1736            generation,
1737            order: Vec::new(),
1738            next_order: 0,
1739            pending: Vec::with_capacity(STRIPE_PARTS),
1740            closed,
1741            views,
1742            profile: None,
1743        })
1744    }
1745
1746    /// Creates a new v10 file and its first table.
1747    ///
1748    /// # Errors
1749    ///
1750    /// If the file exists, a field has no scalar encoding, or the path cannot be written.
1751    pub fn create(
1752        path: impl AsRef<Path>,
1753        name: impl Into<String>,
1754        fields: Vec<Field>,
1755    ) -> Result<Self> {
1756        for field in &fields {
1757            type_tag(&field.ty)?;
1758        }
1759        let file =
1760            OpenOptions::new().write(true).read(true).create_new(true).open(path).map_err(io)?;
1761        let mut header = [0; HEADER as usize];
1762        header[..8].copy_from_slice(MAGIC);
1763        header[8..12].copy_from_slice(&FORMAT.to_le_bytes());
1764        write_at(&file, 0, &header)?;
1765        Ok(Self {
1766            file,
1767            at: HEADER,
1768            dictionaries: fields
1769                .iter()
1770                .map(|field| (field.ty == LogicalType::Varchar).then(GlobalDictionary::new))
1771                .collect(),
1772            coded: fields
1773                .iter()
1774                .map(|field| AtomicBool::new(field.ty == LogicalType::Varchar))
1775                .collect(),
1776            gathers: fields.iter().map(|field| stats::Gather::new(&field.ty, 1)).collect(),
1777            table: Table {
1778                name: name.into(),
1779                dictionaries: vec![None; fields.len()],
1780                dictionary_payloads: Vec::new(),
1781                distincts: vec![None; fields.len()],
1782                fields,
1783                stripes: Vec::new(),
1784                rows: 0,
1785                frequencies: Vec::new(),
1786                pair_frequencies: Vec::new(),
1787                frequency_texts: Vec::new(),
1788                host_groups: None,
1789                clustering: None,
1790                generation: 1,
1791                sections: Vec::new(),
1792            },
1793            generation: 1,
1794            order: Vec::new(),
1795            next_order: 0,
1796            pending: Vec::with_capacity(STRIPE_PARTS),
1797            closed: Vec::new(),
1798            views: Vec::new(),
1799            profile: None,
1800        })
1801    }
1802
1803    /// Creates a new file that holds no table at all, committed and ready to open.
1804    ///
1805    /// A database somebody dropped the last table out of is still a database, and until this there
1806    /// was no way to write one down. Every other way into this file goes through a table, because
1807    /// [`Writer::create`] takes the first one and [`Writer::finish`] commits the one it is on, so a
1808    /// catalog with nothing in it could be read and not written. The format already allowed it: the
1809    /// catalog is a count and that many entries, and a count of nought encodes and decodes the same
1810    /// way every other count does, which is why nothing here is a version change.
1811    ///
1812    /// It hands back nothing rather than a writer, because a writer with no table is a writer with
1813    /// nothing to append to. A file that is going to hold a table is [`Writer::create`], and one
1814    /// that is going to have a table added to it later is [`Writer::open`], which reads what this
1815    /// wrote the same way it reads any other generation.
1816    ///
1817    /// It takes the views anyway, because a database with no table can still have views in it. A
1818    /// view over `range` or over another view names no table, so dropping the last table out of a
1819    /// database does not have to leave the catalog with nothing worth writing down.
1820    ///
1821    /// # Errors
1822    ///
1823    /// If the file exists or the path cannot be written.
1824    pub fn empty(path: impl AsRef<Path>, views: &[ViewEntry]) -> Result<()> {
1825        let file =
1826            OpenOptions::new().write(true).read(true).create_new(true).open(path).map_err(io)?;
1827        let mut header = [0; HEADER as usize];
1828        header[..8].copy_from_slice(MAGIC);
1829        header[8..12].copy_from_slice(&FORMAT.to_le_bytes());
1830        write_at(&file, 0, &header)?;
1831        let catalog = encode_catalog(&[], views)?;
1832        write_at(&file, HEADER, &catalog)?;
1833        // The same two syncs in the same order as [`Writer::finish`], and for the same reason. The
1834        // catalog is on the disk before the slot names it, so a file this is interrupted in the
1835        // middle of is a header with no valid slot rather than a slot pointing at nothing.
1836        file.sync_all().map_err(io)?;
1837        let slot = Slot {
1838            offset: HEADER,
1839            length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
1840            generation: 1,
1841            hash: checksum(&catalog),
1842        };
1843        write_at(&file, slot_offset(1), &slot.bytes())?;
1844        file.sync_all().map_err(io)?;
1845        Ok(())
1846    }
1847
1848    /// Closes the table this writer is on and starts another one in the same file.
1849    ///
1850    /// Nothing is published here. The closed table's directory is written so that the bytes are on
1851    /// disk and its span is known, and the catalog that names it is only written by
1852    /// [`Writer::finish`], so a crash between two tables leaves the previous generation intact.
1853    ///
1854    /// # Errors
1855    ///
1856    /// If the name repeats a table already closed, a field has no scalar encoding, or the table
1857    /// being closed cannot be written.
1858    pub fn next(mut self, name: impl Into<String>, fields: Vec<Field>) -> Result<Self> {
1859        for field in &fields {
1860            type_tag(&field.ty)?;
1861        }
1862        let name = name.into();
1863        let entry = self.close()?;
1864        if self.closed.iter().chain(std::iter::once(&entry)).any(|held| held.name == name) {
1865            return Err(invalid("two tables in one native file have the same name"));
1866        }
1867        let Self { file, at, generation, mut closed, views, .. } = self;
1868        closed.push(entry);
1869        Ok(Self {
1870            file,
1871            at,
1872            generation,
1873            closed,
1874            views,
1875            profile: None,
1876            dictionaries: fields
1877                .iter()
1878                .map(|field| (field.ty == LogicalType::Varchar).then(GlobalDictionary::new))
1879                .collect(),
1880            coded: fields
1881                .iter()
1882                .map(|field| AtomicBool::new(field.ty == LogicalType::Varchar))
1883                .collect(),
1884            gathers: fields.iter().map(|field| stats::Gather::new(&field.ty, generation)).collect(),
1885            table: Table {
1886                name,
1887                dictionaries: vec![None; fields.len()],
1888                dictionary_payloads: Vec::new(),
1889                distincts: vec![None; fields.len()],
1890                fields,
1891                stripes: Vec::new(),
1892                rows: 0,
1893                frequencies: Vec::new(),
1894                pair_frequencies: Vec::new(),
1895                frequency_texts: Vec::new(),
1896                host_groups: None,
1897                clustering: None,
1898                generation,
1899                sections: Vec::new(),
1900            },
1901            order: Vec::new(),
1902            next_order: 0,
1903            pending: Vec::with_capacity(STRIPE_PARTS),
1904        })
1905    }
1906
1907    /// Sets the views the next commit writes down, replacing whatever was carried forward.
1908    ///
1909    /// It replaces rather than adds because the caller has the whole catalog in front of it and the
1910    /// writer does not. A view that was dropped is a view that is not in the list any more, and
1911    /// there is no other way for the writer to hear about that, since nothing else it is told about
1912    /// mentions views at all.
1913    ///
1914    /// A writer that is never told anything writes back the views it read at [`Writer::open`], so a
1915    /// checkpoint that only had a table to append does not quietly drop them.
1916    #[must_use]
1917    pub fn with_views(mut self, views: Vec<ViewEntry>) -> Self {
1918        self.views = views;
1919        self
1920    }
1921
1922    /// Charges the stages this writer runs to `profile`.
1923    ///
1924    /// For the table being written now. [`Writer::next`] starts the next table without one,
1925    /// because a second table's stripes charged to the first table's load would be a profile of
1926    /// neither.
1927    #[must_use]
1928    pub fn with_profile(mut self, profile: Arc<LoadProfile>) -> Self {
1929        self.profile = Some(profile);
1930        self
1931    }
1932
1933    /// Records the order this table's rows are meant to be stored in.
1934    ///
1935    /// The declaration goes in the table directory and comes back out of
1936    /// [`Table::clustering`]. Nothing here sorts anything, and nothing here checks that the rows
1937    /// handed to [`Writer::append`] arrive in the order this claims. That is deliberate for now:
1938    /// the thing that was missing was a place to write the order down, and a loader that honours
1939    /// the declaration is the next piece rather than this one.
1940    ///
1941    /// The declaration applies to the table the writer is currently on, so it is set after
1942    /// [`Writer::next`] rather than once for the file.
1943    ///
1944    /// # Errors
1945    ///
1946    /// If the declaration names a column this table does not have.
1947    pub fn declare(mut self, clustering: Clustering) -> Result<Self> {
1948        // Rebuilt against this table's own column count rather than trusted, because the caller
1949        // built it against a catalog entry and the two could have drifted.
1950        self.table.clustering = Some(Clustering::new(
1951            clustering.columns().to_vec(),
1952            clustering.width(),
1953            &self.table.fields,
1954        )?);
1955        Ok(self)
1956    }
1957
1958    /// Appends bytes at the end of the file and moves the writer's own offset past them.
1959    ///
1960    /// Every write in here goes through this, so that [`Writer::at`] is the only answer to where
1961    /// anything is and the file's cursor is never consulted for it.
1962    fn put(&mut self, bytes: &[u8]) -> Result<()> {
1963        write_at(&self.file, self.at, bytes)?;
1964        self.at = self
1965            .at
1966            .checked_add(bytes.len() as u64)
1967            .ok_or_else(|| invalid("native file length overflow"))?;
1968        Ok(())
1969    }
1970
1971    /// Writes one chunk as independently readable column pages.
1972    ///
1973    /// # Errors
1974    ///
1975    /// If its width or types differ from the declared table, or a page exceeds its bound.
1976    pub fn append(&mut self, chunk: &Chunk) -> Result<()> {
1977        let order = (self.next_order, 0);
1978        self.next_order = self.next_order.saturating_add(1);
1979        self.append_at(order, chunk)
1980    }
1981
1982    /// Writes one chunk and records its source position for directory ordering.
1983    ///
1984    /// Pages may be encoded by parallel pipeline instances and reach the file in completion order.
1985    /// The stripe they land in is sorted by this key at commit, and [`Self::finish`] rejects a
1986    /// sequence whose parts do not come out in source order once the stripes are sorted, because a
1987    /// stripe groups whatever arrived together and cannot put a late part back where it belongs.
1988    ///
1989    /// # Errors
1990    ///
1991    /// The same as [`Self::append`].
1992    pub fn append_at(&mut self, order: (u64, u64), chunk: &Chunk) -> Result<()> {
1993        if chunk.is_empty() {
1994            return Ok(());
1995        }
1996        self.admit(chunk)?;
1997        if self.pending.last().is_some_and(|last| last.order > order) {
1998            self.flush_pending()?;
1999        }
2000        // Cloned rather than encoded, and a clone of a chunk that owns its buffers is a copy of
2001        // them. Sixty four parts of a hundred and five columns is tens of megabytes held for the
2002        // length of a stripe and a few seconds of memory traffic over a whole ClickBench load,
2003        // against the hundreds of seconds of encode this is what lets off one thread.
2004        self.pending.push(PendingChunk { order, chunk: chunk.clone() });
2005        if self.pending.len() == STRIPE_PARTS {
2006            self.flush_pending()?;
2007        }
2008        Ok(())
2009    }
2010
2011    /// Writes a run of chunks as one stripe of its own.
2012    ///
2013    /// [`Self::append_at`] decides where a stripe ends by watching the orders go past, which works
2014    /// when one caller hands over every chunk in source order and does not when several do. A
2015    /// writer being fed by more than one pipeline instance sees the orders interleave, and a stripe
2016    /// that ends every time two of them cross is a stripe of one or two parts.
2017    ///
2018    /// So the grouping moves to the caller. Whoever is buffering hands over a run it already knows
2019    /// is contiguous and in order, and gets a stripe holding exactly that run. The orders still
2020    /// have to come out in source order once the stripes are sorted, which [`Self::finish`] checks,
2021    /// so the runs from different callers may interleave with each other but may not overlap.
2022    ///
2023    /// # Errors
2024    ///
2025    /// The same as [`Self::append`], and if the run is longer than [`STRIPE_PARTS`].
2026    pub fn append_stripe(&mut self, parts: Vec<((u64, u64), Chunk)>) -> Result<()> {
2027        if parts.len() > STRIPE_PARTS {
2028            return Err(invalid("a stripe was handed more parts than it holds"));
2029        }
2030        // Whatever an earlier caller left behind is its own stripe rather than the front of this
2031        // one, because the two runs are from different places in the source and a stripe is a run.
2032        self.flush_pending()?;
2033        for (order, chunk) in parts {
2034            if chunk.is_empty() {
2035                continue;
2036            }
2037            self.admit(&chunk)?;
2038            self.pending.push(PendingChunk { order, chunk });
2039        }
2040        self.flush_pending()
2041    }
2042
2043    /// Checks a chunk against the declared table and counts its rows in.
2044    fn admit(&mut self, chunk: &Chunk) -> Result<()> {
2045        if chunk.width() != self.table.fields.len() {
2046            return Err(invalid("chunk width differs from table schema"));
2047        }
2048        for (index, field) in self.table.fields.iter().enumerate() {
2049            if chunk.column(index)?.logical_type() != &field.ty {
2050                return Err(invalid("chunk type differs from table schema"));
2051            }
2052        }
2053        self.table.rows = self
2054            .table
2055            .rows
2056            .checked_add(chunk.len())
2057            .ok_or_else(|| invalid("row count overflow"))?;
2058        Ok(())
2059    }
2060
2061    /// One column's parts of a stripe as pages, for a column with no global dictionary.
2062    fn encode_pages(columns: &[&Vector]) -> Result<ColumnStripe> {
2063        let mut stripe = ColumnStripe {
2064            pages: Vec::with_capacity(columns.len()),
2065            codes: Vec::with_capacity(columns.len()),
2066            sieves: Vec::with_capacity(columns.len()),
2067            ranges: Vec::with_capacity(columns.len()),
2068        };
2069        let mut settling = Settling::default();
2070        for &column in columns {
2071            let bytes = encode(column, &mut settling)?;
2072            if bytes.len() > MAX_PAGE {
2073                return Err(invalid("column page exceeds the configured bound"));
2074            }
2075            // The range is built first because the sieve reads it rather than walking the column a
2076            // second time to find out how wide it is.
2077            let range = Range::of(column);
2078            // A sieve at least as large as the part it indexes is not written. A reader reads the
2079            // sieve to decide whether to read the part, so when the sieve is the larger of the two
2080            // it has already spent more than the read it is trying to avoid, and that holds even if
2081            // it rejects every time. It is a necessary condition rather than the whole rule, which
2082            // is that a sieve pays when its bytes are under the rejection rate times the part's,
2083            // but the rejection rate depends on what a query probes for and the writer does not
2084            // know that. The necessary half needs two numbers that are both in hand here.
2085            //
2086            // A column with a global dictionary gets none, because it already has an exact
2087            // membership index per stripe. Those do not come through here. See [`prepare`].
2088            let sieve =
2089                Sieve::of(column, &range, SIEVE_BUDGET).filter(|sieve| sieve.len() < bytes.len());
2090            stripe.pages.push(bytes);
2091            stripe.codes.push(None);
2092            stripe.sieves.push(sieve);
2093            stripe.ranges.push(range);
2094        }
2095        Ok(stripe)
2096    }
2097
2098    /// Writes every encoded dictionary block that is not in the file yet and forgets its bytes.
2099    ///
2100    /// This is what keeps a load from holding its dictionaries' payload. The blocks land between
2101    /// stripes wherever the writer is, which is fine because the index says where each one is.
2102    fn place_blocks(&mut self) -> Result<()> {
2103        let mut dictionaries = std::mem::take(&mut self.dictionaries);
2104        let placed = dictionaries.iter_mut().flatten().try_for_each(|dictionary| {
2105            for block in std::mem::take(&mut dictionary.blocks) {
2106                let start = self.at;
2107                self.put(&block)?;
2108                dictionary.placed.push(Placed {
2109                    start,
2110                    length: block.len() as u64,
2111                    hash: checksum(&block),
2112                });
2113            }
2114            Ok(())
2115        });
2116        self.dictionaries = dictionaries;
2117        placed
2118    }
2119
2120    /// Writes the buffered parts as one stripe, each column's parts contiguous on disk.
2121    ///
2122    /// The same four steps a caller holding this writer behind a lock takes, with nobody else
2123    /// waiting between them. See [`prepare`].
2124    fn flush_pending(&mut self) -> Result<()> {
2125        if self.pending.is_empty() {
2126            return Ok(());
2127        }
2128        let held = std::mem::take(&mut self.pending);
2129        let prepared = self.preparer().prepare_held(held)?;
2130        let merged = self.merge_held(prepared)?;
2131        let paged = merged.pages()?;
2132        self.write_paged(paged)
2133    }
2134
2135    /// Writes one stripe whose pages are built, each column's parts contiguous on disk.
2136    fn write_stripe(&mut self, held: &[Part], encoded: Vec<ColumnStripe>) -> Result<()> {
2137        let width = self.table.fields.len();
2138        let parts = held.len();
2139        if encoded.len() != width {
2140            return Err(Error::internal("a stripe came to the writer with the wrong columns"));
2141        }
2142        let profile = self.profile.clone();
2143        if let Some(profile) = &profile {
2144            let rows = held.iter().map(|part| part.rows as u64).sum();
2145            let raw = held.iter().map(|part| part.footprint as u64).sum();
2146            let pages =
2147                encoded.iter().flat_map(|stripe| &stripe.pages).map(|page| page.len() as u64).sum();
2148            profile.moved(Stage::Pages, raw, pages, rows);
2149        }
2150        // Before a byte of the stripe is written, because the raw bytes this frees are the bytes the
2151        // load peaks on and the threads it uses are idle between here and the next chunk arriving.
2152        let timing = profile.as_deref().map(|profile| profile.span(Stage::Dictionary));
2153        let before = self.at;
2154        encode_ready(&mut self.dictionaries)?;
2155        self.place_blocks()?;
2156        drop(timing);
2157        if let Some(profile) = &profile {
2158            profile.moved(Stage::Dictionary, 0, self.at - before, 0);
2159        }
2160        let timing = profile.as_deref().map(|profile| profile.span(Stage::Write));
2161        let before = self.at;
2162        let mut pages = Vec::with_capacity(width);
2163        let mut memberships = vec![None; width];
2164        let mut ranges = Vec::with_capacity(width);
2165        let mut index = Vec::with_capacity(width.saturating_mul(index_section(parts)?));
2166        for stripe in &encoded {
2167            let offset = self.at;
2168            let section = index.len();
2169            let mut length = 0_usize;
2170            for bytes in &stripe.pages {
2171                write_at(&self.file, self.at + length as u64, bytes)?;
2172                put_u32(
2173                    &mut index,
2174                    u32::try_from(bytes.len()).map_err(|_| invalid("part length overflow"))?,
2175                );
2176                put_u64(&mut index, checksum(bytes));
2177                length = length
2178                    .checked_add(bytes.len())
2179                    .ok_or_else(|| invalid("column page length overflow"))?;
2180            }
2181            let hash = checksum(&index[section..]);
2182            put_u64(&mut index, hash);
2183            if length > MAX_PAGE {
2184                return Err(invalid("column page exceeds the configured bound"));
2185            }
2186            self.at = self
2187                .at
2188                .checked_add(length as u64)
2189                .ok_or_else(|| invalid("native file length overflow"))?;
2190            pages.push(Span {
2191                offset,
2192                length: u32::try_from(length).map_err(|_| invalid("page length overflow"))?,
2193            });
2194            ranges.push(merged_range(stripe.ranges.iter().cloned()));
2195        }
2196        for (membership, stripe) in memberships.iter_mut().zip(&encoded) {
2197            if stripe.codes.iter().all(Option::is_none) {
2198                continue;
2199            }
2200            let lists = stripe
2201                .codes
2202                .iter()
2203                .map(|codes| codes.clone().unwrap_or_default())
2204                .collect::<Vec<_>>();
2205            let bytes = encode_membership(&merged_codes(lists));
2206            let offset = self.at;
2207            self.put(&bytes)?;
2208            *membership = Some(Page {
2209                offset,
2210                length: u32::try_from(bytes.len())
2211                    .map_err(|_| invalid("membership page length overflow"))?,
2212                hash: checksum(&bytes),
2213            });
2214        }
2215        let mut sieves = vec![None; width];
2216        for (page, stripe) in sieves.iter_mut().zip(&encoded) {
2217            if stripe.sieves.iter().all(Option::is_none) {
2218                continue;
2219            }
2220            let bytes = encode_sieves(stripe.sieves.iter())?;
2221            let offset = self.at;
2222            self.put(&bytes)?;
2223            *page = Some(Page {
2224                offset,
2225                length: u32::try_from(bytes.len())
2226                    .map_err(|_| invalid("sieve page length overflow"))?,
2227                hash: checksum(&bytes),
2228            });
2229        }
2230        // A stripe of one part has the same rows in it as that part, so its own bounds are already
2231        // the part's and a page here would say what the directory says. Everywhere else the page is
2232        // written unless it comes to more than the column it indexes, which is the rule the sieves
2233        // go by and for the same reason: a reader reads this to decide whether to read the column,
2234        // so a page larger than the column has spent more than the read it is avoiding.
2235        let mut part_ranges = vec![None; width];
2236        if parts > 1 {
2237            for ((page, stripe), span) in part_ranges.iter_mut().zip(&encoded).zip(&pages) {
2238                let bytes = encode_part_ranges(&stripe.ranges)?;
2239                if bytes.len() >= span.length as usize {
2240                    continue;
2241                }
2242                let offset = self.at;
2243                self.put(&bytes)?;
2244                *page = Some(Page {
2245                    offset,
2246                    length: u32::try_from(bytes.len())
2247                        .map_err(|_| invalid("part range page length overflow"))?,
2248                    hash: checksum(&bytes),
2249                });
2250            }
2251        }
2252        let offset = self.at;
2253        self.put(&index)?;
2254        let index = Span {
2255            offset,
2256            length: u32::try_from(index.len())
2257                .map_err(|_| invalid("index page length overflow"))?,
2258        };
2259        let mut rows = 0_usize;
2260        let mut lengths = Vec::with_capacity(parts);
2261        let mut span = None;
2262        for part in held {
2263            rows = rows.checked_add(part.rows).ok_or_else(|| invalid("row count overflow"))?;
2264            lengths.push(u32::try_from(part.rows).map_err(|_| invalid("part row count overflow"))?);
2265            span = Some(span.map_or((part.order, part.order), |(first, _)| (first, part.order)));
2266        }
2267        self.order.push(span.ok_or_else(|| invalid("a stripe was flushed with no parts"))?);
2268        self.table.stripes.push(Stripe {
2269            rows,
2270            parts: lengths,
2271            index,
2272            pages,
2273            memberships: Pages::from_slots(memberships)?,
2274            sieves: Pages::from_slots(sieves)?,
2275            part_ranges: Pages::from_slots(part_ranges)?,
2276            zone: Zone::from_ranges(ranges),
2277        });
2278        drop(timing);
2279        if let Some(profile) = &profile {
2280            profile.moved(Stage::Write, 0, self.at - before, rows as u64);
2281        }
2282        Ok(())
2283    }
2284
2285    /// Finds exact heavy hitters without keeping a hash table for every numeric column while the
2286    /// load is live. The pages are already in the target file, so one column at a time uses a
2287    /// bounded Misra-Gries candidate table and then recounts only those candidates.
2288    ///
2289    /// The first of those passes also counts the column's distinct values exactly, up to the cap in
2290    /// [`distinct`], which is the number a string column gets from its dictionary. It comes back
2291    /// beside the summary because a column whose heavy hitters cannot be proved can still have been
2292    /// counted.
2293    ///
2294    /// The tables are keyed by a value's sixty four bits rather than by [`FrequencyValue`], and a
2295    /// null is counted beside them. Every integer type the format stores fits in those bits, so
2296    /// within one column two values share bits only if they are the same value, and a sixteen byte
2297    /// entry keeps the whole candidate table in the second level cache where the forty eight byte
2298    /// one did not. The null takes part in the candidate table exactly as a key would: it holds a
2299    /// place while its count is above zero, and it is decremented with the rest.
2300    fn numeric_frequency(&self, column: usize) -> Result<(Option<FrequencySummary>, Option<u64>)> {
2301        let signed = match self.table.fields[column].ty {
2302            LogicalType::TinyInt
2303            | LogicalType::SmallInt
2304            | LogicalType::Integer
2305            | LogicalType::BigInt
2306            | LogicalType::Date
2307            | LogicalType::Timestamp => true,
2308            LogicalType::UTinyInt
2309            | LogicalType::USmallInt
2310            | LogicalType::UInteger
2311            | LogicalType::UBigInt => false,
2312            _ => return Ok((None, None)),
2313        };
2314        let value_of = |bits: Option<u64>| match bits {
2315            None => FrequencyValue::Null,
2316            Some(bits) if signed => FrequencyValue::Integer(i128::from(bits as i64)),
2317            Some(bits) => FrequencyValue::Integer(i128::from(bits)),
2318        };
2319        let mut candidates: FrequencyMap<u32> = FrequencyMap::default();
2320        let mut nulls = 0_u32;
2321        let mut decrements = 0_u64;
2322        let mut distinct = distinct::ExactDistinct::new();
2323        self.visit_numeric(column, signed, |_, bits| {
2324            let held = match bits {
2325                Some(bits) => {
2326                    distinct.insert(bits);
2327                    candidates.get_mut(&bits)
2328                }
2329                None if nulls != 0 => Some(&mut nulls),
2330                None => None,
2331            };
2332            if let Some(count) = held {
2333                *count = count.saturating_add(1);
2334            } else if candidates.len() + usize::from(nulls != 0) < FREQUENCY_CANDIDATES {
2335                match bits {
2336                    Some(bits) => {
2337                        candidates.insert(bits, 1);
2338                    }
2339                    None => nulls = 1,
2340                }
2341            } else {
2342                candidates.retain(|_, count| {
2343                    *count -= 1;
2344                    *count != 0
2345                });
2346                nulls = nulls.saturating_sub(1);
2347                decrements = decrements.saturating_add(1);
2348            }
2349        })?;
2350        let (exact, null_count) = if decrements == 0 {
2351            let exact = candidates
2352                .into_iter()
2353                .map(|(bits, count)| (bits, u64::from(count)))
2354                .collect::<FrequencyMap<_>>();
2355            (exact, (nulls != 0).then_some(u64::from(nulls)))
2356        } else {
2357            let mut lower = candidates.values().copied().collect::<Vec<_>>();
2358            if nulls != 0 {
2359                lower.push(nulls);
2360            }
2361            lower.sort_unstable_by(|left, right| right.cmp(left));
2362            if lower.len() < FREQUENCY_BUILD_RANK
2363                || u64::from(lower[FREQUENCY_BUILD_RANK - 1]) <= decrements
2364            {
2365                return Ok((None, distinct.count()));
2366            }
2367            let mut exact =
2368                candidates.into_keys().map(|bits| (bits, 0_u64)).collect::<FrequencyMap<_>>();
2369            let mut null_count = (nulls != 0).then_some(0_u64);
2370            self.visit_numeric(column, signed, |_, bits| {
2371                let held = match bits {
2372                    Some(bits) => exact.get_mut(&bits),
2373                    None => null_count.as_mut(),
2374                };
2375                if let Some(count) = held {
2376                    *count = count.saturating_add(1);
2377                }
2378            })?;
2379            (exact, null_count)
2380        };
2381        let mut entries = exact
2382            .into_iter()
2383            .map(|(bits, count)| FrequencyEntry { value: value_of(Some(bits)), count })
2384            .chain(null_count.map(|count| FrequencyEntry { value: FrequencyValue::Null, count }))
2385            .collect::<Vec<_>>();
2386        let omitted_max = keep_most_frequent(&mut entries).max(decrements);
2387        let kept_rows = entries.iter().try_fold(0_u64, |total, entry| {
2388            total.checked_add(entry.count).filter(|&total| total <= FREQUENCY_ORDINALS as u64)
2389        });
2390        let mut ordinals = Vec::new();
2391        let mut ordinal_entries = Vec::new();
2392        if let Some(kept_rows) = kept_rows {
2393            let mut kept = FrequencyMap::default();
2394            let mut null_kept = None;
2395            for (at, entry) in entries.iter().enumerate() {
2396                let at = u16::try_from(at)
2397                    .map_err(|_| invalid("too many retained frequency entries"))?;
2398                match entry.value {
2399                    FrequencyValue::Integer(value) => {
2400                        kept.insert(value as u64, at);
2401                    }
2402                    FrequencyValue::Null => null_kept = Some(at),
2403                    FrequencyValue::Code(_) => {}
2404                }
2405            }
2406            ordinals.reserve(usize::try_from(kept_rows).unwrap_or(FREQUENCY_ORDINALS));
2407            ordinal_entries.reserve(usize::try_from(kept_rows).unwrap_or(FREQUENCY_ORDINALS));
2408            self.visit_numeric(column, signed, |ordinal, bits| {
2409                let held = match bits {
2410                    Some(bits) => kept.get(&bits).copied(),
2411                    None => null_kept,
2412                };
2413                if let Some(entry) = held {
2414                    ordinals.push(ordinal);
2415                    ordinal_entries.push(entry);
2416                }
2417            })?;
2418        }
2419        Ok((
2420            Some(FrequencySummary { entries, omitted_max, ordinals, ordinal_entries }),
2421            distinct.count(),
2422        ))
2423    }
2424
2425    /// Hands every row of an integer column to `visit` as its ordinal and its sixty four bits, or
2426    /// `None` for a null.
2427    ///
2428    /// `signed` says which of the two readings the column has. A packed unsigned column would come
2429    /// back from `signed_block` as a base plus a code in `i64`, which wraps for a value past the top
2430    /// of `BIGINT`, so only a signed column takes the block path.
2431    fn visit_numeric(
2432        &self,
2433        column: usize,
2434        signed: bool,
2435        mut visit: impl FnMut(u64, Option<u64>),
2436    ) -> Result<()> {
2437        let ty = &self.table.fields[column].ty;
2438        let mut start = 0_u64;
2439        let mut block = Vec::new();
2440        for stripe in &self.table.stripes {
2441            let spans = read_index(&self.file, stripe, column)?;
2442            let page = stripe.pages[column];
2443            let mut bytes = vec![0; page.length as usize];
2444            read_at(&self.file, page.offset, &mut bytes)?;
2445            for (span, &rows) in spans.iter().zip(&stripe.parts) {
2446                let part = part_bytes(&bytes, *span)?;
2447                if checksum(part) != span.hash {
2448                    return Err(invalid("column page checksum differs while building frequencies"));
2449                }
2450                let rows = rows as usize;
2451                let vector = decode(ty, rows, part, None)?;
2452                // Every signed layout a numeric column decodes to, which is every column of `hits`,
2453                // comes out as one run of `i64` and is walked as a slice. The row path below is for
2454                // the unsigned types and anything else that cannot be handed over that way.
2455                if signed && vector.signed_block(&mut block) && block.len() == rows {
2456                    if vector.none_null() {
2457                        for (row, &value) in block.iter().enumerate() {
2458                            visit(start.saturating_add(row as u64), Some(value as u64));
2459                        }
2460                    } else {
2461                        for (row, &value) in block.iter().enumerate() {
2462                            let bits = (!vector.is_null_at(row)).then_some(value as u64);
2463                            visit(start.saturating_add(row as u64), bits);
2464                        }
2465                    }
2466                    start = start.saturating_add(rows as u64);
2467                    continue;
2468                }
2469                // row at a time: frequency construction visits decoded values to update bounded candidates.
2470                for row in 0..rows {
2471                    let bits = if vector.is_null_at(row) {
2472                        None
2473                    } else {
2474                        // An unsigned column has no signed reading, and the documented fallback is
2475                        // the value itself. Every width the format stores fits in sixty four bits,
2476                        // so nothing is lost on the way through.
2477                        let widened = match vector.signed_at(row) {
2478                            Some(value) => Some(value as u64),
2479                            None => match vector.value_at(row) {
2480                                Value::UTinyInt(value) => Some(u64::from(value)),
2481                                Value::USmallInt(value) => Some(u64::from(value)),
2482                                Value::UInteger(value) => Some(u64::from(value)),
2483                                Value::UBigInt(value) => Some(value),
2484                                _ => None,
2485                            },
2486                        };
2487                        Some(widened.ok_or_else(|| {
2488                            invalid("numeric frequency page did not contain an integer value")
2489                        })?)
2490                    };
2491                    visit(start.saturating_add(row as u64), bits);
2492                }
2493                start = start.saturating_add(rows as u64);
2494            }
2495        }
2496        Ok(())
2497    }
2498
2499    /// Builds independent numeric synopses concurrently after all column pages are committed.
2500    ///
2501    /// The columns go through a queue rather than being cut into equal runs, because they are not
2502    /// equally expensive and they are not shuffled. A `BIGINT` column carries eight times the bytes
2503    /// of a `TINYINT` through the decode, and a run of them sits together in a schema the way it
2504    /// sits together in `hits`, so a worker that was handed the wrong six columns finishes long
2505    /// after one that was handed the right six and the whole phase waits for it.
2506    fn numeric_frequencies(&self) -> Result<Vec<(Option<FrequencySummary>, Option<u64>)>> {
2507        let mut columns = self
2508            .table
2509            .fields
2510            .iter()
2511            .enumerate()
2512            .filter_map(|(column, field)| {
2513                matches!(
2514                    field.ty,
2515                    LogicalType::TinyInt
2516                        | LogicalType::SmallInt
2517                        | LogicalType::Integer
2518                        | LogicalType::BigInt
2519                        | LogicalType::UTinyInt
2520                        | LogicalType::USmallInt
2521                        | LogicalType::UInteger
2522                        | LogicalType::UBigInt
2523                        | LogicalType::Date
2524                        | LogicalType::Timestamp
2525                )
2526                .then_some(column)
2527            })
2528            .collect::<Vec<_>>();
2529        let workers = std::thread::available_parallelism()
2530            .map_or(1, usize::from)
2531            .min(MAX_FREQUENCY_WORKERS)
2532            .min(columns.len());
2533        let profile = self.profile.as_deref();
2534        if workers <= 1 {
2535            let _timing = profile.map(|profile| profile.span(Stage::Publish));
2536            let mut frequencies = vec![(None, None); self.table.fields.len()];
2537            for column in columns {
2538                frequencies[column] = self.numeric_frequency(column)?;
2539            }
2540            return Ok(frequencies);
2541        }
2542        // Popped from the back, so the expensive columns are the ones taken first and the cheap ones
2543        // are what is left to fill in behind them.
2544        columns.sort_by_key(|&column| weight(&self.table.fields[column].ty));
2545        let queue = Mutex::new(columns);
2546        let pieces = std::thread::scope(|scope| {
2547            (0..workers)
2548                .map(|_| {
2549                    scope.spawn(|| {
2550                        let _timing = profile.map(|profile| profile.span(Stage::Publish));
2551                        let mut mine = Vec::new();
2552                        loop {
2553                            let taken = queue
2554                                .lock()
2555                                .map_err(|_| Error::internal("a native frequency worker panicked"))?
2556                                .pop();
2557                            let Some(column) = taken else { break };
2558                            mine.push((column, self.numeric_frequency(column)?));
2559                        }
2560                        Ok(mine)
2561                    })
2562                })
2563                .collect::<Vec<_>>()
2564                .into_iter()
2565                .map(|handle| {
2566                    handle
2567                        .join()
2568                        .map_err(|_| Error::internal("a native frequency worker panicked"))?
2569                })
2570                .collect::<Result<Vec<_>>>()
2571        })?;
2572        let mut frequencies = vec![(None, None); self.table.fields.len()];
2573        for piece in pieces {
2574            for (column, summary) in piece {
2575                frequencies[column] = summary;
2576            }
2577        }
2578        Ok(frequencies)
2579    }
2580
2581    /// Reads one stable dictionary code column only at sorted table-wide row ordinals.
2582    fn stable_codes_at(&self, column: usize, ordinals: &[u64]) -> Result<Option<Vec<Option<u32>>>> {
2583        if self.dictionaries.get(column).and_then(Option::as_ref).is_none() {
2584            return Ok(None);
2585        }
2586        if ordinals.windows(2).any(|pair| pair[0] >= pair[1]) {
2587            return Err(invalid("frequency ordinals are not sorted and unique"));
2588        }
2589        let mut out = Vec::with_capacity(ordinals.len());
2590        let mut wanted = 0;
2591        let mut stripe_start = 0_u64;
2592        for stripe in &self.table.stripes {
2593            let stripe_end = stripe_start.saturating_add(stripe.rows as u64);
2594            if wanted == ordinals.len() || ordinals[wanted] >= stripe_end {
2595                stripe_start = stripe_end;
2596                continue;
2597            }
2598            let spans = read_index(&self.file, stripe, column)?;
2599            let page = stripe.pages[column];
2600            let mut bytes = vec![0; page.length as usize];
2601            read_at(&self.file, page.offset, &mut bytes)?;
2602            let mut part_start = stripe_start;
2603            for (span, &rows) in spans.iter().zip(&stripe.parts) {
2604                let part_end = part_start.saturating_add(u64::from(rows));
2605                if wanted < ordinals.len() && ordinals[wanted] < part_end {
2606                    let part = part_bytes(&bytes, *span)?;
2607                    if checksum(part) != span.hash {
2608                        return Err(invalid(
2609                            "column page checksum differs while building pair frequencies",
2610                        ));
2611                    }
2612                    let upto = ordinals.partition_point(|&ordinal| ordinal < part_end);
2613                    let positions = ordinals[wanted..upto]
2614                        .iter()
2615                        .map(|&ordinal| {
2616                            usize::try_from(ordinal.saturating_sub(part_start))
2617                                .map_err(|_| invalid("frequency row offset does not fit in memory"))
2618                        })
2619                        .collect::<Result<Vec<_>>>()?;
2620                    if !decode_selected_stable_codes(rows as usize, part, &positions, &mut out)? {
2621                        return Ok(None);
2622                    }
2623                    wanted = upto;
2624                }
2625                part_start = part_end;
2626            }
2627            stripe_start = stripe_end;
2628        }
2629        if wanted != ordinals.len() {
2630            return Err(invalid("frequency ordinal is outside the table"));
2631        }
2632        Ok(Some(out))
2633    }
2634
2635    /// Derives bounded two-key leaders from numeric anchor ordinals and stable string codes.
2636    fn pair_frequencies(&self) -> Result<Vec<PairFrequencySummary>> {
2637        let anchors = self
2638            .table
2639            .frequencies
2640            .iter()
2641            .enumerate()
2642            .filter_map(|(column, summary)| {
2643                // A writer holds every synopsis it counted, so there is nothing stored to skip.
2644                match summary {
2645                    Some(Frequencies::Held(summary)) => Some(summary),
2646                    _ => None,
2647                }
2648                .filter(|summary| {
2649                    !summary.ordinals.is_empty()
2650                        && summary.ordinal_entries.len() == summary.ordinals.len()
2651                })
2652                .cloned()
2653                .map(|summary| (column, summary))
2654            })
2655            .collect::<Vec<_>>();
2656        let strings = self
2657            .dictionaries
2658            .iter()
2659            .enumerate()
2660            .filter_map(|(column, dictionary)| dictionary.as_ref().map(|_| column))
2661            .collect::<Vec<_>>();
2662        let mut summaries = Vec::new();
2663        for (first, anchors) in anchors {
2664            for &second in &strings {
2665                if summaries.len() == MAX_PAIR_FREQUENCIES {
2666                    return Ok(summaries);
2667                }
2668                let Some(codes) = self.stable_codes_at(second, &anchors.ordinals)? else {
2669                    continue;
2670                };
2671                if codes.len() != anchors.ordinal_entries.len() {
2672                    return Err(invalid("pair frequency columns have different lengths"));
2673                }
2674                let mut counts = HashMap::<(u16, Option<u32>), u64>::new();
2675                for (&anchor, code) in anchors.ordinal_entries.iter().zip(codes) {
2676                    *counts.entry((anchor, code)).or_default() += 1;
2677                }
2678                let mut entries = counts
2679                    .into_iter()
2680                    .map(|((first_entry, second), count)| PairFrequencyEntry {
2681                        first_entry,
2682                        second,
2683                        count,
2684                    })
2685                    .collect::<Vec<_>>();
2686                entries.sort_unstable_by(|left, right| {
2687                    right
2688                        .count
2689                        .cmp(&left.count)
2690                        .then_with(|| left.first_entry.cmp(&right.first_entry))
2691                        .then_with(|| left.second.cmp(&right.second))
2692                });
2693                let pair_omitted = entries.get(FREQUENCY_ENTRIES).map_or(0, |entry| entry.count);
2694                entries.truncate(FREQUENCY_ENTRIES);
2695                summaries.push(PairFrequencySummary {
2696                    first: u16::try_from(first)
2697                        .map_err(|_| invalid("pair frequency column index overflows"))?,
2698                    second: u16::try_from(second)
2699                        .map_err(|_| invalid("pair frequency column index overflows"))?,
2700                    entries,
2701                    omitted_max: anchors.omitted_max.max(pair_omitted),
2702                });
2703            }
2704        }
2705        Ok(summaries)
2706    }
2707
2708    /// Writes the directory of the table this writer is on and says where it went.
2709    ///
2710    /// Everything [`Writer::finish`] used to do except the two writes that publish. Pulling it out
2711    /// is what lets a second table follow a first: the bytes of a closed table are complete and
2712    /// addressable while nothing yet points at them, and the pointer is the last write of the
2713    /// commit.
2714    ///
2715    /// # Errors
2716    ///
2717    /// If directory encoding or writing fails.
2718    fn close(&mut self) -> Result<Entry> {
2719        self.flush_pending()?;
2720        // The rest of a table is its statistics, its dictionaries and its directory. The dictionary
2721        // work is charged as its own stage, because ranking a global dictionary can be most of what
2722        // this costs, and the rest as publish.
2723        let profile = self.profile.clone();
2724        let timing = profile.as_deref().map(|profile| profile.span(Stage::Publish));
2725        let before = self.at;
2726        let mut stripes = std::mem::take(&mut self.order)
2727            .into_iter()
2728            .zip(std::mem::take(&mut self.table.stripes))
2729            .collect::<Vec<_>>();
2730        stripes.sort_by_key(|(order, _)| order.0);
2731        let mut previous: Option<(u64, u64)> = None;
2732        for ((first, last), _) in &stripes {
2733            if previous.is_some_and(|previous| previous >= *first) {
2734                return Err(invalid("chunks did not arrive in source order"));
2735            }
2736            previous = Some(*last);
2737        }
2738        self.table.stripes = stripes.into_iter().map(|(_, stripe)| stripe).collect();
2739        // The frequencies charge themselves, one span to each thread that counts, because they run
2740        // on threads of their own and a span on this one would see their wall time and none of
2741        // their CPU.
2742        drop(timing);
2743        let (frequencies, distincts): (Vec<Option<FrequencySummary>>, _) =
2744            self.numeric_frequencies()?.into_iter().unzip();
2745        self.table.frequencies =
2746            frequencies.into_iter().map(|held| held.map(Frequencies::Held)).collect();
2747        self.table.distincts = distincts;
2748        let timing = profile.as_deref().map(|profile| profile.span(Stage::Dictionary));
2749        let placing = self.at;
2750        finish_dictionaries(&mut self.dictionaries)?;
2751        self.place_blocks()?;
2752        self.table.pair_frequencies = self.pair_frequencies()?;
2753        let dictionaries = std::mem::take(&mut self.dictionaries);
2754        self.table.dictionary_payloads = vec![0; self.table.fields.len()];
2755        self.table.frequency_texts = vec![Vec::new(); self.table.fields.len()];
2756        self.table.host_groups = None;
2757        // One column at a time, and every column's values dropped before the next column's are read
2758        // back. Sorting the columns across threads is the obvious thing and was what this did, but
2759        // sorting a column now means decoding it, and five ClickBench string columns decoded at once
2760        // is the peak this change is about.
2761        for (index, dictionary) in dictionaries.into_iter().enumerate() {
2762            let Some(dictionary) = dictionary else { continue };
2763            let (order, flat, bases) = dictionary.ranked_with_values(Some(&self.file))?;
2764            // A code nothing counted is a code no non-null row of this column holds, which is the
2765            // empty string a null was written as and nothing else, because a code is only ever made
2766            // by a row asking for one.
2767            self.table.distincts[index] =
2768                Some(dictionary.counts.iter().filter(|count| **count != 0).count() as u64);
2769            let (frequencies, texts) = code_frequency(&dictionary, &flat, &bases)?;
2770            self.table.frequencies[index] = Some(Frequencies::Held(frequencies));
2771            self.table.frequency_texts[index] = texts;
2772            if self.table.fields[index].name.eq_ignore_ascii_case("Referer") {
2773                self.table.host_groups = host::build(index, &dictionary, &flat, &bases)?;
2774            }
2775            drop(flat);
2776            drop(bases);
2777            let encoded = encode_global_dictionary(&dictionary, &order, &dictionary.placed, true)?;
2778            drop(order);
2779            let offset = self.at;
2780            self.put(&encoded.index)?;
2781            self.put(&encoded.ranks)?;
2782            self.put(&encoded.grams)?;
2783            self.table.dictionary_payloads[index] = dictionary
2784                .placed
2785                .iter()
2786                .try_fold(0_u64, |sum, place| sum.checked_add(place.length))
2787                .ok_or_else(|| invalid("global dictionary payload overflow"))?;
2788            let length = encoded
2789                .index
2790                .len()
2791                .checked_add(encoded.ranks.len())
2792                .and_then(|len| len.checked_add(encoded.grams.len()))
2793                .ok_or_else(|| invalid("dictionary page length overflow"))?;
2794            self.table.dictionaries[index] = Some(Page {
2795                offset,
2796                length: u32::try_from(length)
2797                    .map_err(|_| invalid("dictionary page length overflow"))?,
2798                hash: checksum(&encoded.index),
2799            });
2800        }
2801        drop(timing);
2802        let timing = profile.as_deref().map(|profile| profile.span(Stage::Publish));
2803        let placed = self.at - placing;
2804        self.write_stats()?;
2805        let directory = encode_directory(&self.table)?;
2806        if directory.len() > MAX_DIRECTORY {
2807            return Err(invalid("directory exceeds the configured bound"));
2808        }
2809        let offset = self.at;
2810        self.put(&directory)?;
2811        drop(timing);
2812        if let Some(profile) = &profile {
2813            profile.moved(Stage::Dictionary, 0, placed, 0);
2814            profile.moved(Stage::Publish, 0, self.at - before - placed, 0);
2815        }
2816        Ok(Entry {
2817            name: self.table.name.clone(),
2818            fields: self.table.fields.clone(),
2819            rows: self.table.rows,
2820            nonzero: table_nonzero_counts(&self.table),
2821            aggregates: table_aggregate_sums(&self.table),
2822            directory: Page {
2823                offset,
2824                length: u32::try_from(directory.len())
2825                    .map_err(|_| invalid("directory length overflow"))?,
2826                hash: checksum(&directory),
2827            },
2828        })
2829    }
2830
2831    /// Writes the statistics sections for the table being closed, as far as the budget reaches.
2832    ///
2833    /// Called from [`Self::close`] after the last stripe and after the dictionaries, which is the
2834    /// first moment the table's column bytes are final and the last moment before the directory is
2835    /// encoded. Both halves matter: the budget is a share of the column bytes, and a section that
2836    /// went in after the directory would be a section the directory does not name.
2837    ///
2838    /// Nothing here can fail the write. A column whose gather came back blind gets no sections, a
2839    /// column the budget could not reach gets none, and section 3.1 says both of those plan the way
2840    /// they planned before statistics existed. The two errors that are returned are an encode
2841    /// failure and a section count past the bound, and neither is a thing a column can cause.
2842    fn write_stats(&mut self) -> Result<()> {
2843        let gathers = std::mem::take(&mut self.gathers);
2844        let rows = self.table.rows as u64;
2845        let mut payloads = Vec::new();
2846        for (column, gather) in gathers.into_iter().enumerate() {
2847            let Some(gather) = gather else { continue };
2848            // A gather that saw a different number of rows than the table committed is a gather
2849            // that missed some, and a distinct count over some of a column is the one error an
2850            // estimator cannot see coming. This has no way of happening today, since a table is
2851            // written once and every chunk goes through `flush_pending`, and that is exactly why it
2852            // is worth a line: it stays true only while that stays true.
2853            if gather.rows() != rows {
2854                continue;
2855            }
2856            let Some(stats) = gather.finish() else { continue };
2857            let mut summary = Vec::new();
2858            stats.summary.encode(&mut summary)?;
2859            let mut sketches = Vec::new();
2860            stats.sketches.encode(&mut sketches)?;
2861            payloads.push((column, summary, sketches));
2862        }
2863        if payloads.is_empty() {
2864            return Ok(());
2865        }
2866        let costs = payloads
2867            .iter()
2868            .map(|(_, summary, sketches)| summary.len() + sketches.len())
2869            .collect::<Vec<_>>();
2870        let allowance = stats::allowance(stats::column_bytes(&self.table), stats::BUDGET_SHARE);
2871        // Nothing is spent yet. A table this writer is closing is one it wrote from nothing, so the
2872        // only statistics sections it can have are the ones about to go in.
2873        let keep = stats::within(&costs, allowance, 0);
2874        for ((column, summary, sketches), _) in
2875            payloads.iter().zip(&keep).filter(|&(_, &keep)| keep)
2876        {
2877            let id = u64::try_from(*column).map_err(|_| invalid("column index overflow"))?;
2878            for (kind, bytes, header_bytes) in [
2879                // A summary is a header the whole way down: there is nothing behind it a reader
2880                // could decide not to read.
2881                (*section::SUMMARY, summary, summary.len() as u32),
2882                (*section::SKETCHES, sketches, rudb_stats::sketches::HEADER_BYTES),
2883            ] {
2884                let written = write_section(
2885                    &self.file,
2886                    &mut self.at,
2887                    &section::Attachment { kind, id, flags: 0, header_bytes, bytes },
2888                    self.generation,
2889                )?;
2890                self.table.sections.push(written);
2891            }
2892        }
2893        if self.table.sections.len() > MAX_SECTIONS {
2894            return Err(invalid("the table would name more sections than the bound allows"));
2895        }
2896        Ok(())
2897    }
2898
2899    /// Commits every table this writer has written and syncs the file before publishing its header
2900    /// slot.
2901    ///
2902    /// The table handed back is the one the writer was on, which is the last of them. Callers that
2903    /// wrote several already know the others, since they named them.
2904    ///
2905    /// # Errors
2906    ///
2907    /// If directory encoding, writing, or syncing fails.
2908    pub fn finish(mut self) -> Result<Table> {
2909        let entry = self.close()?;
2910        let profile = self.profile.take();
2911        let _timing = profile.as_deref().map(|profile| profile.span(Stage::Publish));
2912        let mut tables = std::mem::take(&mut self.closed);
2913        tables.push(entry);
2914        let catalog = encode_catalog(&tables, &self.views)?;
2915        if catalog.len() > MAX_DIRECTORY {
2916            return Err(invalid("catalog exceeds the configured bound"));
2917        }
2918        let offset = self.at;
2919        self.put(&catalog)?;
2920        if let Some(profile) = &profile {
2921            profile.moved(Stage::Publish, 0, catalog.len() as u64, 0);
2922        }
2923        // Every page and every table directory is on the disk before anything points at them. The
2924        // slot write below is what makes this generation the one a reader picks, so the order of
2925        // these two syncs is the whole of the commit.
2926        synced(&self.file, profile.as_deref())?;
2927        let slot = Slot {
2928            offset,
2929            length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
2930            generation: self.generation,
2931            hash: checksum(&catalog),
2932        };
2933        // The one write that is not an append, and the last one. It goes back over the slot in the
2934        // header, so it names its offset rather than going through `put`, and `at` does not move.
2935        // Which of the two slots it is alternates with the generation, so the one naming the
2936        // generation before this is still intact and still valid until this write lands.
2937        write_at(&self.file, slot_offset(self.generation), &slot.bytes())?;
2938        synced(&self.file, profile.as_deref())?;
2939        Ok(self.table)
2940    }
2941
2942    /// Commits a generation that changes the views and leaves every table exactly where it is.
2943    ///
2944    /// There was no way to do this before views existed, because everything that could change the
2945    /// catalog also wrote a table, so the only way to say something new about a file was to go
2946    /// through a table. A view is the first thing that can change on its own. Without this, adding
2947    /// a view to a database with eight tables in it would rewrite all eight, since the append path
2948    /// needs a table to append and the fallback is the whole file.
2949    ///
2950    /// It is the same commit as [`Writer::finish`] with nothing appended before it. The table
2951    /// entries are carried forward by directory pointer the way an append carries them, the new
2952    /// catalog goes on the end, and the slot write at the end is what publishes it.
2953    ///
2954    /// # Errors
2955    ///
2956    /// If the file has no valid committed directory, is not this build's format, or cannot be
2957    /// written.
2958    pub fn restate(path: impl AsRef<Path>, views: &[ViewEntry]) -> Result<()> {
2959        let path = path.as_ref();
2960        let (_, size, slot, bytes, _) = slot_bytes(path)?;
2961        let (closed, _) = decode_catalog(&bytes, size)?;
2962        let generation = slot
2963            .generation
2964            .checked_add(1)
2965            .ok_or_else(|| invalid("native file generation overflow"))?;
2966        let catalog = encode_catalog(&closed, views)?;
2967        if catalog.len() > MAX_DIRECTORY {
2968            return Err(invalid("catalog exceeds the configured bound"));
2969        }
2970        let file = OpenOptions::new().write(true).read(true).open(path).map_err(io)?;
2971        write_at(&file, size, &catalog)?;
2972        file.sync_all().map_err(io)?;
2973        let slot = Slot {
2974            offset: size,
2975            length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
2976            generation,
2977            hash: checksum(&catalog),
2978        };
2979        write_at(&file, slot_offset(generation), &slot.bytes())?;
2980        file.sync_all().map_err(io)?;
2981        Ok(())
2982    }
2983
2984    /// Adds exact integer aggregate certificates to an older file's catalog without rewriting
2985    /// table pages. The old committed slot remains readable until the new catalog is fully synced.
2986    pub fn certify_summaries(path: impl AsRef<Path>) -> Result<()> {
2987        let path = path.as_ref();
2988        let (_, size, slot, bytes, _) = slot_bytes(path)?;
2989        let (mut entries, views) = decode_catalog(&bytes, size)?;
2990        let native = Catalog::open(path)?;
2991        for entry in &mut entries {
2992            let reader = native.table(&entry.name)?;
2993            entry.nonzero = reader_nonzero_counts(&reader)?;
2994            entry.aggregates = reader_aggregate_sums(&reader)?;
2995        }
2996        let generation = slot
2997            .generation
2998            .checked_add(1)
2999            .ok_or_else(|| invalid("native file generation overflow"))?;
3000        let catalog = encode_catalog(&entries, &views)?;
3001        if catalog.len() > MAX_DIRECTORY {
3002            return Err(invalid("catalog exceeds the configured bound"));
3003        }
3004        let file = OpenOptions::new().write(true).read(true).open(path).map_err(io)?;
3005        write_at(&file, size, &catalog)?;
3006        file.sync_all().map_err(io)?;
3007        let slot = Slot {
3008            offset: size,
3009            length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
3010            generation,
3011            hash: checksum(&catalog),
3012        };
3013        write_at(&file, slot_offset(generation), &slot.bytes())?;
3014        file.sync_all().map_err(io)?;
3015        Ok(())
3016    }
3017
3018    /// The earlier name for [`Self::certify_summaries`].
3019    pub fn certify_counts(path: impl AsRef<Path>) -> Result<()> {
3020        Self::certify_summaries(path)
3021    }
3022}
3023
3024/// Appends one run of bytes at `at` and moves it past them, answering where they went.
3025///
3026/// The append half of [`attach`], which cannot use [`Writer::put`] because it is not writing a
3027/// table. Every byte a section costs goes through here, so the offsets in an extent table come
3028/// from one place.
3029fn append(file: &File, at: &mut u64, bytes: &[u8]) -> Result<u64> {
3030    let offset = *at;
3031    write_at(file, offset, bytes)?;
3032    *at =
3033        at.checked_add(bytes.len() as u64).ok_or_else(|| invalid("native file length overflow"))?;
3034    Ok(offset)
3035}
3036
3037/// Writes one attachment's payload as extents and returns the entry that names it.
3038///
3039/// The split is by bytes, and the `first` of each extent is therefore a byte count. A section kind
3040/// whose extents should break on a row boundary instead will want to hand its extents over already
3041/// split; nothing needs that yet, and guessing at the shape of it now would be guessing.
3042fn write_section(
3043    file: &File,
3044    at: &mut u64,
3045    one: &section::Attachment<'_>,
3046    generation: u64,
3047) -> Result<Section> {
3048    // A payload of nothing is the exception, and it is not a special case so much as a different
3049    // reading of the same field: an entry with no bytes has no header to be longer than them, and
3050    // `header_bytes` is what the structure would have cost. See `Section::refused`.
3051    if !one.bytes.is_empty() && one.header_bytes as usize > one.bytes.len() {
3052        return Err(invalid("a section's header is longer than its payload"));
3053    }
3054    let mut extents = Vec::new();
3055    let mut first = 0_u64;
3056    for chunk in one.bytes.chunks(section::MAX_EXTENT as usize) {
3057        let offset = append(file, at, chunk)?;
3058        extents.push(section::Extent {
3059            offset,
3060            length: u32::try_from(chunk.len()).map_err(|_| invalid("extent length overflow"))?,
3061            hash: checksum(chunk),
3062            first,
3063        });
3064        first += chunk.len() as u64;
3065    }
3066    let mut table = Vec::with_capacity(extents.len() * section::EXTENT_BYTES);
3067    section::encode_extents(&extents, &mut table)?;
3068    // A payload of nothing is a section of no extents and no extent table, and its `extent_page`
3069    // is zero rather than the end of the file. Section 3.7 wants that entry to exist: it is how a
3070    // relationship that did not fit the budget is recorded as not built rather than forgotten.
3071    let extent_page = if table.is_empty() { 0 } else { append(file, at, &table)? };
3072    Ok(Section {
3073        kind: one.kind,
3074        id: one.id,
3075        generation,
3076        extents: u32::try_from(extents.len()).map_err(|_| invalid("too many extents"))?,
3077        extent_page,
3078        extent_bytes: u32::try_from(table.len()).map_err(|_| invalid("extent table overflow"))?,
3079        hash: checksum(&table),
3080        flags: one.flags,
3081        header_bytes: one.header_bytes,
3082    })
3083}
3084
3085/// Attaches graph sections to a table already committed in a file, without rewriting a page.
3086///
3087/// This is the second pass spec/graph/03-the-file-format.md section 3.8 asks for. A key map has to
3088/// exist before the link that uses it can be built, and it is built by reading the key column back,
3089/// so the structures of a table cannot be written during the load that wrote the table. They are
3090/// written afterwards, by this, and the file in between the two is a correct file that answers
3091/// every query more slowly.
3092///
3093/// Nothing is overwritten. The payloads, the extent tables, the new directory for this table and
3094/// the new catalog all go on the end of the file past the committed generation, and the last write
3095/// is the header slot, exactly as [`Writer::finish`] does it. So a crash anywhere in here leaves
3096/// the generation before it intact, and leaves unreferenced trailing bytes that the next commit
3097/// writes past.
3098///
3099/// An attachment replaces any section of the same kind and id, and every other section is carried
3100/// through untouched, including one whose kind this build does not know. The table's own generation
3101/// is carried through too, because attaching a section moves no row: see [`Table::generation`].
3102///
3103/// # Errors
3104///
3105/// If the file has no valid committed directory, is an older format than this build writes, holds
3106/// no table of that name, names a section whose payload cannot be written, or would end up naming
3107/// more sections than the format allows.
3108pub fn attach(
3109    path: impl AsRef<Path>,
3110    table: &str,
3111    attachments: &[section::Attachment<'_>],
3112) -> Result<Table> {
3113    let path = path.as_ref();
3114    let (_, size, slot, bytes, _) = slot_bytes(path)?;
3115    let (mut entries, views) = decode_catalog(&bytes, size)?;
3116    let at = entries
3117        .iter()
3118        .position(|entry| entry.name == table)
3119        .ok_or_else(|| invalid(&format!("the file holds no table called {table}")))?;
3120    let file = OpenOptions::new().write(true).read(true).open(path).map_err(io)?;
3121    let mut version = [0; 4];
3122    read_at(&file, 8, &mut version)?;
3123    let version = u32::from_le_bytes(version);
3124    // Readable is not the same as writable. A format 22 file has no section table, and giving its
3125    // directory one without moving the number in its header would leave a file that claims to be
3126    // format 22 and is not, which is worse than refusing. Rewriting it with this build is the
3127    // answer, and the format is at 0.3.x, so nobody has one of these that this project did not
3128    // just make.
3129    if version != FORMAT {
3130        return Err(invalid(&format!(
3131            "the file is format {version} and a graph section needs format {FORMAT}, so it has \
3132             to be written again"
3133        )));
3134    }
3135    let mut directory = vec![0; entries[at].directory.length as usize];
3136    read_at(&file, entries[at].directory.offset, &mut directory)?;
3137    if checksum(&directory) != entries[at].directory.hash {
3138        return Err(invalid(&format!("the directory of table {table} does not checksum")));
3139    }
3140    let mut held = decode_directory(&directory, size)?;
3141    let mut cursor = size;
3142    for one in attachments {
3143        let written = write_section(&file, &mut cursor, one, held.generation)?;
3144        held.sections.retain(|old| !(old.kind == one.kind && old.id == one.id));
3145        held.sections.push(written);
3146    }
3147    if held.sections.len() > MAX_SECTIONS {
3148        return Err(invalid("the table would name more sections than the bound allows"));
3149    }
3150    let encoded = encode_directory(&held)?;
3151    if encoded.len() > MAX_DIRECTORY {
3152        return Err(invalid("directory exceeds the configured bound"));
3153    }
3154    let offset = append(&file, &mut cursor, &encoded)?;
3155    entries[at].directory = Page {
3156        offset,
3157        length: u32::try_from(encoded.len()).map_err(|_| invalid("directory length overflow"))?,
3158        hash: checksum(&encoded),
3159    };
3160    // The views the file already had, written back unchanged. Attaching a section to a table says
3161    // nothing about a view and must not drop one.
3162    let catalog = encode_catalog(&entries, &views)?;
3163    if catalog.len() > MAX_DIRECTORY {
3164        return Err(invalid("catalog exceeds the configured bound"));
3165    }
3166    let offset = append(&file, &mut cursor, &catalog)?;
3167    file.sync_all().map_err(io)?;
3168    let generation =
3169        slot.generation.checked_add(1).ok_or_else(|| invalid("native file generation overflow"))?;
3170    let committed = Slot {
3171        offset,
3172        length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
3173        generation,
3174        hash: checksum(&catalog),
3175    };
3176    write_at(&file, slot_offset(generation), &committed.bytes())?;
3177    file.sync_all().map_err(io)?;
3178    Ok(held)
3179}
3180
3181/// One column's frequency synopsis as values with their row counts, shared by every clone of a
3182/// reader.
3183type Synopsis = Arc<Vec<(Value, u64)>>;
3184
3185/// Reads committed native column pages without holding the table in memory.
3186#[derive(Debug, Clone)]
3187pub struct Reader {
3188    file: Arc<File>,
3189    table: Arc<Table>,
3190    dictionaries: Arc<Vec<OnceLock<Arc<Vector>>>>,
3191    /// Held while a global dictionary is being opened, one per column.
3192    ///
3193    /// The [`OnceLock`] above says whether one has been opened, which is the question a reader that
3194    /// already has it needs answered and is free. It does not say whether one is being opened, and
3195    /// the difference matters because every worker of a scan wants the same dictionary at the same
3196    /// moment. Without this they all miss, all read the page, all verify it and all decode it, and
3197    /// all but one throw the answer away. ClickBench 38 reads the URL dictionary, which is 515,958
3198    /// entries, and was paying for it twice.
3199    loading: Arc<Vec<Mutex<()>>>,
3200    /// Each column's frequency synopsis as values, the first time anything asks for it. See
3201    /// [`Reader::decode_frequencies`].
3202    frequency_values: Arc<Vec<OnceLock<Synopsis>>>,
3203    /// Stored frequency sections are decoded once per open table. A small directory can hold the
3204    /// summary inline, but a larger one otherwise rereads and decodes the same section on every
3205    /// plan and every summary-backed aggregate.
3206    frequency_summaries: Arc<Vec<OnceLock<Arc<FrequencySummary>>>>,
3207    /// How many global dictionaries have been opened. A scan of a dictionary column should open its
3208    /// dictionary once however many workers it has, and the test that says so is the only thing
3209    /// keeping it that way.
3210    opened: Arc<AtomicUsize>,
3211    /// The membership sieves of one stripe of one column, by column and then by stripe, read the
3212    /// first time a probe asks about them. A query filters on one or two columns and never looks at
3213    /// the rest, so reading these at open would be the whole index for the sake of a fraction of it.
3214    sieves: Arc<Vec<Vec<SieveSlot>>>,
3215    /// The per part ranges of one stripe of one column, by column and then by stripe, read the
3216    /// first time something compares that column and kept after that.
3217    part_ranges: Arc<Vec<Vec<RangeSlot>>>,
3218    /// Which stripe and which part of it every part of the table is, by table wide part number.
3219    places: Arc<Vec<Place>>,
3220    cache: Arc<Shelf>,
3221    /// Where the pages above are counted against the database's budget. See [`PagePool`].
3222    pool: PagePool,
3223    /// How many whole stripe pages have been read, which is what the sharing above is judged on. A
3224    /// scan of a column should read each of its stripes once however many workers it has.
3225    pages: Arc<AtomicUsize>,
3226    /// How many index sections have been read. A scan of a column should read each of its stripes
3227    /// once here too, and the test that says so is the only thing keeping it that way.
3228    indexes: Arc<AtomicUsize>,
3229    /// The file's size when it was opened, for [`Reader::layout`].
3230    size: u64,
3231    /// The committed directory's size, for [`Reader::layout`].
3232    directory: u64,
3233    /// What opening the file cost, which is a number rather than a claim.
3234    opening: Opening,
3235}
3236
3237/// What [`Reader::open`] read before it returned.
3238///
3239/// `spec/stats/04-in-memory.md` section 4.2 says opening a table reads the header and the directory
3240/// and nothing else, and once that document's statistics are in the file the tempting change is to
3241/// load a column summary or two on the way past, because they are small and the next query will
3242/// want them. A hundred milliseconds of that is a hundred milliseconds nobody asked for, and an
3243/// embedded database is opened by processes that are about to run one trivial query.
3244///
3245/// So the claim gets a number. Both of these are fixed by the schema and the stripe count and are
3246/// independent of how many rows the file holds, and the test that says so is what stops the
3247/// tempting change from landing quietly.
3248#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
3249pub struct Opening {
3250    /// How many times the file was read. The header, then each directory slot that looked valid
3251    /// enough to check, so three at the most.
3252    pub reads: u32,
3253    /// How many bytes those reads asked for.
3254    pub bytes: u64,
3255}
3256
3257/// What a reader has read, while it was being opened and since.
3258#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
3259pub struct Reads {
3260    /// What opening cost, before any query had been planned.
3261    pub opening: Opening,
3262    /// Whole stripe pages read since.
3263    pub pages: usize,
3264    /// Index sections read since.
3265    pub indexes: usize,
3266    /// Global dictionaries opened since. One per dictionary column that a query touched, however
3267    /// many workers touched it, which is a claim only a test can keep true.
3268    pub dictionaries: usize,
3269}
3270
3271/// Where one table wide part number lands.
3272#[derive(Debug, Clone, Copy)]
3273struct Place {
3274    stripe: u32,
3275    part: u32,
3276    rows: u32,
3277}
3278
3279/// One part's bytes inside one column page.
3280#[derive(Debug, Clone, Copy)]
3281struct PartSpan {
3282    start: usize,
3283    length: usize,
3284    hash: u64,
3285}
3286
3287/// What a reader holds for one stripe of one column.
3288///
3289/// The index is small and is loaded whether the caller wants the whole page or one part of it. The
3290/// page is loaded only by a scan, because a sparse fetch that wants a thousand rows out of sixty
3291/// four thousand would be reading sixty four times what it uses.
3292#[derive(Debug, Clone)]
3293struct CachedColumn {
3294    stripe: usize,
3295    index: Arc<Vec<PartSpan>>,
3296    page: Option<Arc<Vec<u8>>>,
3297}
3298
3299/// One column's stripes a reader holds, and which of them somebody is reading right now.
3300///
3301/// The pages are one slot per stripe of the table rather than a list of the ones being kept, so
3302/// finding a page is an index and not a walk. That matters because the walk happened under the
3303/// lock, once per part per column, and a scan that gives a whole stripe to each of thirty two
3304/// workers keeps enough pages that walking them was the longest thing the lock was held for. The
3305/// slots cost a pointer per stripe per column, which on the ClickBench file is eight kilobytes
3306/// against the forty megabytes of pages they point at. `order` is which of them are filled, oldest
3307/// first, because that is the one thing the slots cannot say by themselves.
3308///
3309/// `loading` is what keeps a scan from reading the same page once per worker. It is a list and not
3310/// a set because it holds at most one stripe per worker on the column and is walked far less often
3311/// than a hash of it would be built.
3312///
3313/// `index` is every index this reader has ever read for the column, one slot per stripe, and it is
3314/// never evicted. An index is a few hundred bytes and a page is a quarter of a megabyte, so the two
3315/// do not belong under the same budget. Riding in the page cache meant a worker that came back to a
3316/// stripe after its page had been evicted read the index again with it, which on the full
3317/// ClickBench file was about thirteen hundred reads out of a hundred and fourteen thousand.
3318#[derive(Debug, Default)]
3319struct Cached {
3320    pages: Vec<Option<Resident>>,
3321    loading: Vec<usize>,
3322    index: Vec<Option<Arc<Vec<PartSpan>>>>,
3323}
3324
3325/// One page a reader holds, and whether anyone has read it since the pool last looked.
3326#[derive(Debug, Clone)]
3327struct Resident {
3328    page: Arc<Vec<u8>>,
3329    used: Arc<AtomicBool>,
3330}
3331
3332/// Every column's pages of one reader, with how many each column holds and the floor under that.
3333#[derive(Debug)]
3334struct Shelf {
3335    columns: Vec<Mutex<Cached>>,
3336    /// How many pages each column holds right now. Counted outside the column locks so that the
3337    /// pool can tell whether a column is at its floor without taking a lock it might be under.
3338    held: Vec<AtomicUsize>,
3339    /// How many stripes of one column are kept whatever the budget says. See
3340    /// [`CACHED_STRIPES_PER_COLUMN`] for what sets it and [`Reader::keep_stripes`] for who raises it.
3341    kept: AtomicUsize,
3342}
3343
3344/// The pages every reader of one database keeps, under one budget in bytes.
3345///
3346/// A reader lives as long as the database does, so the pages it holds are what the next query finds
3347/// already in memory. They used to be four stripes a column, oldest out first, which on TPC-H SF1
3348/// meant every query read every page of lineitem off the file again and paid the system call for
3349/// it. Keeping every page there costs 38 MB and took a third of the system time off the suite.
3350///
3351/// So the question is no longer how many stripes a column keeps but how many bytes the database
3352/// does, and one budget answers it for every reader at once. A table nobody queries gives its pages
3353/// up to one that is being queried, which a count per column cannot do.
3354///
3355/// Pages leave by the clock. Each has a bit a read sets, and when the pool is over budget it walks
3356/// from the oldest: a page with the bit set loses the bit and goes round again, and a page without
3357/// it goes. That keeps what is read over and over and lets a page one scan read once go first.
3358///
3359/// The old count is still a floor. A column never gives up a page while it holds four or fewer,
3360/// because a scan whose workers evict each other's pages reads a quarter of a megabyte for every
3361/// part it takes, and a budget of zero is the cache as it was before the pool existed.
3362#[derive(Debug, Clone, Default)]
3363pub struct PagePool {
3364    ring: Arc<Mutex<Ring>>,
3365    budget: Arc<AtomicUsize>,
3366}
3367
3368#[derive(Debug, Default)]
3369struct Ring {
3370    held: VecDeque<Held>,
3371    bytes: usize,
3372}
3373
3374/// One page in the pool, pointing back at the reader that holds it.
3375///
3376/// Weak, because a reader that has gone, which every reader does at a checkpoint, should take its
3377/// pages with it and not have them kept alive by the pool.
3378#[derive(Debug)]
3379struct Held {
3380    shelf: Weak<Shelf>,
3381    column: usize,
3382    stripe: usize,
3383    bytes: usize,
3384    used: Arc<AtomicBool>,
3385}
3386
3387impl PagePool {
3388    /// A pool that keeps up to `budget` bytes of pages beyond each column's floor.
3389    #[must_use]
3390    pub fn new(budget: usize) -> Self {
3391        let pool = Self::default();
3392        pool.budget.store(budget, Atomic::Relaxed);
3393        pool
3394    }
3395
3396    /// The bytes of pages the pool is counting now.
3397    ///
3398    /// # Panics
3399    ///
3400    /// If the pool's lock is poisoned, which takes a panic while it was held.
3401    #[must_use]
3402    pub fn bytes(&self) -> usize {
3403        self.ring.lock().map_or(0, |ring| ring.bytes)
3404    }
3405
3406    /// Counts a page a reader has just taken in, and lets pages go until the pool is back under its
3407    /// budget or it has looked at every page once.
3408    ///
3409    /// Called with no column lock held. The pages that go are chosen under the pool's lock and
3410    /// dropped under their column's lock afterwards, so no thread ever holds both.
3411    fn admit(&self, held: Held) {
3412        let budget = self.budget.load(Atomic::Relaxed);
3413        let mut gone = Vec::new();
3414        {
3415            let Ok(mut ring) = self.ring.lock() else { return };
3416            ring.bytes += held.bytes;
3417            ring.held.push_back(held);
3418            // One lap and no more. A page read since the last pass loses its bit on this one and
3419            // can only go on a later one, which is the second chance the clock is named for.
3420            let mut looked = 0;
3421            let limit = ring.held.len();
3422            while ring.bytes > budget && looked < limit {
3423                looked += 1;
3424                let Some(entry) = ring.held.pop_front() else { break };
3425                let Some(shelf) = entry.shelf.upgrade() else {
3426                    ring.bytes -= entry.bytes;
3427                    continue;
3428                };
3429                if entry.used.swap(false, Atomic::Relaxed) {
3430                    ring.held.push_back(entry);
3431                    continue;
3432                }
3433                let count = &shelf.held[entry.column];
3434                if count.load(Atomic::Relaxed) <= shelf.kept.load(Atomic::Relaxed).max(1) {
3435                    ring.held.push_back(entry);
3436                    continue;
3437                }
3438                count.fetch_sub(1, Atomic::Relaxed);
3439                ring.bytes -= entry.bytes;
3440                gone.push((shelf, entry));
3441            }
3442            // A reader that has gone leaves its entries behind, and with a budget nobody reaches
3443            // they would pile up one checkpoint after another. The front is where the oldest are.
3444            while ring.held.front().is_some_and(|entry| entry.shelf.strong_count() == 0) {
3445                if let Some(entry) = ring.held.pop_front() {
3446                    ring.bytes -= entry.bytes;
3447                }
3448            }
3449        }
3450        for (shelf, entry) in gone {
3451            let Ok(mut cached) = shelf.columns[entry.column].lock() else { continue };
3452            if let Some(slot) = cached.pages.get_mut(entry.stripe) {
3453                if slot.as_ref().is_some_and(|slot| Arc::ptr_eq(&slot.used, &entry.used)) {
3454                    *slot = None;
3455                }
3456            }
3457        }
3458    }
3459}
3460
3461/// Stripes of one column a reader keeps the bytes of, when nobody has asked for more.
3462///
3463/// This has to hold at least as many stripes as a column has workers in it at once, or the workers
3464/// evict each other's pages and read them again. Four is what a scan that hands parts out in order
3465/// needs, because then every worker is within a few parts of every other and at most a couple of
3466/// stripes are open at a time. A scan that hands a whole stripe to each worker has one stripe open
3467/// per worker for the length of that stripe, and it says so with [`Reader::keep_stripes`] rather
3468/// than paying for sixteen slots on every table that is read one part at a time.
3469///
3470/// It multiplies by the page size, which is a quarter of a megabyte for a four byte column, and by
3471/// the number of columns a query touches.
3472const CACHED_STRIPES_PER_COLUMN: usize = 4;
3473
3474/// The sieves of one stripe of one column, once somebody has asked for them.
3475type SieveSlot = OnceLock<Arc<Vec<Option<Sieve>>>>;
3476
3477type RangeSlot = OnceLock<Arc<Vec<Range>>>;
3478
3479#[derive(Debug)]
3480struct NativeText {
3481    file: Arc<File>,
3482    /// How many values the dictionary holds.
3483    values: usize,
3484    /// Where each value ends inside its payload block, packed at `offset_bits` in runs of
3485    /// [`TEXT_OFFSET_RUN`].
3486    ///
3487    /// Ends rather than starts, because then a block of 1,024 values is 1,024 numbers rather than
3488    /// 1,025: the start of a value is the end of the one before it, and the first value of a block
3489    /// starts at zero by construction. Relative to the block rather than to the payload, because a
3490    /// reader decodes a whole block and slices it, so an offset into the payload is a number it
3491    /// would have to subtract a base from anyway.
3492    offsets: Vec<u8>,
3493    /// Bits one offset is packed at, which is what the largest block of this column spans and is the
3494    /// same for every block of it.
3495    offset_bits: usize,
3496    /// The same ends unpacked, built once enough readers have asked for one at a time.
3497    ///
3498    /// Reading one offset out of the packed form costs about fifty instructions: a division to find
3499    /// the run, a bounds check to slice it, a shift to reach the bit the value starts at and a
3500    /// narrowing on the way out. That is the right price for a reader that wants a handful. It is
3501    /// the wrong price for `STRLEN` over a column, which asks for one per row and nothing else, and
3502    /// where a million of them was a third of ClickBench 28.
3503    ///
3504    /// With the ends unpacked every read is a load, and a vector of lengths is one loop over them.
3505    /// The table is built only once the reads say it will be used, which is what
3506    /// [`Self::ends_worth_unpacking`] decides and [`Self::ends_asked`] counts towards, because a
3507    /// table built for a reader that wanted three values is four bytes a value spent on nothing.
3508    value_ends: OnceLock<Option<Vec<u32>>>,
3509    /// The length of every value, worked out of [`Self::value_ends`] the first time a vector of
3510    /// lengths is asked for.
3511    ///
3512    /// A length out of the ends is two loads, a test for whether the value opens its block and a
3513    /// check that it does not end before it starts, which came to thirteen instructions a row on
3514    /// ClickBench 28. Out of this it is one load. The order is checked once for the whole table
3515    /// while it is built, and a column that fails it gets no table and goes on reading the ends,
3516    /// which is where the error is reported. Four bytes a value, and only for a column something
3517    /// has asked the length of a vector at a time.
3518    value_lens: OnceLock<Option<Vec<u32>>>,
3519    /// How many single offset reads have come in while the table is not built.
3520    ///
3521    /// Relaxed, and read only against a threshold, so two threads racing here means the table is
3522    /// built one read early or one read late. Counting stops the moment the table exists, because
3523    /// [`OnceLock::get`] settles it before this is touched.
3524    ends_asked: AtomicUsize,
3525    /// How many entries the sorted order has, which is the value count.
3526    ranks: usize,
3527    /// Where the sorted order starts in the file. It is read a block at a time and only when
3528    /// something searches it, so a query that never compares this column against a literal never
3529    /// touches it at all.
3530    rank_at: u64,
3531    /// Where each block of the sorted order ends, as a byte offset from `rank_at`. A block is packed
3532    /// at whatever width its own heads need, so unlike the entries it replaced its length is not
3533    /// arithmetic on the block number.
3534    rank_ends: Vec<u64>,
3535    rank_hashes: Vec<u64>,
3536    rank_blocks: Vec<OnceLock<Result<Vec<u8>>>>,
3537    /// Bits one code is packed at, which is what the value count needs and is the same for every
3538    /// block of the column.
3539    code_bits: usize,
3540    /// The sorted order turned round, built the first time a reader asks for it.
3541    ///
3542    /// Four bytes per value against the four the offsets already hold, so a column that has this is
3543    /// carrying half again what it carried before rather than something of a new order. It is built
3544    /// only when something asks, which is a grouped min or max over this column and nothing else,
3545    /// and that reader was going to read the payload of this column once per row otherwise.
3546    code_ranks: OnceLock<Option<Vec<u32>>>,
3547    /// Where each block of the payload starts in the file, and how many stored bytes it is.
3548    ///
3549    /// Absolute rather than an offset from a base the blocks share, because a block is written the
3550    /// moment it fills and what comes after it in the file is whatever the load wrote next. A file
3551    /// old enough to have them back to back is read into these same two lists by adding the base to
3552    /// the ends it carries, so nothing below here knows which kind of file it came from.
3553    starts: Vec<u64>,
3554    lengths: Vec<u64>,
3555    hashes: Vec<u64>,
3556    /// Conservative four-byte substring signatures, read only by a compatible LIKE filter.
3557    grams: Option<NativeGrams>,
3558    /// The payload, read and decoded a block at a time and kept after that.
3559    blocks: Vec<OnceLock<Result<Vec<u8>>>>,
3560    /// How many decoded payload bytes this column keeps before a sweep stops keeping what it reads.
3561    /// [`TEXT_KEEP_BUDGET`] everywhere but in the test of the ceiling.
3562    keep_budget: usize,
3563    /// Roughly how many decoded payload bytes are being kept, which is what [`TEXT_KEEP_BUDGET`]
3564    /// is measured against.
3565    ///
3566    /// Roughly, because two threads that keep the same block at the same time both add its length
3567    /// while [`OnceLock`] keeps one of the two. That makes the count read high and the budget bind
3568    /// a little early, which is the harmless direction, and it costs one relaxed add a block rather
3569    /// than a lock on the path every scan of a string column goes through.
3570    payload_kept: AtomicUsize,
3571    /// Which payload blocks a sweep has decoded before, one flag a block.
3572    ///
3573    /// A sweep keeps a block the second time it decodes it and not the first. A process that runs
3574    /// one statement, which is how a benchmark or a script uses the engine, sweeps each block once
3575    /// and so keeps nothing: on ten million rows a `URL LIKE` held 396 MB with every block kept and
3576    /// 97 MB with none, for the same processor time. A session that asks again pays the decode one
3577    /// more time and reads kept blocks from then on, under the same [`TEXT_KEEP_BUDGET`].
3578    swept: Vec<AtomicBool>,
3579    /// The boundaries this dictionary has already been searched for, by the value searched for.
3580    ///
3581    /// A search is the expensive thing this type does. It settles a probe on the stored head where
3582    /// it can and reads a value where it cannot, and reading a value decodes the payload block it
3583    /// sits in, so one search can cost several blocks. The thing that makes remembering worth it is
3584    /// that the same search comes back: a top N asks once a chunk whether anything left can beat its
3585    /// worst candidate, and the worst candidate settles long before the chunks run out.
3586    ///
3587    /// Shared across the instances of a scan rather than kept per instance, because each of them has
3588    /// its own worst candidate and all of them are searching the same dictionary. One lock per chunk
3589    /// is nothing next to a probe of a file.
3590    ///
3591    /// Bounded by [`TEXT_SEARCH_MEMO`] and emptied rather than evicted when it is full. What fills
3592    /// it is a top N improving its bound, which happens a few dozen times and then stops, so the
3593    /// bound is there for the filter that searches for a different literal every chunk rather than
3594    /// for anything this is meant to help.
3595    searched: Mutex<HashMap<Vec<u8>, (usize, bool)>>,
3596}
3597
3598#[derive(Debug)]
3599struct NativeGrams {
3600    start: u64,
3601    length: usize,
3602    hash: u64,
3603    loaded: OnceLock<Result<Vec<u8>>>,
3604}
3605
3606/// How many searched for values a column's dictionary remembers the boundary of.
3607///
3608/// See [`NativeText::searched`]. Small because the case it is for repeats one value, not because a
3609/// larger one would be wrong.
3610const TEXT_SEARCH_MEMO: usize = 64;
3611
3612/// How many values of a dictionary go in one block of the payload.
3613///
3614/// The block is the unit the string cascade encodes, the unit a checksum covers, and the unit a
3615/// reader has to decode to get at a single value, so it is the one number the payload format turns
3616/// on. Blocking by values rather than by bytes is what keeps a value out of two blocks at once: the
3617/// block holding a code is `code / TEXT_PAYLOAD_VALUES` and nothing has to be stitched.
3618///
3619/// A probe on the five ClickBench columns that have a dictionary worth the name, written up on
3620/// #347, measured the ratio and the decode speed at 128, 256, 512, 1,024 and 4,096 values. Both get
3621/// better all the way up, because front coding and the LZ matcher have more to look back at and
3622/// because the per chunk setup is spread over more values. What stops it is the point read: a query
3623/// that wants ten values has to decode ten blocks, so the block is what a lookup costs. At 1,024
3624/// values a block is between 67 KB and 394 KB decoded across those five columns, and the ratios are
3625/// 2.3 to 4.5. Going up to 4,096 buys two to six percent more and makes a block as much as 1.5 MB.
3626/// Going down to 512 gives up five to nine percent.
3627const TEXT_PAYLOAD_VALUES: usize = 1024;
3628
3629/// Two KiB per payload block makes a four-byte substring a useful negative test without keeping a
3630/// large lookup table. The load and file-size costs must pass the same end-to-end gate as queries.
3631const TEXT_GRAM_BYTES: usize = 2048;
3632
3633/// A fast mixing step for exactly four bytes, shared by load and query.
3634fn gram_bits(bytes: &[u8]) -> [usize; 2] {
3635    let original = u32::from_le_bytes(bytes.try_into().expect("a four-byte gram"));
3636    let mut first = original ^ (original >> 16);
3637    first = first.wrapping_mul(0x7feb_352d);
3638    first ^= first >> 15;
3639    let mut second = original ^ (original >> 17);
3640    second = second.wrapping_mul(0x846c_a68b);
3641    second ^= second >> 16;
3642    let mask = TEXT_GRAM_BYTES * 8 - 1;
3643    [(first as usize) & mask, (second as usize) & mask]
3644}
3645
3646/// How many decoded payload bytes one dictionary keeps before a sweep stops keeping what it reads.
3647///
3648/// A sweep of the whole dictionary decodes every block whatever it does, and the only question is
3649/// whether it hangs on to them. Keeping all of them is 4.2 GB on ClickBench `URL` at a hundred
3650/// million rows, which is what #997 was right to stop. Keeping none of them means the next query
3651/// asking the same thing decodes all of it again, and on the same column at a million rows that
3652/// took a `LIKE` from 2.7 ms to 16.2 ms, because the decode used to be paid once by a session and
3653/// is now paid by every statement in it. Neither end is the answer. A bound is.
3654///
3655/// So a sweep keeps what it decodes until the column is holding this much and decodes without
3656/// keeping after that. At a million rows the five ClickBench string columns decode to between 8 MB
3657/// and 85 MB, so they sit inside it and a repeated `LIKE` reads a decoded block rather than a
3658/// stored one. At a hundred million rows `URL` fills it and the rest of that column is read and
3659/// dropped, which is the old cost on the part that does not fit and none of the old footprint.
3660///
3661/// Two hundred and fifty six megabytes a column is a number and not a policy, and the policy is
3662/// what should replace it: this wants to be a buffer pool over the whole database, sized against
3663/// the memory limit the session was given, with the blocks of every column competing for it and the
3664/// least useful one evicted. That is F2 work. What is here is the part of it that can be written
3665/// without an eviction order, which is a ceiling.
3666const TEXT_KEEP_BUDGET: usize = 256 * 1024 * 1024;
3667
3668/// The length of every value out of where each one ends inside its payload block, or `None` for
3669/// ends that go backwards somewhere inside a block.
3670///
3671/// A value that opens a block starts at zero and every other one starts where the value before it
3672/// ends, so a block is a run of differences.
3673fn lengths_of(ends: &[u32]) -> Option<Vec<u32>> {
3674    let mut lens = Vec::with_capacity(ends.len());
3675    for block in ends.chunks(TEXT_PAYLOAD_VALUES) {
3676        let mut start = 0;
3677        for &end in block {
3678            lens.push(end.checked_sub(start)?);
3679            start = end;
3680        }
3681    }
3682    Some(lens)
3683}
3684
3685/// How many offsets go in one packed run.
3686///
3687/// A payload block holds 1,024 values and `bitpack::pack_tail` takes fewer than 1,024 at a time,
3688/// since a whole unit of that many belongs in the transposed layout instead. So the offsets of a
3689/// block go in two runs. Five hundred and twelve values at any width is a whole number of bytes, so
3690/// a run starts where a multiply says it does and nothing is padded.
3691const TEXT_OFFSET_RUN: usize = 512;
3692
3693/// Bytes at the front of a global dictionary index: the value count, the values a payload block
3694/// holds, the block count and the bits an offset is packed at.
3695const DICTIONARY_HEADER: usize = 16;
3696
3697/// Set beside the offset width in the fourth word of a global dictionary index, meaning each
3698/// payload block says where in the file it starts and how long it is, rather than sitting directly
3699/// behind the block before it.
3700///
3701/// In that word rather than in a word of its own because the width is at most 32 and lives in a
3702/// `u32`, so the top of it has never been anything. A build old enough not to know the flag reads
3703/// the file's format before it reads any of this and refuses it there, and if it somehow did get
3704/// here it would find an offset width of two billion and say so.
3705///
3706/// The point of the flag is that a block written the moment it fills does not know what will be
3707/// written after it, so the payload of a column cannot be one run of bytes unless the whole column
3708/// is held until the file is closed. That is the memory the load cannot afford. What it costs is
3709/// eight bytes a block, against the block being a thousand values.
3710const DICTIONARY_SCATTERED: u32 = 1 << 31;
3711/// The dictionary index carries one four-byte substring signature per payload block.
3712const DICTIONARY_GRAMS: u32 = 1 << 30;
3713
3714/// How many entries of a dictionary's sorted order sit in one block that is read and checked as a
3715/// unit.
3716///
3717/// Five hundred and twelve entries is between two and three kilobytes on the ClickBench string
3718/// columns, which is well under a page. A binary search over half a million entries makes nineteen
3719/// probes, and the first ten land in ten different blocks while the last nine land in the one block
3720/// that holds the answer, so the whole search reads about thirty kilobytes of a megabyte of order. A
3721/// smaller block would save a little on the early probes, cost a checksum and an end list four times
3722/// as long, and give the heads less to share a base with. A larger one would read more than it uses
3723/// on every probe.
3724const TEXT_RANK_BLOCK: usize = 512;
3725
3726/// Bytes at the front of a rank block, which is the base of its heads and the width they are packed
3727/// at.
3728///
3729/// An entry used to be twelve bytes flat, eight for the head and four for the code, and on the five
3730/// ClickBench columns that have a dictionary worth the name that was 744 MB of a 12.2 GB file. Both
3731/// halves of it are nearly empty. The heads are the first eight bytes of the values in sorted order,
3732/// so a block of five hundred and twelve of them spans a tiny slice of the column, and on a column of
3733/// URLs they are all `http://w` and the block holds one distinct head. The codes are positions in a
3734/// dictionary of eighteen million, which is twenty five bits and not thirty two.
3735///
3736/// So a block now writes the smallest head in it, the bits the largest is above that, and the heads
3737/// and the codes packed at the width each needs. A block where every head agrees costs nine bytes
3738/// and the codes.
3739const RANK_BLOCK_HEADER: usize = size_of::<u64>() + 1;
3740
3741impl NativeText {
3742    /// One block of the payload, read and decoded the first time anything asks for a value in it.
3743    ///
3744    /// The bytes handed back are the values of the block laid end to end, which is what the offsets
3745    /// describe, so a caller slices it with the offsets it already has. Where the block sits in the
3746    /// file is the only thing the caller cannot work out for itself, because the stored form is
3747    /// shorter than the decoded one and by a different amount in every block.
3748    fn payload_block(&self, block: usize) -> Result<Option<&[u8]>> {
3749        let Some(slot) = self.blocks.get(block) else { return Ok(None) };
3750        let bytes = slot.get_or_init(|| self.decode_block(block)).as_ref().map_err(Clone::clone)?;
3751        Ok(Some(bytes.as_slice()))
3752    }
3753
3754    /// Reads and decodes one block of the payload, without deciding who keeps it.
3755    ///
3756    /// [`Self::payload_block`] keeps it forever, which is what a point read wants and what a walk
3757    /// of the whole dictionary must not do. Both call this and they differ in nothing else.
3758    fn decode_block(&self, block: usize) -> Result<Vec<u8>> {
3759        let len = self.lengths[block];
3760        let mut stored = vec![
3761            0;
3762            usize::try_from(len).map_err(|_| invalid(
3763                "global dictionary block does not fit in memory"
3764            ))?
3765        ];
3766        read_at(&self.file, self.starts[block], &mut stored)?;
3767        if checksum(&stored) != self.hashes[block] {
3768            return Err(invalid("global dictionary payload checksum differs"));
3769        }
3770        let first = block * TEXT_PAYLOAD_VALUES;
3771        let last = (first + TEXT_PAYLOAD_VALUES).min(self.values);
3772        let want = self.end_within(last - 1)? as usize;
3773        let values = string::decode_flat(&stored)?;
3774        if values.len() != last - first {
3775            return Err(invalid("global dictionary block holds the wrong value count"));
3776        }
3777        let bytes = values.into_bytes();
3778        if bytes.len() != want {
3779            return Err(invalid("global dictionary block decodes to the wrong length"));
3780        }
3781        Ok(bytes)
3782    }
3783
3784    /// How many single offset reads make [`Self::value_ends`] worth building.
3785    ///
3786    /// As many reads as the dictionary has values. Building the table costs about thirty
3787    /// instructions a value once the fresh pages it lands in are counted, and a read out of it saves
3788    /// about thirty five, so it repays itself after roughly one read per value. The reads so far are
3789    /// the only guess there is at the reads to come, and waiting until they match the size of the
3790    /// dictionary is betting that a column read that much will be read that much again.
3791    ///
3792    /// A sixteenth was the first answer, from counting the unpacking alone at three instructions a
3793    /// value. ClickBench 38 showed what that missed: it reads about twenty thousand titles a
3794    /// statement out of a dictionary of three hundred and fifty thousand, crossed a sixteenth in its
3795    /// second statement and was two percent slower for a table it did not read enough to repay. A
3796    /// scan asking for the length of every row crosses it part way through its first statement on
3797    /// ClickBench, where a string column has about two rows for every value, and a filter that keeps
3798    /// a few thousand rows never does. The floor is there
3799    /// because a short dictionary would otherwise build a table for a handful of reads.
3800    fn ends_worth_unpacking(&self) -> usize {
3801        self.values.max(TEXT_PAYLOAD_VALUES)
3802    }
3803
3804    /// The unpacked ends, if they are built or if this read is the one that makes them worth it.
3805    fn value_ends(&self) -> Option<&[u32]> {
3806        if let Some(built) = self.value_ends.get() {
3807            return built.as_deref();
3808        }
3809        if self.ends_asked.fetch_add(1, Atomic::Relaxed) < self.ends_worth_unpacking() {
3810            return None;
3811        }
3812        self.value_ends.get_or_init(|| self.unpack_ends()).as_deref()
3813    }
3814
3815    /// Every end of the column, a run at a time.
3816    ///
3817    /// `None` rather than an error on anything wrong, because this is a cache in front of a reader
3818    /// that answers the same question. A column whose offsets are short or whose ends do not fit in
3819    /// four bytes gets no table and the same error it would have got, from the read that wanted it.
3820    fn unpack_ends(&self) -> Option<Vec<u32>> {
3821        let mut ends = vec![0u32; self.values];
3822        for (run, into) in ends.chunks_mut(TEXT_OFFSET_RUN).enumerate() {
3823            let bytes = self.offsets.get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)?;
3824            bitpack::unpack_tail_into(bytes, self.offset_bits, into, |bits| {
3825                u32::try_from(bits).unwrap_or(u32::MAX)
3826            })
3827            .ok()?;
3828        }
3829        // An end that did not fit was stored as the sentinel, and a real one cannot reach it because
3830        // a payload block is far smaller than four gigabytes. So the column keeps the packed reader.
3831        if ends.contains(&u32::MAX) { None } else { Some(ends) }
3832    }
3833
3834    /// Where the value at `index` ends inside its payload block.
3835    fn end_within(&self, index: usize) -> Result<u32> {
3836        if let Some(ends) = self.value_ends() {
3837            return ends
3838                .get(index)
3839                .copied()
3840                .ok_or_else(|| invalid("global dictionary offsets are short"));
3841        }
3842        let run = index / TEXT_OFFSET_RUN;
3843        let bytes = self
3844            .offsets
3845            .get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)
3846            .ok_or_else(|| invalid("global dictionary offsets are short"))?;
3847        let end = bitpack::tail_at(bytes, self.offset_bits, index % TEXT_OFFSET_RUN)
3848            .map_err(|_| invalid("global dictionary offsets are short"))?;
3849        u32::try_from(end).map_err(|_| invalid("global dictionary offset is past the payload"))
3850    }
3851
3852    /// Where every value in `first..last` ends inside its payload block, in one pass over the runs.
3853    ///
3854    /// [`Self::end_within`] answers for one value and pays for it twice over: it shifts a window to
3855    /// the bit the value starts at, and the copy that fills that window is a length the compiler does
3856    /// not know, so it is a call to `memcpy` rather than a load. A sweep asked for two of those per
3857    /// value, one for the end and one for the start that is the end before it, and on the ClickBench
3858    /// `URL` dictionary of eighteen million that was most of the half second a `LIKE` over it took.
3859    ///
3860    /// [`bitpack::unpack_tail_into`] walks the run instead, which makes the window a fixed width and
3861    /// so an unaligned load, and reads the bit position off a counter. A run is five hundred and
3862    /// twelve values and a block is two of them, so a block of a thousand and twenty four values
3863    /// costs two calls here and nothing per value.
3864    ///
3865    /// The answer is written straight into the result. A run that is wanted from its first value,
3866    /// which is every run but the one the sweep starts in, unpacks into its own window of the result
3867    /// and is never copied. Only a run joined part way through needs the scratch buffer, and there is
3868    /// at most one of those per sweep, so the buffer is allocated the first time one turns up.
3869    fn ends_within(&self, first: usize, last: usize) -> Result<Vec<u64>> {
3870        let mut ends = vec![0u64; last.saturating_sub(first)];
3871        let mut scratch = Vec::new();
3872        let mut at = first;
3873        while at < last {
3874            let run = at / TEXT_OFFSET_RUN;
3875            let stop = ((run + 1) * TEXT_OFFSET_RUN).min(last);
3876            let held = self.values.saturating_sub(run * TEXT_OFFSET_RUN).min(TEXT_OFFSET_RUN);
3877            let bytes = self
3878                .offsets
3879                .get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)
3880                .ok_or_else(|| invalid("global dictionary offsets are short"))?;
3881            let from = at % TEXT_OFFSET_RUN;
3882            let upto = stop - run * TEXT_OFFSET_RUN;
3883            if upto > held || bytes.len() < bitpack::tail_len(held, self.offset_bits) {
3884                return Err(invalid("global dictionary offsets are short"));
3885            }
3886            let into = &mut ends[at - first..stop - first];
3887            if from == 0 {
3888                bitpack::unpack_tail_into(bytes, self.offset_bits, into, |bits| bits)
3889                    .map_err(|_| invalid("global dictionary offsets are short"))?;
3890            } else {
3891                scratch.resize(held, 0);
3892                bitpack::unpack_tail_into(bytes, self.offset_bits, &mut scratch, |bits| bits)
3893                    .map_err(|_| invalid("global dictionary offsets are short"))?;
3894                into.copy_from_slice(&scratch[from..upto]);
3895            }
3896            at = stop;
3897        }
3898        Ok(ends)
3899    }
3900
3901    /// Where the value at `index` starts inside its payload block, which is where the value before
3902    /// it ended unless it is the first of the block.
3903    fn start_within(&self, index: usize) -> Result<u32> {
3904        if index % TEXT_PAYLOAD_VALUES == 0 { Ok(0) } else { self.end_within(index - 1) }
3905    }
3906
3907    /// Where the value at `index` starts and ends inside its payload block.
3908    ///
3909    /// The two offsets sit next to each other in the same run unless the value opens one, and a run
3910    /// of seventeen bit offsets, which is what a block of a thousand strings needs, puts a pair of
3911    /// them inside one eight byte load. So the common case reads the packed bytes once rather than
3912    /// twice and does the bounds arithmetic once. This is asked once per string a text column hands
3913    /// out, and on ClickBench 27 the two reads together were a quarter of the query.
3914    fn span_within(&self, index: usize) -> Result<(u32, u32)> {
3915        if let Some(ends) = self.value_ends() {
3916            let end =
3917                *ends.get(index).ok_or_else(|| invalid("global dictionary offsets are short"))?;
3918            // The value before it in the same block, and zero where there is no value before it.
3919            // `index` is inside the table, so the one under it is too.
3920            let start = if index % TEXT_PAYLOAD_VALUES == 0 { 0 } else { ends[index - 1] };
3921            if start > end {
3922                return Err(invalid("global dictionary value ends before it starts"));
3923            }
3924            return Ok((start, end));
3925        }
3926        let within = index % TEXT_OFFSET_RUN;
3927        let (start, end) = if within == 0 {
3928            (self.start_within(index)?, self.end_within(index)?)
3929        } else {
3930            let run = index / TEXT_OFFSET_RUN;
3931            let bytes = self
3932                .offsets
3933                .get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)
3934                .ok_or_else(|| invalid("global dictionary offsets are short"))?;
3935            let (start, end) = bitpack::tail_pair(bytes, self.offset_bits, within)
3936                .map_err(|_| invalid("global dictionary offsets are short"))?;
3937            let ends = u32::try_from(end)
3938                .map_err(|_| invalid("global dictionary offset is past the payload"))?;
3939            let starts = u32::try_from(start)
3940                .map_err(|_| invalid("global dictionary offset is past the payload"))?;
3941            (starts, ends)
3942        };
3943        if start > end {
3944            return Err(invalid("global dictionary value ends before it starts"));
3945        }
3946        Ok((start, end))
3947    }
3948
3949    /// The block of the sorted order that holds `rank`, and where in it that rank sits.
3950    ///
3951    /// The block is read from the file and checked against the hash the index carries for it the
3952    /// first time anything asks, and kept after that, the same way a payload block is. A search
3953    /// makes about as many probes as the order has bits, so the whole search reads a handful of
3954    /// these and never the rest.
3955    fn rank_parts(&self, rank: usize) -> Result<(&[u8], usize)> {
3956        let slot = self
3957            .rank_blocks
3958            .get(rank / TEXT_RANK_BLOCK)
3959            .ok_or_else(|| invalid("global dictionary rank is past the order"))?;
3960        let block = slot
3961            .get_or_init(|| {
3962                let which = rank / TEXT_RANK_BLOCK;
3963                let start = if which == 0 { 0 } else { self.rank_ends[which - 1] };
3964                let end = self.rank_ends[which];
3965                let mut bytes = vec![0; (end - start) as usize];
3966                read_at(&self.file, self.rank_at + start, &mut bytes)?;
3967                if checksum(&bytes)
3968                    != *self
3969                        .rank_hashes
3970                        .get(rank / TEXT_RANK_BLOCK)
3971                        .ok_or_else(|| invalid("global dictionary rank block has no checksum"))?
3972                {
3973                    return Err(invalid("global dictionary rank checksum differs"));
3974                }
3975                Ok(bytes)
3976            })
3977            .as_ref()
3978            .map_err(Clone::clone)?;
3979        Ok((block.as_slice(), rank % TEXT_RANK_BLOCK))
3980    }
3981
3982    /// The first eight bytes of the value at `rank`, as the integer a comparison reads.
3983    fn head_at(&self, rank: usize) -> Result<u64> {
3984        let (block, within) = self.rank_parts(rank)?;
3985        let (base, width, packed) = rank_heads(block)?;
3986        let above = bitpack::tail_at(packed, width, within)
3987            .map_err(|_| invalid("global dictionary rank block is short of heads"))?;
3988        Ok(base.wrapping_add(above))
3989    }
3990
3991    /// The packed codes of one rank block, which follow the heads on the next byte boundary.
3992    fn rank_codes<'block>(&self, block: &'block [u8], count: usize) -> Result<&'block [u8]> {
3993        let (_, width, packed) = rank_heads(block)?;
3994        packed
3995            .get(bitpack::tail_len(count, width)..)
3996            .ok_or_else(|| invalid("global dictionary rank block is short of codes"))
3997    }
3998
3999    /// How many entries the block holding `rank` has, which is a full block except at the end.
4000    fn rank_block_len(&self, rank: usize) -> usize {
4001        let first = rank / TEXT_RANK_BLOCK * TEXT_RANK_BLOCK;
4002        TEXT_RANK_BLOCK.min(self.ranks - first)
4003    }
4004}
4005
4006/// The base, the width and the packed bytes of one rank block's heads.
4007fn rank_heads(block: &[u8]) -> Result<(u64, usize, &[u8])> {
4008    let header = block
4009        .get(..RANK_BLOCK_HEADER)
4010        .ok_or_else(|| invalid("global dictionary rank block is short"))?;
4011    let base = u64::from_le_bytes(header[..8].try_into().expect("eight bytes"));
4012    let width = header[8] as usize;
4013    if width > 64 {
4014        return Err(invalid("global dictionary rank block packs heads past a word"));
4015    }
4016    Ok((base, width, &block[RANK_BLOCK_HEADER..]))
4017}
4018
4019/// Bits one offset of a dictionary takes, which is what its widest payload block spans.
4020///
4021/// One width for the whole column rather than one a block. A block is 1,024 values of the same
4022/// column, so the blocks of a column are within a factor of two of each other on every ClickBench
4023/// string column, and a width a block would save a fraction of a bit and cost a byte a block plus
4024/// the arithmetic that finds where a block starts.
4025fn offset_width(ends: &[u32]) -> usize {
4026    // The ends are already relative to the block the value is in, so the last end of a block is that
4027    // block's total and the largest end anywhere is the widest block. There is no subtraction left
4028    // to do and no need to walk the blocks to find where one starts.
4029    let span = ends.iter().copied().max().unwrap_or(0);
4030    (u32::BITS - span.leading_zeros()) as usize
4031}
4032
4033/// How many bytes `values` offsets take at `bits`, which is what the reader has to know before it
4034/// has read any of them.
4035fn offset_bytes(values: usize, bits: usize) -> usize {
4036    let full = values / TEXT_OFFSET_RUN;
4037    let rest = values % TEXT_OFFSET_RUN;
4038    full * TEXT_OFFSET_RUN / 8 * bits + bitpack::tail_len(rest, bits)
4039}
4040
4041/// The end of every value within its payload block, packed a run at a time.
4042/// A run never straddles a block, because [`TEXT_OFFSET_RUN`] divides [`TEXT_PAYLOAD_VALUES`], which
4043/// is what lets this be a walk of the ends rather than arithmetic against a per block base.
4044fn encode_offsets(ends: &[u32], bits: usize, out: &mut Vec<u8>) -> Result<()> {
4045    let mut run = Vec::with_capacity(TEXT_OFFSET_RUN);
4046    for chunk in ends.chunks(TEXT_OFFSET_RUN) {
4047        run.clear();
4048        run.extend(chunk.iter().map(|&end| u64::from(end)));
4049        bitpack::pack_tail(&run, bits, out)
4050            .map_err(|_| invalid("global dictionary offsets do not pack"))?;
4051    }
4052    Ok(())
4053}
4054
4055/// How many bits a code of a dictionary of `values` entries takes.
4056fn code_width(values: usize) -> usize {
4057    match u64::try_from(values).unwrap_or(u64::MAX) {
4058        0 | 1 => 0,
4059        last => (u64::BITS - (last - 1).leading_zeros()) as usize,
4060    }
4061}
4062
4063impl TextSource for NativeText {
4064    fn len(&self) -> usize {
4065        self.values
4066    }
4067
4068    fn might_contain(&self, first: usize, literal: &[u8]) -> Result<bool> {
4069        let Some(grams) = &self.grams else { return Ok(true) };
4070        if literal.len() < 4 || first >= self.values {
4071            return Ok(true);
4072        }
4073        let bytes = grams
4074            .loaded
4075            .get_or_init(|| {
4076                let mut bytes = vec![0; grams.length];
4077                read_at(&self.file, grams.start, &mut bytes)?;
4078                if checksum(&bytes) != grams.hash {
4079                    return Err(invalid("global dictionary substring signatures checksum differs"));
4080                }
4081                Ok(bytes)
4082            })
4083            .as_ref()
4084            .map_err(Clone::clone)?;
4085        let block = first / TEXT_PAYLOAD_VALUES;
4086        let Some(bits) = bytes.get(block * TEXT_GRAM_BYTES..(block + 1) * TEXT_GRAM_BYTES) else {
4087            return Ok(true);
4088        };
4089        Ok(literal.windows(4).all(|gram| {
4090            gram_bits(gram).into_iter().all(|bit| bits[bit / 8] & (1 << (bit % 8)) != 0)
4091        }))
4092    }
4093
4094    fn bytes_at(&self, index: usize) -> Result<Option<&[u8]>> {
4095        if index >= self.values {
4096            return Ok(None);
4097        }
4098        let (start, end) = self.span_within(index)?;
4099        if start == end {
4100            return Ok(Some(&[]));
4101        }
4102        // A block holds a fixed number of values rather than a fixed number of bytes, so the value
4103        // is in one block and the offsets already say where in it.
4104        let block = index / TEXT_PAYLOAD_VALUES;
4105        let Some(bytes) = self.payload_block(block)? else { return Ok(None) };
4106        Ok(bytes.get(start as usize..end as usize))
4107    }
4108
4109    fn bytes_len_at(&self, index: usize) -> Result<Option<usize>> {
4110        if index >= self.values {
4111            return Ok(None);
4112        }
4113        let (start, end) = self.span_within(index)?;
4114        Ok(Some((end - start) as usize))
4115    }
4116
4117    /// Every length out of the unpacked ends in one loop, which is the point of having them.
4118    ///
4119    /// The whole run of positions counts towards [`Self::ends_worth_unpacking`] at once, because a
4120    /// caller asking for a vector of lengths has said how many it wants, and a vector of them is
4121    /// usually enough on its own. Until the table is worth building this is the row at a time read,
4122    /// the same as the default.
4123    fn bytes_lens_at(&self, indices: &[u32], into: &mut [i64]) -> Result<()> {
4124        self.ends_asked.fetch_add(indices.len(), Atomic::Relaxed);
4125        let Some(ends) = self.value_ends() else {
4126            for (slot, &index) in into.iter_mut().zip(indices) {
4127                *slot = self
4128                    .bytes_len_at(index as usize)?
4129                    .map_or(0, |len| i64::try_from(len).unwrap_or(i64::MAX));
4130            }
4131            return Ok(());
4132        };
4133        if let Some(lens) = self.value_lens.get_or_init(|| lengths_of(ends)) {
4134            for (slot, &index) in into.iter_mut().zip(indices) {
4135                // Past the end is no value and so no length, which is what a row at a time read
4136                // says.
4137                *slot = lens.get(index as usize).map_or(0, |&len| i64::from(len));
4138            }
4139            return Ok(());
4140        }
4141        for (slot, &index) in into.iter_mut().zip(indices) {
4142            let index = index as usize;
4143            // Past the end is no value and so no length, which is what a row at a time read says.
4144            let Some(&end) = ends.get(index) else {
4145                *slot = 0;
4146                continue;
4147            };
4148            let start = if index % TEXT_PAYLOAD_VALUES == 0 { 0 } else { ends[index - 1] };
4149            if start > end {
4150                return Err(invalid("global dictionary value ends before it starts"));
4151            }
4152            *slot = i64::from(end - start);
4153        }
4154        Ok(())
4155    }
4156
4157    /// The rest of the block holding `first`, decoded into a buffer that may die with the call.
4158    ///
4159    /// A block is the unit this format decodes, so a walk that wants every value is going to decode
4160    /// every block whatever it does. The question is whether it keeps them, and both answers are
4161    /// wrong on their own. [`Self::payload_block`] keeps every block it is asked for, so a reader
4162    /// that walked the whole dictionary through `bytes_at` ended up holding the whole dictionary
4163    /// decoded, 4.2 GB on ClickBench `URL`. Keeping none of them makes the next statement asking
4164    /// the same question decode all of it again, which on the same column at a million rows is a
4165    /// `LIKE` going from 2.7 ms to 16.2 ms.
4166    ///
4167    /// So a sweep keeps what it decodes for the second time while the column is under
4168    /// [`TEXT_KEEP_BUDGET`] and drops it after that. A block already in hand is used where it is there and costs nothing either way.
4169    fn sweep(
4170        &self,
4171        first: usize,
4172        limit: usize,
4173        body: &mut dyn FnMut(usize, &[u8]) -> Result<()>,
4174    ) -> Result<usize> {
4175        let limit = limit.min(self.values);
4176        if first >= limit {
4177            return Ok(first);
4178        }
4179        let block = first / TEXT_PAYLOAD_VALUES;
4180        let last = ((block + 1) * TEXT_PAYLOAD_VALUES).min(limit);
4181        let decoded;
4182        let kept = self.blocks.get(block).and_then(OnceLock::get);
4183        let again = kept.is_none()
4184            && self.swept.get(block).is_some_and(|swept| swept.swap(true, Atomic::Relaxed));
4185        let bytes: &[u8] = match kept {
4186            Some(Ok(kept)) => kept,
4187            _ if again && self.payload_kept.load(Atomic::Relaxed) < self.keep_budget => {
4188                let kept = self
4189                    .payload_block(block)?
4190                    .ok_or_else(|| invalid("global dictionary block is past the payload"))?;
4191                self.payload_kept.fetch_add(kept.len(), Atomic::Relaxed);
4192                kept
4193            }
4194            _ => {
4195                decoded = self.decode_block(block)?;
4196                &decoded
4197            }
4198        };
4199        let ends = self.ends_within(first, last)?;
4200        if ends.len() != last - first {
4201            return Err(invalid("global dictionary offsets are short"));
4202        }
4203        let mut start = u64::from(self.start_within(first)?);
4204        // row at a time: the caller is handed one value after another, and what it does with one is
4205        // its own business, so there is no shape here for anything but a walk.
4206        for (index, &end) in (first..last).zip(&ends) {
4207            let value = usize::try_from(start)
4208                .ok()
4209                .zip(usize::try_from(end).ok())
4210                .and_then(|(from, to)| bytes.get(from..to))
4211                .ok_or_else(|| invalid("global dictionary value is past its block"))?;
4212            body(index, value)?;
4213            start = end;
4214        }
4215        Ok(last)
4216    }
4217
4218    /// Each block the indices land in, decoded once and dropped, or read where it is already kept.
4219    ///
4220    /// Never kept, unlike [`Self::sweep`] under its budget, because a scattered read is a one off:
4221    /// a synopsis turned into values is turned once and remembered by the reader as values, a few
4222    /// kilobytes, where the blocks it went through are megabytes nobody asks for again.
4223    fn visit(
4224        &self,
4225        indices: &[usize],
4226        body: &mut dyn FnMut(usize, &[u8]) -> Result<()>,
4227    ) -> Result<()> {
4228        let mut at = 0;
4229        while at < indices.len() {
4230            let block = indices[at] / TEXT_PAYLOAD_VALUES;
4231            let upto =
4232                at + indices[at..].partition_point(|&index| index / TEXT_PAYLOAD_VALUES == block);
4233            let wanted = &indices[at..upto];
4234            if wanted.iter().any(|&index| index >= self.values) {
4235                return Err(invalid("a visited value is past the global dictionary"));
4236            }
4237            let decoded;
4238            let bytes: &[u8] = match self.blocks.get(block).and_then(OnceLock::get) {
4239                Some(Ok(kept)) => kept,
4240                _ => {
4241                    decoded = self.decode_block(block)?;
4242                    &decoded
4243                }
4244            };
4245            for (offset, &index) in wanted.iter().enumerate() {
4246                let (start, end) = self.span_within(index)?;
4247                let value = bytes
4248                    .get(start as usize..end as usize)
4249                    .ok_or_else(|| invalid("global dictionary value is past its block"))?;
4250                body(at + offset, value)?;
4251            }
4252            at = upto;
4253        }
4254        Ok(())
4255    }
4256
4257    fn ranks(&self) -> Option<usize> {
4258        (self.ranks > 0).then_some(self.ranks)
4259    }
4260
4261    /// The boundary for `wanted`, out of [`Self::searched`] where it is there and put there where
4262    /// it is not.
4263    ///
4264    /// The lock is held over the search rather than dropped and taken again, so that two threads
4265    /// asking for the same value at the same time do the work once between them. That is the shape
4266    /// the scan actually arrives in: sixteen instances of a top N, all reading the same column, all
4267    /// improving their bound over the same early chunks.
4268    fn below(&self, ranks: usize, wanted: &[u8]) -> Result<(usize, bool)> {
4269        let mut memo = self.searched.lock().map_err(|_| invalid("a poisoned dictionary search"))?;
4270        if let Some(&answer) = memo.get(wanted) {
4271            return Ok(answer);
4272        }
4273        let answer = search_below(self, ranks, wanted)?;
4274        if memo.len() >= TEXT_SEARCH_MEMO {
4275            memo.clear();
4276        }
4277        memo.insert(wanted.to_vec(), answer);
4278        Ok(answer)
4279    }
4280
4281    fn compare_rank(&self, rank: usize, wanted: &[u8]) -> Result<Ordering> {
4282        // The head settles the probe unless the two values start with the same eight bytes, and
4283        // only then is a value read. On a column of URLs that is the difference between a search
4284        // that touches one block of the payload and a search that touches nineteen of them.
4285        let settled = self.head_at(rank)?.cmp(&head(wanted));
4286        if settled != Ordering::Equal {
4287            return Ok(settled);
4288        }
4289        let code = self.code_at_rank(rank)?;
4290        let bytes = self
4291            .bytes_at(code as usize)?
4292            .ok_or_else(|| invalid("global dictionary order names a code it does not have"))?;
4293        Ok(bytes.cmp(wanted))
4294    }
4295
4296    fn code_at_rank(&self, rank: usize) -> Result<u32> {
4297        let (block, within) = self.rank_parts(rank)?;
4298        let codes = self.rank_codes(block, self.rank_block_len(rank))?;
4299        let code = bitpack::tail_at(codes, self.code_bits, within)
4300            .map_err(|_| invalid("global dictionary rank block is short of codes"))?;
4301        let code = u32::try_from(code)
4302            .map_err(|_| invalid("global dictionary order names a code it does not have"))?;
4303        if code as usize >= self.len() {
4304            return Err(invalid("global dictionary order names a code it does not have"));
4305        }
4306        Ok(code)
4307    }
4308
4309    fn code_ranks(&self) -> Option<&[u32]> {
4310        // The order is a permutation of the positions, so inverting it needs every position to be
4311        // named exactly once. Anything else and the slice would have holes, and a caller indexing
4312        // it by a code would read a rank that belongs to nothing.
4313        if self.ranks == 0 || self.ranks != self.len() {
4314            return None;
4315        }
4316        self.code_ranks
4317            .get_or_init(|| {
4318                let mut ranks = vec![u32::MAX; self.ranks];
4319                // A block at a time rather than a rank at a time, because reading it per rank pays
4320                // for the bounds check, the division and the lock on every one of them.
4321                for first in (0..self.ranks).step_by(TEXT_RANK_BLOCK) {
4322                    let (block, _) = self.rank_parts(first).ok()?;
4323                    let count = self.rank_block_len(first);
4324                    let codes = self.rank_codes(block, count).ok()?;
4325                    for (within, code) in bitpack::unpack_tail(codes, self.code_bits, count)
4326                        .ok()?
4327                        .into_iter()
4328                        .enumerate()
4329                    {
4330                        let code = usize::try_from(code).ok()?;
4331                        *ranks.get_mut(code)? = u32::try_from(first + within).ok()?;
4332                    }
4333                }
4334                if ranks.contains(&u32::MAX) {
4335                    return None;
4336                }
4337                Some(ranks)
4338            })
4339            .as_deref()
4340    }
4341
4342    fn footprint(&self) -> usize {
4343        self.offsets.capacity()
4344            + self
4345                .value_ends
4346                .get()
4347                .and_then(Option::as_ref)
4348                .map_or(0, |ends| ends.capacity() * size_of::<u32>())
4349            + self
4350                .value_lens
4351                .get()
4352                .and_then(Option::as_ref)
4353                .map_or(0, |lens| lens.capacity() * size_of::<u32>())
4354            + self
4355                .code_ranks
4356                .get()
4357                .and_then(Option::as_ref)
4358                .map_or(0, |ranks| ranks.capacity() * size_of::<u32>())
4359            + self.rank_hashes.capacity() * size_of::<u64>()
4360            + self.rank_ends.capacity() * size_of::<u64>()
4361            + self.rank_blocks.capacity() * size_of::<OnceLock<Result<Vec<u8>>>>()
4362            + self
4363                .rank_blocks
4364                .iter()
4365                .filter_map(OnceLock::get)
4366                .filter_map(|result| result.as_ref().ok())
4367                .map(Vec::capacity)
4368                .sum::<usize>()
4369            + self.blocks.capacity() * size_of::<OnceLock<Result<Vec<u8>>>>()
4370            + self.hashes.capacity() * size_of::<u64>()
4371            + self.starts.capacity() * size_of::<u64>()
4372            + self.lengths.capacity() * size_of::<u64>()
4373            + self
4374                .grams
4375                .as_ref()
4376                .and_then(|grams| grams.loaded.get())
4377                .and_then(|result| result.as_ref().ok())
4378                .map_or(0, Vec::capacity)
4379            + self
4380                .blocks
4381                .iter()
4382                .filter_map(OnceLock::get)
4383                .filter_map(|result| result.as_ref().ok())
4384                .map(Vec::capacity)
4385                .sum::<usize>()
4386    }
4387}
4388
4389/// Every table wide part number in order, with the stripe it belongs to.
4390fn places(table: &Table) -> Result<Vec<Place>> {
4391    let mut places = Vec::with_capacity(table.stripes.len().saturating_mul(STRIPE_PARTS));
4392    for (at, stripe) in table.stripes.iter().enumerate() {
4393        let index = u32::try_from(at).map_err(|_| invalid("too many stripes"))?;
4394        for (part, &rows) in stripe.parts.iter().enumerate() {
4395            places.push(Place {
4396                stripe: index,
4397                part: u32::try_from(part).map_err(|_| invalid("too many parts in a stripe"))?,
4398                rows,
4399            });
4400        }
4401    }
4402    Ok(places)
4403}
4404
4405/// Reads one column's section of a stripe's index page.
4406///
4407/// The section carries its own checksum, so a reader that wants one column out of a hundred and
4408/// five preads a few hundred bytes and still knows that what it got is what was written.
4409fn read_index(file: &File, stripe: &Stripe, column: usize) -> Result<Vec<PartSpan>> {
4410    let parts = stripe.parts.len();
4411    let section = index_section(parts)?;
4412    let at = column.checked_mul(section).ok_or_else(|| invalid("index page offset overflow"))?;
4413    let end = at.checked_add(section).ok_or_else(|| invalid("index page offset overflow"))?;
4414    if end > stripe.index.length as usize {
4415        return Err(invalid("index page is shorter than its columns"));
4416    }
4417    let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
4418    let mut bytes = vec![0; section];
4419    let offset = stripe
4420        .index
4421        .offset
4422        .checked_add(at as u64)
4423        .ok_or_else(|| invalid("index page offset overflow"))?;
4424    read_at(file, offset, &mut bytes)?;
4425    let entries = section - size_of::<u64>();
4426    let stored = u64::from_le_bytes(bytes[entries..].try_into().expect("eight bytes"));
4427    if checksum(&bytes[..entries]) != stored {
4428        // With where it was read from, because the two ways this fires look identical from the
4429        // message alone: a file somebody damaged, and a file we wrote to the wrong offset.
4430        return Err(invalid(&format!(
4431            "index page section checksum differs, column {column} of {parts} parts at {offset}, \
4432             wanted {stored:016x} and got {:016x}",
4433            checksum(&bytes[..entries]),
4434        )));
4435    }
4436    let mut spans = Vec::with_capacity(parts);
4437    let mut start = 0_usize;
4438    for part in 0..parts {
4439        let at = part * INDEX_ENTRY;
4440        let length = u32::from_le_bytes(bytes[at..at + 4].try_into().expect("four bytes")) as usize;
4441        let hash = u64::from_le_bytes(bytes[at + 4..at + 12].try_into().expect("eight bytes"));
4442        spans.push(PartSpan { start, length, hash });
4443        start = start.checked_add(length).ok_or_else(|| invalid("column page length overflow"))?;
4444    }
4445    if start != page.length as usize {
4446        return Err(invalid("column page length differs from its index"));
4447    }
4448    Ok(spans)
4449}
4450
4451/// One part's bytes out of a whole column page.
4452fn part_bytes(page: &[u8], span: PartSpan) -> Result<&[u8]> {
4453    let end = span.start.checked_add(span.length).ok_or_else(|| invalid("part range overflow"))?;
4454    page.get(span.start..end).ok_or_else(|| invalid("part exceeds its column page"))
4455}
4456
4457/// Puts one stripe of one column in the cache, and hands back the page for the pool to count when
4458/// it is a page the column did not already hold.
4459///
4460/// The index goes in its own slot and stays. Only the page is under the budget, and the pool is
4461/// what enforces it, once the caller has let go of the column's lock.
4462fn remember(cached: &mut Cached, held: &CachedColumn) -> Option<(usize, Arc<AtomicBool>)> {
4463    if let Some(slot) = cached.index.get_mut(held.stripe) {
4464        if slot.is_none() {
4465            *slot = Some(Arc::clone(&held.index));
4466        }
4467    }
4468    let page = held.page.clone()?;
4469    let slot = cached.pages.get_mut(held.stripe)?;
4470    if slot.is_some() {
4471        return None;
4472    }
4473    let bytes = page.len();
4474    // Set, so that the page a worker has just paid to read is not the one the pass it pays for
4475    // lets go of before the worker has read a part out of it.
4476    let used = Arc::new(AtomicBool::new(true));
4477    *slot = Some(Resident { page, used: Arc::clone(&used) });
4478    Some((bytes, used))
4479}
4480
4481/// Every table a native file holds, without the directory of any of them.
4482///
4483/// This is what opening a database reads. It is the small level of the directory, so the cost is
4484/// proportional to how many tables there are rather than to how much data they hold, and a session
4485/// that touches two tables of eight decodes two table directories.
4486///
4487/// The file handle is shared with every reader this hands out. Eight tables in one file is one open
4488/// file descriptor, not eight, which is the other thing one file buys over a file per table.
4489#[derive(Debug, Clone)]
4490pub struct Catalog {
4491    file: Arc<File>,
4492    size: u64,
4493    entries: Arc<Vec<Entry>>,
4494    /// The views the file holds, whole, since a view has no second level to read later.
4495    views: Arc<Vec<ViewEntry>>,
4496    opening: Opening,
4497    /// Where every reader this hands out counts its pages.
4498    pool: PagePool,
4499}
4500
4501/// Signed integer sums and non-null counts for selected columns, plus total table rows.
4502#[derive(Debug, Clone, PartialEq, Eq)]
4503pub struct CertifiedSums {
4504    pub columns: Vec<(i128, u64)>,
4505    pub rows: u64,
4506}
4507
4508impl Catalog {
4509    /// Reads the highest valid catalog slot and nothing under it.
4510    ///
4511    /// The readers it hands out keep pages in a pool of their own with no budget, so each column
4512    /// holds its floor of four stripes and no more. A database opens with [`Catalog::open_in`].
4513    ///
4514    /// # Errors
4515    ///
4516    /// If the file has no valid committed catalog or a catalog pointer is out of bounds.
4517    pub fn open(path: impl AsRef<Path>) -> Result<Self> {
4518        Self::open_in(path, &PagePool::default())
4519    }
4520
4521    /// The same, with every reader it hands out keeping its pages in `pool`.
4522    ///
4523    /// # Errors
4524    ///
4525    /// If the file has no valid committed catalog or a catalog pointer is out of bounds.
4526    pub fn open_in(path: impl AsRef<Path>, pool: &PagePool) -> Result<Self> {
4527        let (file, size, _, bytes, opening) = slot_bytes(path)?;
4528        let (entries, views) = decode_catalog(&bytes, size)?;
4529        Ok(Self {
4530            file: Arc::new(file),
4531            size,
4532            entries: Arc::new(entries),
4533            views: Arc::new(views),
4534            opening,
4535            pool: pool.clone(),
4536        })
4537    }
4538
4539    /// The tables in the file, in the order they were written.
4540    pub fn names(&self) -> impl ExactSizeIterator<Item = &str> {
4541        self.entries.iter().map(|entry| entry.name.as_str())
4542    }
4543
4544    /// The same tables with how many rows each of them holds.
4545    ///
4546    /// The names alone answer which tables the file has, which is what a checkpoint needs to know.
4547    /// A load asks a second question: whether a table already in the file is really in the way of
4548    /// the one it wants to write. A table with no rows is not, because it has no pages the next
4549    /// generation would have to carry, so the count has to come out of the catalog beside the name.
4550    pub fn rows(&self) -> impl ExactSizeIterator<Item = (&str, usize)> {
4551        self.entries.iter().map(|entry| (entry.name.as_str(), entry.rows))
4552    }
4553
4554    /// The views in the file, in the order they were written.
4555    ///
4556    /// Whole, unlike [`Catalog::names`], which hands back names and makes the caller ask for a table
4557    /// by one. A view is a few strings and a column list and it was all read at open, so there is
4558    /// nothing left to go and fetch and no reason to make the caller ask twice.
4559    pub fn views(&self) -> impl ExactSizeIterator<Item = &ViewEntry> {
4560        self.views.iter()
4561    }
4562
4563    /// How many tables the file holds.
4564    #[must_use]
4565    pub fn len(&self) -> usize {
4566        self.entries.len()
4567    }
4568
4569    /// Whether the file holds no table at all, which is what [`Writer::empty`] writes and what a
4570    /// database somebody dropped the last table out of comes back as.
4571    #[must_use]
4572    pub fn is_empty(&self) -> bool {
4573        self.entries.is_empty()
4574    }
4575
4576    /// Opens one table by name, decoding its directory now.
4577    ///
4578    /// # Errors
4579    ///
4580    /// If there is no table by that name, or its directory is torn or points outside the file.
4581    pub fn table(&self, name: &str) -> Result<Reader> {
4582        let entry = self
4583            .entries
4584            .iter()
4585            .find(|entry| entry.name == name)
4586            .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
4587        // Checked and then decoded a window at a time, so that the directory's own bytes are never
4588        // all in memory beside the table they decode into. It is read twice, and the second read
4589        // comes out of the page cache the first one filled.
4590        let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
4591        if file_checksum(&self.file, offset, length)? != entry.directory.hash {
4592            return Err(invalid(&format!("the directory of table {name} does not checksum")));
4593        }
4594        let mut opening = self.opening;
4595        opening.reads += 1;
4596        opening.bytes += u64::from(entry.directory.length);
4597        Reader::build(
4598            Arc::clone(&self.file),
4599            self.size,
4600            read_directory(Cursor::over(&self.file, offset, length), self.size, Some(offset))?,
4601            u64::from(entry.directory.length),
4602            opening,
4603            self.pool.clone(),
4604        )
4605    }
4606
4607    /// Counts non-null, nonzero values from a validated native directory without building a
4608    /// reader for every stripe. Returns `None` when the bounded frequency synopsis cannot prove
4609    /// the count, so callers can use the ordinary query path.
4610    pub fn nonzero_count(&self, name: &str, column: usize) -> Result<Option<u64>> {
4611        let entry = self
4612            .entries
4613            .iter()
4614            .find(|entry| entry.name == name)
4615            .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
4616        let Some(field) = entry.fields.get(column) else {
4617            return Err(invalid("frequency column index out of range"));
4618        };
4619        if !matches!(
4620            field.ty,
4621            LogicalType::TinyInt
4622                | LogicalType::SmallInt
4623                | LogicalType::Integer
4624                | LogicalType::BigInt
4625                | LogicalType::UTinyInt
4626                | LogicalType::USmallInt
4627                | LogicalType::UInteger
4628                | LogicalType::UBigInt
4629        ) {
4630            return Ok(None);
4631        }
4632        let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
4633        if file_checksum(&self.file, offset, length)? != entry.directory.hash {
4634            return Err(invalid(&format!("the directory of table {name} does not checksum")));
4635        }
4636        if let Some(count) = entry.nonzero.get(column).copied().flatten() {
4637            return Ok(Some(count));
4638        }
4639        quick_nonzero(
4640            Cursor::over(&self.file, offset, length),
4641            &entry.name,
4642            &entry.fields,
4643            entry.rows,
4644            column,
4645        )
4646    }
4647
4648    /// Exact signed-integer sums and non-null counts from the small catalog. The table directory
4649    /// checksum is still checked once before any certificate can answer a query.
4650    pub fn aggregate_sums(&self, name: &str, columns: &[usize]) -> Result<Option<CertifiedSums>> {
4651        let entry = self
4652            .entries
4653            .iter()
4654            .find(|entry| entry.name == name)
4655            .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
4656        let mut sums = Vec::with_capacity(columns.len());
4657        for &column in columns {
4658            let Some(field) = entry.fields.get(column) else {
4659                return Err(invalid("aggregate column index out of range"));
4660            };
4661            if !signed_integer(&field.ty) {
4662                return Ok(None);
4663            }
4664            let Some(sum) = entry.aggregates[column] else {
4665                return Ok(None);
4666            };
4667            sums.push(sum);
4668        }
4669        let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
4670        if file_checksum(&self.file, offset, length)? != entry.directory.hash {
4671            return Err(invalid(&format!("the directory of table {name} does not checksum")));
4672        }
4673        Ok(Some(CertifiedSums { columns: sums, rows: entry.rows as u64 }))
4674    }
4675
4676    /// The schema copied into the small file catalog, available without opening the table directory.
4677    pub fn table_fields(&self, name: &str) -> Option<&[Field]> {
4678        self.entries.iter().find(|entry| entry.name == name).map(|entry| entry.fields.as_slice())
4679    }
4680}
4681
4682/// Where the slot naming `generation` goes, which is the one the generation before it did not use.
4683///
4684/// Generation 1 takes the slot at 16, so a file written once is byte for byte the file this wrote
4685/// before there was a second generation to write.
4686fn slot_offset(generation: u64) -> u64 {
4687    16 + (generation - 1) % 2 * SLOT_BYTES as u64
4688}
4689
4690/// The header and the bytes the highest valid slot points at.
4691///
4692/// Both levels of the directory are reached this way, so the magic check, the version check and the
4693/// choice between the two slots live here rather than being written out twice.
4694fn slot_bytes(path: impl AsRef<Path>) -> Result<(File, u64, Slot, Vec<u8>, Opening)> {
4695    let mut file = File::open(path).map_err(io)?;
4696    let size = file.metadata().map_err(io)?.len();
4697    if size < HEADER {
4698        return Err(invalid("file is shorter than its header"));
4699    }
4700    let mut header = [0; HEADER as usize];
4701    file.read_exact(&mut header).map_err(io)?;
4702    let mut opening = Opening { reads: 1, bytes: HEADER };
4703    let version = u32::from_le_bytes([header[8], header[9], header[10], header[11]]);
4704    // The two halves are worth telling apart. A wrong magic is a file that was never ours and
4705    // the answer is to look at the path. A wrong version is our own file from another build,
4706    // and the number this build wants is the only thing that tells the reader whether to
4707    // rebuild the file or to go back to the binary that wrote it.
4708    if &header[..8] != MAGIC {
4709        return Err(invalid("the header does not begin with a rudb native magic"));
4710    }
4711    if !READABLE.contains(&version) {
4712        return Err(invalid(&format!(
4713            "the file is format {version} and this build reads format {FORMAT}, so it has to \
4714                 be written again"
4715        )));
4716    }
4717    let mut selected = None;
4718    for start in [16, 16 + SLOT_BYTES] {
4719        let slot = Slot::read(&header[start..start + SLOT_BYTES]);
4720        if slot.generation == 0 || slot.length == 0 || slot.length as usize > MAX_DIRECTORY {
4721            continue;
4722        }
4723        let Some(end) = slot.offset.checked_add(u64::from(slot.length)) else { continue };
4724        if slot.offset < HEADER || end > size {
4725            continue;
4726        }
4727        let mut bytes = vec![0; slot.length as usize];
4728        file.seek(SeekFrom::Start(slot.offset)).map_err(io)?;
4729        file.read_exact(&mut bytes).map_err(io)?;
4730        opening.reads += 1;
4731        opening.bytes += u64::from(slot.length);
4732        if checksum(&bytes) == slot.hash
4733            && selected
4734                .as_ref()
4735                .is_none_or(|(old, _): &(Slot, Vec<u8>)| old.generation < slot.generation)
4736        {
4737            selected = Some((slot, bytes));
4738        }
4739    }
4740    let (slot, bytes) = selected.ok_or_else(|| invalid("no committed directory slot is valid"))?;
4741    Ok((file, size, slot, bytes, opening))
4742}
4743
4744impl Reader {
4745    /// Opens a file that holds exactly one table.
4746    ///
4747    /// # Errors
4748    ///
4749    /// If the file has no valid committed directory, a directory pointer is out of bounds, or the
4750    /// file holds more than one table, which is a file that has to be opened by name.
4751    pub fn open(path: impl AsRef<Path>) -> Result<Self> {
4752        let catalog = Catalog::open(path)?;
4753        let mut names = catalog.names();
4754        let name = names.next().ok_or_else(|| invalid("the file holds no table"))?.to_string();
4755        if names.next().is_some() {
4756            return Err(invalid(
4757                "the file holds more than one table, so it has to be opened by name",
4758            ));
4759        }
4760        catalog.table(&name)
4761    }
4762
4763    /// Builds a reader over one decoded table directory.
4764    fn build(
4765        file: Arc<File>,
4766        size: u64,
4767        table: Table,
4768        directory: u64,
4769        opening: Opening,
4770        pool: PagePool,
4771    ) -> Result<Self> {
4772        let places = places(&table)?;
4773        let dictionaries = (0..table.fields.len()).map(|_| OnceLock::new()).collect();
4774        let table_fields = table.fields.len();
4775        let stripes = table.stripes.len();
4776        let columns = (0..table.fields.len())
4777            .map(|_| {
4778                Mutex::new(Cached {
4779                    pages: (0..stripes).map(|_| None).collect(),
4780                    index: (0..stripes).map(|_| None).collect(),
4781                    ..Cached::default()
4782                })
4783            })
4784            .collect::<Vec<_>>();
4785        let cache = Shelf {
4786            columns,
4787            held: (0..table_fields).map(|_| AtomicUsize::new(0)).collect(),
4788            kept: AtomicUsize::new(CACHED_STRIPES_PER_COLUMN),
4789        };
4790        let sieves: Vec<Vec<SieveSlot>> = (0..table.fields.len())
4791            .map(|_| table.stripes.iter().map(|_| OnceLock::new()).collect())
4792            .collect();
4793        let part_ranges: Vec<Vec<RangeSlot>> = (0..table.fields.len())
4794            .map(|_| table.stripes.iter().map(|_| OnceLock::new()).collect())
4795            .collect();
4796        Ok(Self {
4797            file,
4798            table: Arc::new(table),
4799            dictionaries: Arc::new(dictionaries),
4800            loading: Arc::new((0..table_fields).map(|_| Mutex::new(())).collect()),
4801            frequency_values: Arc::new((0..table_fields).map(|_| OnceLock::new()).collect()),
4802            frequency_summaries: Arc::new((0..table_fields).map(|_| OnceLock::new()).collect()),
4803            opened: Arc::new(AtomicUsize::new(0)),
4804            sieves: Arc::new(sieves),
4805            part_ranges: Arc::new(part_ranges),
4806            places: Arc::new(places),
4807            cache: Arc::new(cache),
4808            pool,
4809            pages: Arc::new(AtomicUsize::new(0)),
4810            indexes: Arc::new(AtomicUsize::new(0)),
4811            size,
4812            directory,
4813            opening,
4814        })
4815    }
4816
4817    /// What this reader has read so far, and what opening it cost.
4818    ///
4819    /// Public because the claim of `spec/stats/04-in-memory.md` section 4.2 is about this number
4820    /// and a claim nobody can check is a comment. A caller that wants to know whether opening a
4821    /// file touched the data asks here, and gets an answer that does not depend on what the page
4822    /// cache happened to hold.
4823    #[must_use]
4824    pub fn reads(&self) -> Reads {
4825        Reads {
4826            opening: self.opening,
4827            pages: self.pages.load(Atomic::Relaxed),
4828            indexes: self.indexes.load(Atomic::Relaxed),
4829            dictionaries: self.opened.load(Atomic::Relaxed),
4830        }
4831    }
4832
4833    /// Where the file's bytes went, from the directory alone.
4834    ///
4835    /// No page is read, so this costs the same on a 45 GB table as on an empty one. See [`Layout`]
4836    /// for what is charged where and for why the three things that are not columns stay separate.
4837    #[must_use]
4838    pub fn layout(&self) -> Layout {
4839        let table = &self.table;
4840        let stripes = table.stripes.as_slice();
4841        let columns = table
4842            .fields
4843            .iter()
4844            .enumerate()
4845            .map(|(at, field)| ColumnLayout {
4846                name: field.name.clone(),
4847                kind: field.ty.to_string(),
4848                pages: sum(stripes.iter().map(|stripe| span_bytes(&stripe.pages, at))),
4849                memberships: sum(stripes.iter().map(|stripe| stripe.memberships.bytes(at))),
4850                sieves: sum(stripes.iter().map(|stripe| stripe.sieves.bytes(at))),
4851                part_ranges: sum(stripes.iter().map(|stripe| stripe.part_ranges.bytes(at))),
4852                dictionary: dictionary_bytes(table, at),
4853            })
4854            .collect();
4855        Layout {
4856            file: self.size,
4857            rows: table.rows,
4858            stripes: stripes.len(),
4859            parts: self.places.len(),
4860            columns,
4861            indexes: sum(stripes.iter().map(|stripe| u64::from(stripe.index.length))),
4862            directory: self.directory,
4863            header: HEADER,
4864        }
4865    }
4866
4867    /// What every part of one column is stored as, which is what `pragma_storage_info` reports.
4868    ///
4869    /// Unlike [`Self::layout`] this reads the data, because the encoder's choice is in the page and
4870    /// nowhere else. The directory says how many bytes a column took and says nothing about what
4871    /// shape they are in, and the shape is the question worth asking: the same rows in a different
4872    /// order come back bit packed on one file and plain on another, and that is the difference a
4873    /// clustered load makes to a scan.
4874    ///
4875    /// One read per stripe rather than one per part. A part is a few kilobytes out of a page that
4876    /// is a quarter of a megabyte, so asking part by part would read the same page sixty four
4877    /// times. Nothing is put in the page cache, because a caller asking what a file looks like is
4878    /// not about to scan it and evicting the pages a real query wants would be a poor trade.
4879    ///
4880    /// # Errors
4881    ///
4882    /// If the column is outside the schema, or a page, index section or checksum is invalid.
4883    pub fn stored(&self, column: usize) -> Result<Vec<StoredPart>> {
4884        let field = self
4885            .table
4886            .fields
4887            .get(column)
4888            .ok_or_else(|| invalid("stored column index out of range"))?;
4889        let mut stored = Vec::with_capacity(self.places.len());
4890        let mut row = 0;
4891        for (at, stripe) in self.table.stripes.iter().enumerate() {
4892            let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
4893            let index = read_index(&self.file, stripe, column)?;
4894            let mut bytes = vec![0; page.length as usize];
4895            read_at(&self.file, page.offset, &mut bytes)?;
4896            let ranges = self.stripe_part_ranges(at, column);
4897            for (part, &rows) in stripe.parts.iter().enumerate() {
4898                let span = *index.get(part).ok_or_else(|| invalid("part index out of range"))?;
4899                let held = part_bytes(&bytes, span)?;
4900                let range = ranges.and_then(|held| held.get(part));
4901                stored.push(StoredPart {
4902                    stripe: at,
4903                    part,
4904                    row,
4905                    rows: rows as usize,
4906                    encoding: page_encoding(&field.ty, rows as usize, held),
4907                    bytes: span.length as u64,
4908                    page: page.offset,
4909                    offset: span.start as u64,
4910                    low: range
4911                        .and_then(|range| range.low.clone())
4912                        .and_then(|bound| bound.into_value(&field.ty)),
4913                    high: range
4914                        .and_then(|range| range.high.clone())
4915                        .and_then(|bound| bound.into_value(&field.ty)),
4916                    nulls: range.map(|range| range.nulls),
4917                });
4918                row += rows as usize;
4919            }
4920        }
4921        Ok(stored)
4922    }
4923
4924    /// How many parts the table has, which is how many chunks a scan of it reads.
4925    #[must_use]
4926    pub fn parts(&self) -> usize {
4927        self.places.len()
4928    }
4929
4930    /// The parts of each stripe, in table wide part numbers.
4931    ///
4932    /// A scan that wants one worker to own the page it reads hands work out in these runs. The
4933    /// stripes are contiguous in part numbering and all but the last hold sixty four parts, but a
4934    /// stripe can be flushed early when rows arrive out of order, so the runs are read off the
4935    /// directory rather than worked out from a constant.
4936    #[must_use]
4937    pub fn stripe_parts(&self) -> Vec<std::ops::Range<usize>> {
4938        let mut runs = Vec::with_capacity(self.table.stripes.len());
4939        let mut start = 0;
4940        for stripe in &self.table.stripes {
4941            let end = start + stripe.parts.len();
4942            runs.push(start..end);
4943            start = end;
4944        }
4945        runs
4946    }
4947
4948    /// How many rows one stripe holds, in the numbering [`Self::stripe_parts`] hands back.
4949    ///
4950    /// Off the directory, which is already in memory, rather than by the caller asking for each
4951    /// part in turn through the catalog. Nothing past the end holds any rows.
4952    #[must_use]
4953    pub fn stripe_rows(&self, stripe: usize) -> usize {
4954        self.table.stripes.get(stripe).map_or(0, |held| held.rows)
4955    }
4956
4957    /// Asks the page cache to keep `stripes` stripes of every column instead of the default.
4958    ///
4959    /// This only ever raises the number. A scan that gives each worker a whole stripe has one page
4960    /// per column per worker open at once, and a cache smaller than that is worse than no cache at
4961    /// all: every worker's page is evicted by the others before it has finished its stripe, so it
4962    /// reads a quarter of a megabyte for every part it takes out of it.
4963    pub fn keep_stripes(&self, stripes: usize) {
4964        self.cache.kept.fetch_max(stripes, Atomic::Relaxed);
4965    }
4966
4967    /// Rows in one part, or zero when the part number is past the table.
4968    #[must_use]
4969    pub fn part_rows(&self, at: usize) -> usize {
4970        self.places.get(at).map_or(0, |place| place.rows as usize)
4971    }
4972
4973    /// The committed table directory.
4974    #[must_use]
4975    pub fn table(&self) -> &Table {
4976        &self.table
4977    }
4978
4979    /// Exact leading frequencies when the stored synopsis proves a count-descending prefix.
4980    ///
4981    /// The returned list can be longer than `top`. Keeping the stored tail lets a later TopN apply
4982    /// additional ordering keys without losing a value tied with the requested boundary.
4983    ///
4984    /// # Errors
4985    ///
4986    /// If the column is outside the schema or a stored value does not fit its declared type.
4987    pub fn top_frequencies(&self, column: usize, top: usize) -> Result<Option<Vec<(Value, u64)>>> {
4988        let field = self
4989            .table
4990            .fields
4991            .get(column)
4992            .ok_or_else(|| invalid("frequency column index out of range"))?;
4993        let Some(summary) = self.frequency_summary(column)? else {
4994            return Ok(None);
4995        };
4996        if top == 0 || summary.entries.len() < top {
4997            return Ok(None);
4998        }
4999        let boundary = summary.entries[top - 1].count;
5000        if boundary <= summary.omitted_max {
5001            return Ok(None);
5002        }
5003        self.decode_frequencies(column, &field.ty, &summary.entries).map(Some)
5004    }
5005
5006    /// Exact leading counts for a numeric key paired with a stable-dictionary string key.
5007    ///
5008    /// The stored prefix is returned only when its requested boundary strictly beats the bound on
5009    /// every pair omitted at load time. The returned tail may be longer than `top`, as with
5010    /// [`Self::top_frequencies`], so downstream ordering can settle ties without reading rows.
5011    ///
5012    /// # Errors
5013    ///
5014    /// If either column is outside the schema or persisted pair metadata is inconsistent with the
5015    /// frequency synopsis or dictionary it names.
5016    pub fn top_pair_frequencies(
5017        &self,
5018        first: usize,
5019        second: usize,
5020        top: usize,
5021    ) -> Result<Option<PairFrequencyCounts>> {
5022        if first >= self.table.fields.len() || second >= self.table.fields.len() {
5023            return Err(invalid("pair frequency column index out of range"));
5024        }
5025        let Some(summary) =
5026            self.table.pair_frequencies.iter().find(|summary| {
5027                summary.first as usize == first && summary.second as usize == second
5028            })
5029        else {
5030            return Ok(None);
5031        };
5032        if top == 0 || summary.entries.len() < top {
5033            return Ok(None);
5034        }
5035        let boundary = summary.entries[top - 1].count;
5036        if boundary <= summary.omitted_max {
5037            return Ok(None);
5038        }
5039        let first_summary = self
5040            .frequency_summary(first)?
5041            .ok_or_else(|| invalid("pair frequency first column has no synopsis"))?;
5042        let anchors = self
5043            .decode_frequencies(first, &self.table.fields[first].ty, &first_summary.entries)?
5044            .into_iter()
5045            .map(|(value, _)| value)
5046            .collect::<Vec<_>>();
5047        let dictionary = self
5048            .dictionary(second)?
5049            .ok_or_else(|| invalid("pair frequency second column has no dictionary"))?;
5050        let mut codes = summary.entries.iter().filter_map(|entry| entry.second).collect::<Vec<_>>();
5051        codes.sort_unstable();
5052        codes.dedup();
5053        let texts = dictionary
5054            .try_values_visited(&codes.iter().map(|&code| code as usize).collect::<Vec<_>>())?;
5055        let mut out = Vec::with_capacity(summary.entries.len());
5056        for entry in &summary.entries {
5057            if entry.count < boundary {
5058                break;
5059            }
5060            let first = anchors
5061                .get(entry.first_entry as usize)
5062                .cloned()
5063                .ok_or_else(|| invalid("pair frequency anchor is outside its values"))?;
5064            let second = match entry.second {
5065                None => Value::Null,
5066                Some(code) => {
5067                    let at = codes
5068                        .binary_search(&code)
5069                        .map_err(|_| invalid("pair frequency code was not among the codes read"))?;
5070                    texts[at].clone()
5071                }
5072            };
5073            out.push((vec![first, second], entry.count));
5074        }
5075        Ok(Some(out))
5076    }
5077
5078    /// Every value of one column with the number of rows holding it, when the synopsis is complete.
5079    ///
5080    /// The heavy hitter pass keeps a bounded set of candidates and decrements them all when it runs
5081    /// out of room, so what it usually ends with is the leading values and a bound on everything it
5082    /// dropped. `omitted_max` of zero says that never happened: no candidate was ever decremented and
5083    /// the entries did not overflow the stored budget, so the list is every distinct value of the
5084    /// column with an exact count, and a null counts as a value of its own rather than being skipped.
5085    ///
5086    /// That makes a whole class of question answerable without reading a row. How many rows hold a
5087    /// value, how many do not, and what a `GROUP BY` of that column with a count over it produces are
5088    /// all in here. It is only ever true of a column with few enough distinct values, which is the
5089    /// case worth having, because that is exactly the column a grouping or an equality filter would
5090    /// otherwise walk every row to answer.
5091    ///
5092    /// `None` when the column has no synopsis, or has one that dropped anything.
5093    ///
5094    /// # Errors
5095    ///
5096    /// If the column is outside the schema or a stored value does not fit its declared type.
5097    pub fn exact_frequencies(&self, column: usize) -> Result<Option<Vec<(Value, u64)>>> {
5098        let Some(prefix) = self.frequency_prefix(column)? else {
5099            return Ok(None);
5100        };
5101        Ok((prefix.omitted_max == 0).then_some(prefix.entries))
5102    }
5103
5104    /// Every value the synopsis lists with the number of rows holding it, and a bound on the rest.
5105    ///
5106    /// The counts are exact whether or not the list is complete. The heavy hitter pass keeps a
5107    /// bounded candidate set and then recounts only the candidates that survived it, so a value that
5108    /// made it into the list carries the number of rows that really hold it rather than whatever the
5109    /// pass had left over. What the pass loses is values, not counts.
5110    ///
5111    /// `omitted_max` is how many rows the most common value left out can hold, and zero says nothing
5112    /// was left out at all, which is what [`exact_frequencies`] asks for. Above zero the list is the
5113    /// leading values of the column and everything else is somewhere between no rows and that bound.
5114    ///
5115    /// That prefix is worth reading on its own. A column with a value in half its rows and a long
5116    /// tail behind it has no complete synopsis and never will, and it is the column where dividing
5117    /// the rows by the distinct count is furthest from the truth.
5118    ///
5119    /// `None` when the column has no synopsis.
5120    ///
5121    /// # Errors
5122    ///
5123    /// If the column is outside the schema or a stored value does not fit its declared type.
5124    ///
5125    /// [`exact_frequencies`]: Self::exact_frequencies
5126    pub fn frequency_prefix(&self, column: usize) -> Result<Option<FrequencyPrefix>> {
5127        let field = self
5128            .table
5129            .fields
5130            .get(column)
5131            .ok_or_else(|| invalid("frequency column index out of range"))?;
5132        let Some(summary) = self.frequency_summary(column)? else {
5133            return Ok(None);
5134        };
5135        let entries = self.decode_frequencies(column, &field.ty, &summary.entries)?;
5136        Ok(Some(FrequencyPrefix { entries, omitted_max: summary.omitted_max }))
5137    }
5138
5139    /// One column's synopsis, read back from the file when the directory left it there.
5140    fn frequency_summary(&self, column: usize) -> Result<Option<Cow<'_, FrequencySummary>>> {
5141        Ok(match self.table.frequencies.get(column) {
5142            None | Some(None) => None,
5143            Some(Some(Frequencies::Held(summary))) => Some(Cow::Borrowed(summary)),
5144            Some(Some(Frequencies::Stored { span, values })) => {
5145                let slot = self
5146                    .frequency_summaries
5147                    .get(column)
5148                    .ok_or_else(|| invalid("frequency column index out of range"))?;
5149                if let Some(summary) = slot.get() {
5150                    return Ok(Some(Cow::Borrowed(summary.as_ref())));
5151                }
5152                let field = self
5153                    .table
5154                    .fields
5155                    .get(column)
5156                    .ok_or_else(|| invalid("frequency column index out of range"))?;
5157                let mut bytes = vec![0; span.length as usize];
5158                read_at(&self.file, span.offset, &mut bytes)?;
5159                let summary =
5160                    decode_summary(&mut Cursor::new(&bytes), field, self.table.rows, *values)?;
5161                let summary = summary.ok_or_else(|| invalid("a stored synopsis is missing"))?;
5162                let _ = slot.set(Arc::new(summary));
5163                Some(Cow::Borrowed(slot.get().expect("the decoded summary was stored").as_ref()))
5164            }
5165        })
5166    }
5167
5168    /// Turns stored frequency entries into values of the column's own type.
5169    ///
5170    /// Remembered per column, because the planner asks once for every estimate that touches the
5171    /// column and the executor asks again, and the answer is a few hundred values. The codes of a
5172    /// string column are read through [`Vector::try_values_visited`], which does not keep the blocks
5173    /// it decodes, so what a query answered out of the synopsis holds is those values and not the
5174    /// hundred or so dictionary blocks they are scattered over.
5175    fn decode_frequencies(
5176        &self,
5177        column: usize,
5178        ty: &LogicalType,
5179        entries: &[FrequencyEntry],
5180    ) -> Result<Vec<(Value, u64)>> {
5181        if let Some(values) = self.frequency_values.get(column).and_then(OnceLock::get) {
5182            return Ok(values.as_ref().clone());
5183        }
5184        let values = self.decode_frequencies_once(column, ty, entries)?;
5185        if let Some(slot) = self.frequency_values.get(column) {
5186            let _ = slot.set(Arc::new(values.clone()));
5187        }
5188        Ok(values)
5189    }
5190
5191    fn decode_frequencies_once(
5192        &self,
5193        column: usize,
5194        ty: &LogicalType,
5195        entries: &[FrequencyEntry],
5196    ) -> Result<Vec<(Value, u64)>> {
5197        let stored_texts = self.table.frequency_texts.get(column).filter(|texts| !texts.is_empty());
5198        if stored_texts.is_some_and(|texts| texts.len() != entries.len()) {
5199            return Err(invalid("frequency text count differs from its synopsis"));
5200        }
5201        let dictionary = if *ty == LogicalType::Varchar && stored_texts.is_none() {
5202            self.dictionary(column)?
5203        } else {
5204            None
5205        };
5206        let mut codes = entries
5207            .iter()
5208            .filter_map(|entry| match entry.value {
5209                FrequencyValue::Code(code) => Some(code as usize),
5210                _ => None,
5211            })
5212            .collect::<Vec<_>>();
5213        codes.sort_unstable();
5214        codes.dedup();
5215        let texts = match &dictionary {
5216            Some(dictionary) if !codes.is_empty() => dictionary.try_values_visited(&codes)?,
5217            _ => Vec::new(),
5218        };
5219        let mut out = Vec::with_capacity(entries.len());
5220        for (entry_at, entry) in entries.iter().enumerate() {
5221            let value = match entry.value {
5222                FrequencyValue::Null => {
5223                    if stored_texts.and_then(|texts| texts[entry_at].as_ref()).is_some() {
5224                        return Err(invalid("a null frequency entry has text"));
5225                    }
5226                    Value::Null
5227                }
5228                FrequencyValue::Integer(value) => match *ty {
5229                    LogicalType::TinyInt => Value::TinyInt(
5230                        i8::try_from(value)
5231                            .map_err(|_| invalid("frequency TINYINT is out of range"))?,
5232                    ),
5233                    LogicalType::UTinyInt => Value::UTinyInt(
5234                        u8::try_from(value)
5235                            .map_err(|_| invalid("frequency UTINYINT is out of range"))?,
5236                    ),
5237                    LogicalType::USmallInt => Value::USmallInt(
5238                        u16::try_from(value)
5239                            .map_err(|_| invalid("frequency USMALLINT is out of range"))?,
5240                    ),
5241                    LogicalType::UInteger => Value::UInteger(
5242                        u32::try_from(value)
5243                            .map_err(|_| invalid("frequency UINTEGER is out of range"))?,
5244                    ),
5245                    LogicalType::UBigInt => Value::UBigInt(
5246                        u64::try_from(value)
5247                            .map_err(|_| invalid("frequency UBIGINT is out of range"))?,
5248                    ),
5249                    LogicalType::SmallInt => Value::SmallInt(
5250                        i16::try_from(value)
5251                            .map_err(|_| invalid("frequency SMALLINT is out of range"))?,
5252                    ),
5253                    LogicalType::Integer => Value::Integer(
5254                        i32::try_from(value)
5255                            .map_err(|_| invalid("frequency INTEGER is out of range"))?,
5256                    ),
5257                    LogicalType::BigInt => Value::BigInt(
5258                        i64::try_from(value)
5259                            .map_err(|_| invalid("frequency BIGINT is out of range"))?,
5260                    ),
5261                    LogicalType::Date => Value::Date(
5262                        i32::try_from(value)
5263                            .map_err(|_| invalid("frequency DATE is out of range"))?,
5264                    ),
5265                    LogicalType::Timestamp => Value::Timestamp(
5266                        i64::try_from(value)
5267                            .map_err(|_| invalid("frequency TIMESTAMP is out of range"))?,
5268                    ),
5269                    _ => return Err(invalid("integer frequency belongs to another type")),
5270                },
5271                FrequencyValue::Code(code) => {
5272                    if let Some(text) = stored_texts.and_then(|texts| texts[entry_at].as_ref()) {
5273                        Value::Varchar(
5274                            String::from_utf8(text.clone())
5275                                .map_err(|_| invalid("frequency text is not UTF-8"))?,
5276                        )
5277                    } else {
5278                        if dictionary.is_none() {
5279                            return Err(invalid("frequency code has no dictionary or stored text"));
5280                        }
5281                        let at = codes
5282                            .binary_search(&(code as usize))
5283                            .map_err(|_| invalid("frequency code was not among the codes read"))?;
5284                        texts[at].clone()
5285                    }
5286                }
5287            };
5288            out.push((value, entry.count));
5289        }
5290        Ok(out)
5291    }
5292
5293    /// Sparse rows belonging to the bounded numeric frequency candidate set.
5294    ///
5295    /// The list is omitted when collecting it would exceed the fixed storage budget. A composite
5296    /// aggregate may accept a result over these rows only when its requested boundary is strictly
5297    /// greater than `omitted_max`.
5298    ///
5299    /// # Errors
5300    ///
5301    /// If the column is outside the schema.
5302    pub fn frequency_occurrences(&self, column: usize) -> Result<Option<FrequencyOccurrences>> {
5303        let field = self
5304            .table
5305            .fields
5306            .get(column)
5307            .ok_or_else(|| invalid("frequency column index out of range"))?;
5308        let Some(summary) = self.frequency_summary(column)? else {
5309            return Ok(None);
5310        };
5311        if summary.ordinals.is_empty() {
5312            return Ok(None);
5313        }
5314        let (anchors, anchor_indices) = if summary.ordinal_entries.len() == summary.ordinals.len() {
5315            let entries = self.decode_frequencies(column, &field.ty, &summary.entries)?;
5316            (entries.into_iter().map(|(value, _)| value).collect(), summary.ordinal_entries.clone())
5317        } else {
5318            (Vec::new(), Vec::new())
5319        };
5320        Ok(Some(FrequencyOccurrences {
5321            omitted_max: summary.omitted_max,
5322            ordinals: summary.ordinals.clone(),
5323            anchors,
5324            anchor_indices,
5325        }))
5326    }
5327
5328    /// How many distinct values one column holds, counting a null as no value.
5329    ///
5330    /// A string column of this format is written against one dictionary that covers the whole table.
5331    /// A code is handed out the first time a value is seen and nothing ever removes one, so the
5332    /// number of codes is the number of distinct values exactly rather than an estimate. That makes
5333    /// `COUNT(DISTINCT column)` over a whole table a question the directory already knows the answer
5334    /// to, and the alternative is a hash table with a row per distinct value built from a pass over
5335    /// every row.
5336    ///
5337    /// A null in the column used to make this `None` and no longer does. A null row is written as
5338    /// the code for the empty string, so a nullable column's dictionary can hold an empty string
5339    /// that no row of it actually has, and the dictionary on its own does not say which case it is.
5340    /// The writer does know, because it counts the non-null rows that use each code on its way to
5341    /// the frequency summary, so it records how many codes any row holds and the directory carries
5342    /// that number. This reads it rather than the size of the dictionary, which also means the
5343    /// dictionary page is not opened to answer.
5344    ///
5345    /// An integer column has no dictionary, and its count comes from the set the writer keeps on its
5346    /// numeric frequency pass instead, which is exact up to a cap. `None` for a column past that cap
5347    /// and for every column that is neither, where a sketch would answer approximately and SQL asked
5348    /// for the exact number.
5349    ///
5350    /// # Errors
5351    ///
5352    /// If the column is outside the schema.
5353    pub fn distinct_values(&self, column: usize) -> Result<Option<u64>> {
5354        self.table
5355            .distincts
5356            .get(column)
5357            .copied()
5358            .ok_or_else(|| invalid("distinct column index out of range"))
5359    }
5360
5361    /// How many rows of one column are null, added up over the stripes.
5362    ///
5363    /// Every stripe records this exactly when it is written, because a null count is not a bound
5364    /// that is allowed to be wide the way a minimum and a maximum are: a filter that reads one too
5365    /// many is slow and a `COUNT` that reads one too many is wrong. Adding up a few hundred numbers
5366    /// already in memory is what makes `COUNT(column)` over a whole table free.
5367    ///
5368    /// # Errors
5369    ///
5370    /// If the column is outside the schema.
5371    pub fn null_count(&self, column: usize) -> Result<u64> {
5372        if column >= self.table.fields.len() {
5373            return Err(invalid("null count column index out of range"));
5374        }
5375        let mut nulls = 0_u64;
5376        for stripe in &self.table.stripes {
5377            let range = stripe
5378                .zone
5379                .column(column)
5380                .ok_or_else(|| invalid("stripe zone is narrower than the schema"))?;
5381            nulls = nulls
5382                .checked_add(range.nulls as u64)
5383                .ok_or_else(|| invalid("null count overflow"))?;
5384        }
5385        Ok(nulls)
5386    }
5387
5388    /// The smallest and the largest value of one string column, from the order beside its values.
5389    ///
5390    /// The dictionary holds exactly the values the column holds, so the first and the last of them
5391    /// in sorted order are the column's minimum and maximum. Two reads of a rank block settle what
5392    /// otherwise walks a million rows.
5393    ///
5394    /// `None` when the column is not a string, when the file was written before version 9 and so has
5395    /// no order, when the column has no values at all, or when it has a null in it, which is the
5396    /// placeholder again: the empty string a null is written as would sort ahead of every real
5397    /// value and be reported as the minimum.
5398    ///
5399    /// # Errors
5400    ///
5401    /// If the column is outside the schema, or a rank names a code the dictionary does not have.
5402    pub fn text_extremes(&self, column: usize) -> Result<Option<(Value, Value)>> {
5403        if self.null_count(column)? > 0 {
5404            return Ok(None);
5405        }
5406        let Some(dictionary) = self.dictionary(column)? else { return Ok(None) };
5407        let Some(ranks) = dictionary.ranks() else { return Ok(None) };
5408        if ranks == 0 {
5409            return Ok(None);
5410        }
5411        let low = text_at_rank(&dictionary, 0)?;
5412        let high = text_at_rank(&dictionary, ranks - 1)?;
5413        Ok(Some((low, high)))
5414    }
5415
5416    /// The smallest and the largest value of one column, when every stripe wrote exact ends.
5417    ///
5418    /// A stripe's ends are allowed to be wider than the truth, because a bound that rules out a
5419    /// chunk that could not match is still correct when it rules out nothing. That is what makes
5420    /// them cheap to write for a bit packed or a dictionary column, and it is also what stops them
5421    /// answering a `MIN`. So each stripe says which of the two it wrote, and this answers only when
5422    /// all of them walked their rows.
5423    ///
5424    /// `None` for a column with no ends, for an empty table, and for a column any stripe of which
5425    /// guessed. Nulls need no special case, because the ends skip them the same way `MIN` does.
5426    ///
5427    /// One case is given up on that did not have to be. A stripe merges the ends of its sixty four
5428    /// parts, and a part with no ends at all erases the merged ones, because a part whose rows are
5429    /// not covered by the stripe's ends is a stripe that would skip rows it should keep. A part of
5430    /// nothing but nulls has no rows to cover and so did not need to erase anything, but the merge
5431    /// cannot tell that part from a part whose layout it could not read. So a column with a chunk
5432    /// of nothing but nulls in the middle of it goes and reads the rows. That is slow and right,
5433    /// and the fix is a row count per part rather than anything here.
5434    ///
5435    /// # Errors
5436    ///
5437    /// If the column is outside the schema.
5438    pub fn exact_extremes(&self, column: usize) -> Result<Option<(Bound, Bound)>> {
5439        if column >= self.table.fields.len() {
5440            return Err(invalid("extremes column index out of range"));
5441        }
5442        let mut low: Option<Bound> = None;
5443        let mut high: Option<Bound> = None;
5444        for stripe in &self.table.stripes {
5445            let range = stripe
5446                .zone
5447                .column(column)
5448                .ok_or_else(|| invalid("stripe zone is narrower than the schema"))?;
5449            if !range.exact {
5450                return Ok(None);
5451            }
5452            // A stripe of nothing but nulls has no ends and says nothing about the column's, which
5453            // is why this skips it rather than giving up on the whole column. A stripe that has
5454            // rows and still has no end is a layout whose values this cannot see, and skipping that
5455            // one would answer with an end taken from the other stripes, so it gives up instead.
5456            let (Some(small), Some(large)) = (range.low.as_ref(), range.high.as_ref()) else {
5457                if stripe.rows > range.nulls {
5458                    return Ok(None);
5459                }
5460                continue;
5461            };
5462            low = Some(low.map_or_else(|| small.clone(), |held| held.smaller(small.clone())));
5463            high = Some(high.map_or_else(|| large.clone(), |held| held.larger(large.clone())));
5464        }
5465        Ok(low.zip(high))
5466    }
5467
5468    /// The sum of one integer column and how many rows went into it, when every stripe wrote one.
5469    ///
5470    /// The count beside the sum is the non-null rows, because that is what a `SUM` adds up and what
5471    /// an `AVG` divides by, and a caller that had to work it out from the row count and the null
5472    /// count would be doing the same walk twice.
5473    ///
5474    /// `None` for anything that is not an integer column, for a file written by something that did
5475    /// not record it, and when adding the stripes together would overflow.
5476    ///
5477    /// # Errors
5478    ///
5479    /// If the column is outside the schema.
5480    pub fn exact_sum(&self, column: usize) -> Result<Option<(i128, u64)>> {
5481        if column >= self.table.fields.len() {
5482            return Err(invalid("sum column index out of range"));
5483        }
5484        let mut total = 0_i128;
5485        let mut rows = 0_u64;
5486        for stripe in &self.table.stripes {
5487            let range = stripe
5488                .zone
5489                .column(column)
5490                .ok_or_else(|| invalid("stripe zone is narrower than the schema"))?;
5491            let Some(part) = range.sum else { return Ok(None) };
5492            let Some(sum) = total.checked_add(part) else { return Ok(None) };
5493            total = sum;
5494            rows = rows.saturating_add(stripe.rows as u64 - range.nulls as u64);
5495        }
5496        Ok(Some((total, rows)))
5497    }
5498
5499    /// Certified host groups over a string column, when the caller's inclusive row-count bound
5500    /// excludes every host the synopsis omitted.
5501    pub fn host_groups(
5502        &self,
5503        column: usize,
5504        minimum_count: u64,
5505    ) -> Result<Option<Vec<host::HostEntry>>> {
5506        if column >= self.table.fields.len() {
5507            return Err(invalid("host group column index out of range"));
5508        }
5509        let Some(summary) = &self.table.host_groups else { return Ok(None) };
5510        if summary.column != column || minimum_count <= summary.omitted_max {
5511            return Ok(None);
5512        }
5513        Ok(Some(summary.entries.clone()))
5514    }
5515
5516    /// The global dictionary of a column, opened once however many workers ask for it at once.
5517    ///
5518    /// The unlocked look is first because it is the answer every time after the first and it costs a
5519    /// load. Everybody who misses it queues on [`Self::loading`] and looks again on the way in, so
5520    /// the one who arrived first does the reading and the rest take what it left. Waiting is the
5521    /// cheaper thing to do: the work behind the lock is a page read, a checksum and the decode of a
5522    /// dictionary that can hold half a million entries, and the alternative is every worker of the
5523    /// scan doing all of it and all but one dropping the result on the floor.
5524    fn dictionary(&self, column: usize) -> Result<Option<Arc<Vector>>> {
5525        let Some(page) = self.table.dictionaries[column] else { return Ok(None) };
5526        if let Some(dictionary) = self.dictionaries[column].get() {
5527            return Ok(Some(Arc::clone(dictionary)));
5528        }
5529        let _queued = self.loading[column].lock().map_err(|_| invalid("a poisoned dictionary"))?;
5530        if let Some(dictionary) = self.dictionaries[column].get() {
5531            return Ok(Some(Arc::clone(dictionary)));
5532        }
5533        self.opened.fetch_add(1, Atomic::Relaxed);
5534        let dictionary = Arc::new(open_global_dictionary(
5535            Arc::clone(&self.file),
5536            page,
5537            &self.table.fields[column].ty,
5538            TEXT_KEEP_BUDGET,
5539        )?);
5540        let _ = self.dictionaries[column].set(Arc::clone(&dictionary));
5541        Ok(Some(dictionary))
5542    }
5543
5544    /// Reads one section's extent table and checks it against the entry that names it.
5545    ///
5546    /// # Errors
5547    ///
5548    /// If the entry points outside the file, the table does not checksum, or it does not decode as
5549    /// a run of extents in element order.
5550    pub fn extents(&self, of: &Section) -> Result<Vec<section::Extent>> {
5551        if of.extent_bytes == 0 {
5552            return Ok(Vec::new());
5553        }
5554        let mut bytes = vec![0; of.extent_bytes as usize];
5555        read_at(&self.file, of.extent_page, &mut bytes)?;
5556        if checksum(&bytes) != of.hash {
5557            return Err(invalid("a section's extent table does not checksum"));
5558        }
5559        let extents = section::decode_extents(&bytes)?;
5560        if extents.len() != of.extents as usize {
5561            return Err(invalid("a section's extent table is not the length the entry says"));
5562        }
5563        Ok(extents)
5564    }
5565
5566    /// Reads and verifies one extent of a section.
5567    ///
5568    /// This is what section 3.2's second rule is for. A reduction that needs one extent of a two
5569    /// gigabyte forward link reads and checksums that extent and nothing else, which is the whole
5570    /// difference between a structure that works at SF100 and issue #745.
5571    ///
5572    /// # Errors
5573    ///
5574    /// If the extent points outside the file, or its bytes do not checksum.
5575    pub fn extent(&self, of: &section::Extent) -> Result<Vec<u8>> {
5576        let end = of
5577            .offset
5578            .checked_add(u64::from(of.length))
5579            .ok_or_else(|| invalid("an extent overflows the file"))?;
5580        if of.offset < HEADER || end > self.size {
5581            return Err(invalid("an extent is outside the file"));
5582        }
5583        let mut bytes = vec![0; of.length as usize];
5584        read_at(&self.file, of.offset, &mut bytes)?;
5585        if checksum(&bytes) != of.hash {
5586            return Err(invalid("an extent does not checksum"));
5587        }
5588        Ok(bytes)
5589    }
5590
5591    /// Reads a whole section's payload, every extent of it, in order.
5592    ///
5593    /// For a structure that is resident anyway, which a key map is. Anything large enough that the
5594    /// split matters should be walking [`Reader::extents`] and taking the one it needs.
5595    ///
5596    /// # Errors
5597    ///
5598    /// If the extent table or any extent fails its check.
5599    pub fn payload(&self, of: &Section) -> Result<Vec<u8>> {
5600        let extents = self.extents(of)?;
5601        let mut bytes =
5602            Vec::with_capacity(sum(extents.iter().map(|one| u64::from(one.length))) as usize);
5603        for one in &extents {
5604            if one.first != bytes.len() as u64 {
5605                return Err(invalid("a section's extents do not join up"));
5606            }
5607            bytes.extend_from_slice(&self.extent(one)?);
5608        }
5609        // The same exception `write_section` makes: a budget record has no bytes, so its
5610        // `header_bytes` is a size rather than a header and there is nothing for it to run past.
5611        if !bytes.is_empty() && of.header_bytes as usize > bytes.len() {
5612            return Err(invalid("a section's header is longer than its payload"));
5613        }
5614        Ok(bytes)
5615    }
5616
5617    /// Reads only the named columns from one part.
5618    ///
5619    /// The whole stripe page each column lives in is read and kept, because a scan asks for the
5620    /// parts of a stripe one after another and this is what turns sixty four reads into one.
5621    ///
5622    /// # Errors
5623    ///
5624    /// If a part, column, page, or checksum is invalid.
5625    pub fn read(&self, part: usize, columns: &[usize]) -> Result<Chunk> {
5626        self.read_impl(part, columns, true)
5627    }
5628
5629    /// Reads named columns from one part without keeping the stripe page it came out of.
5630    ///
5631    /// This is for sparse row fetches after a selective TopN or filter, which reach a few parts of
5632    /// a stripe rather than all of them. A caller that will read most of a stripe should use
5633    /// [`Self::read`] instead, because this reads and discards the page index every time.
5634    ///
5635    /// # Errors
5636    ///
5637    /// If a part, column, page, or checksum is invalid.
5638    pub fn read_sparse(&self, part: usize, columns: &[usize]) -> Result<Chunk> {
5639        self.read_impl(part, columns, false)
5640    }
5641
5642    /// Whether an exact global-code membership index proves that the stripe holding a part cannot
5643    /// contain any of the sorted candidate codes.
5644    ///
5645    /// # Errors
5646    ///
5647    /// If the part, column, index page, checksum, or delta stream is invalid.
5648    pub fn skips_codes(&self, part: usize, column: usize, candidates: &[u32]) -> Result<bool> {
5649        if candidates.is_empty() {
5650            return Ok(true);
5651        }
5652        if candidates.windows(2).any(|pair| pair[0] >= pair[1]) {
5653            return Err(Error::internal("native code candidates are not sorted and unique"));
5654        }
5655        let stripe = self.stripe_of(part)?;
5656        let Some(page) = stripe.memberships.get(column) else {
5657            return Ok(false);
5658        };
5659        let mut bytes = vec![0; page.length as usize];
5660        read_at(&self.file, page.offset, &mut bytes)?;
5661        if checksum(&bytes) != page.hash {
5662            return Err(invalid("membership page checksum differs"));
5663        }
5664        let codes = decode_membership(&bytes)?;
5665        let mut left = 0;
5666        let mut right = 0;
5667        while left < codes.len() && right < candidates.len() {
5668            match codes[left].cmp(&candidates[right]) {
5669                Ordering::Less => left += 1,
5670                Ordering::Greater => right += 1,
5671                Ordering::Equal => return Ok(false),
5672            }
5673        }
5674        Ok(true)
5675    }
5676
5677    fn stripe_of(&self, part: usize) -> Result<&Stripe> {
5678        let place = self.places.get(part).ok_or_else(|| invalid("part index out of range"))?;
5679        self.table
5680            .stripes
5681            .get(place.stripe as usize)
5682            .ok_or_else(|| invalid("stripe index out of range"))
5683    }
5684
5685    /// The page index of one column of one stripe, and its page when the caller wants all of it.
5686    ///
5687    /// A scan hands parts out in order, so every worker on a column crosses into a new stripe within
5688    /// a few parts of the others and they all want the same page at the same moment. This used to
5689    /// let all of them read it, which cost the scan as many copies of every page as it had workers.
5690    /// On the full ClickBench file a `MIN(EventDate), MAX(EventDate)` moved 3.2 GB off the disk to
5691    /// look at 400 MB of column.
5692    ///
5693    /// A worker that finds the page it wants already being read neither waits for it nor reads it
5694    /// again. It comes back with the index alone, which sends [`Reader::read_impl`] down the path
5695    /// that reads the one part it came for, a few kilobytes against a quarter of a megabyte, and it
5696    /// picks the page up from the cache on its next part. Waiting would be the other way to avoid
5697    /// the duplicate read and it is worse: the pages that matter are the wide string ones, they take
5698    /// milliseconds to copy even warm, and every other worker would be stopped for all of it.
5699    ///
5700    /// The file is never read under the lock.
5701    fn held(&self, at: usize, stripe: &Stripe, column: usize, whole: bool) -> Result<CachedColumn> {
5702        let cache =
5703            self.cache.columns.get(column).ok_or_else(|| invalid("column index out of range"))?;
5704        let mut cached = cache.lock().map_err(|_| invalid("column page cache is poisoned"))?;
5705        let known = cached.index.get(at).and_then(Clone::clone);
5706        let page = cached.pages.get(at).and_then(Option::as_ref).map(|slot| {
5707            slot.used.store(true, Atomic::Relaxed);
5708            Arc::clone(&slot.page)
5709        });
5710        if let Some(index) = known.clone() {
5711            if !whole || page.is_some() {
5712                return Ok(CachedColumn { stripe: at, index, page });
5713            }
5714        }
5715        if cached.loading.contains(&at) {
5716            drop(cached);
5717            // The index is almost always already here, because somebody read this stripe to get
5718            // into the loading list in the first place, so this branch usually costs no read at
5719            // all and the one part read in `read_impl` is all the losing worker pays for.
5720            if let Some(index) = known {
5721                return Ok(CachedColumn { stripe: at, index, page: None });
5722            }
5723            let held = self.page_of(stripe, column, at, false, None)?;
5724            let mut cached = cache.lock().map_err(|_| invalid("column page cache is poisoned"))?;
5725            remember(&mut cached, &held);
5726            return Ok(held);
5727        }
5728        cached.loading.push(at);
5729        drop(cached);
5730
5731        let read = self.page_of(stripe, column, at, whole, known);
5732
5733        // The stripe leaves the loading list and its page enters the cache under one lock. Doing
5734        // them separately would leave a moment where another worker sees neither and reads the
5735        // page a second time, which is the whole thing this is here to stop.
5736        let mut cached = cache.lock().map_err(|_| invalid("column page cache is poisoned"))?;
5737        if let Some(position) = cached.loading.iter().position(|loading| *loading == at) {
5738            cached.loading.remove(position);
5739        }
5740        let held = read?;
5741        let taken = remember(&mut cached, &held);
5742        drop(cached);
5743        if let Some((bytes, used)) = taken {
5744            self.cache.held[column].fetch_add(1, Atomic::Relaxed);
5745            self.pool.admit(Held {
5746                shelf: Arc::downgrade(&self.cache),
5747                column,
5748                stripe: at,
5749                bytes,
5750                used,
5751            });
5752        }
5753        Ok(held)
5754    }
5755
5756    /// Reads one stripe's index for a column, and its page when the caller wants all of it.
5757    ///
5758    /// `known` is the index when the reader has already read it, which after the first worker
5759    /// through a stripe it always has, because [`remember`] keeps every index for the life of the
5760    /// reader. Without that a scan reads the index again on every part that misses the page cache.
5761    fn page_of(
5762        &self,
5763        stripe: &Stripe,
5764        column: usize,
5765        at: usize,
5766        whole: bool,
5767        known: Option<Arc<Vec<PartSpan>>>,
5768    ) -> Result<CachedColumn> {
5769        let index = match known {
5770            Some(index) => index,
5771            None => {
5772                self.indexes.fetch_add(1, Atomic::Relaxed);
5773                Arc::new(read_index(&self.file, stripe, column)?)
5774            }
5775        };
5776        let page = if whole {
5777            self.pages.fetch_add(1, Atomic::Relaxed);
5778            let span = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
5779            let mut bytes = vec![0; span.length as usize];
5780            read_at(&self.file, span.offset, &mut bytes)?;
5781            Some(Arc::new(bytes))
5782        } else {
5783            None
5784        };
5785        Ok(CachedColumn { stripe: at, index, page })
5786    }
5787
5788    fn read_impl(&self, at: usize, columns: &[usize], whole: bool) -> Result<Chunk> {
5789        let place = *self.places.get(at).ok_or_else(|| invalid("part index out of range"))?;
5790        let index = place.stripe as usize;
5791        let stripe =
5792            self.table.stripes.get(index).ok_or_else(|| invalid("stripe index out of range"))?;
5793        let rows = place.rows as usize;
5794        let mut picked = Vec::with_capacity(columns.len());
5795        for &column in columns {
5796            let field = self
5797                .table
5798                .fields
5799                .get(column)
5800                .ok_or_else(|| invalid("column index out of range"))?;
5801            let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
5802            let held = self.held(index, stripe, column, whole)?;
5803            let span = *held
5804                .index
5805                .get(place.part as usize)
5806                .ok_or_else(|| invalid("part index out of range"))?;
5807            let owned;
5808            let bytes = match &held.page {
5809                Some(held) => part_bytes(held, span)?,
5810                None => {
5811                    let offset = page
5812                        .offset
5813                        .checked_add(span.start as u64)
5814                        .ok_or_else(|| invalid("part range overflow"))?;
5815                    let mut bytes = vec![0; span.length];
5816                    read_at(&self.file, offset, &mut bytes)?;
5817                    owned = bytes;
5818                    &owned
5819                }
5820            };
5821            if checksum(bytes) != span.hash {
5822                return Err(invalid(&format!(
5823                    "column page checksum differs, column {column} part {} at {}+{} of {} bytes, \
5824                     wanted {:016x} and got {:016x}",
5825                    place.part,
5826                    page.offset,
5827                    span.start,
5828                    span.length,
5829                    span.hash,
5830                    checksum(bytes),
5831                )));
5832            }
5833            let dictionary = self.dictionary(column)?;
5834            // Held as a page, because a column that came out of a file is handed out more than
5835            // once. A group by clones its key columns out of the chunk so the keys outlive it, a
5836            // projection of a bare column name does the same, and a cut of a flat run copies unless
5837            // the run is a page. One `Arc` per column per part buys all of those, and it moves the
5838            // run into the `Arc` without touching a value.
5839            picked.push(decode(&field.ty, rows, bytes, dictionary)?.into_pages());
5840        }
5841        Chunk::with_rows(picked, rows)
5842    }
5843
5844    /// Whether persisted statistics prove that a part cannot match the predicates.
5845    ///
5846    /// Three of them, asked cheapest first.
5847    ///
5848    /// The stripe's bounds are in memory already, so they are free, and they are also the coarsest:
5849    /// every part of a stripe gets the same answer and a scan that skips one part that way skips all
5850    /// sixty four. Then the part's own bounds, which are a read of one page per column per stripe
5851    /// and are sixty four times finer. Then the sieves, which are per part and answer equality, the
5852    /// test bounds are worst at: a column of identifiers has every stripe and nearly every part
5853    /// covering the whole of its type, so bounds keep them all and the sieve keeps the ones that
5854    /// really hold the value.
5855    ///
5856    /// The middle one is what an ordered comparison on a column the rows are not sorted by needs. On
5857    /// ClickBench 24 the stripe bounds leave eight stripes of sixteen alive, which is half the file,
5858    /// and the part bounds leave thirty parts of nine hundred and seventy four.
5859    #[must_use]
5860    pub fn skips(&self, part: usize, probes: &[Probe]) -> bool {
5861        let Some(place) = self.places.get(part).copied() else { return false };
5862        let Some(stripe) = self.table.stripes.get(place.stripe as usize) else { return false };
5863        if stripe.zone.skips(probes) {
5864            return true;
5865        }
5866        probes.iter().any(|probe| self.outside(place, probe) || self.sifted(place, probe))
5867    }
5868
5869    /// Whether the bounds of one part rule out one probe.
5870    ///
5871    /// The part's own two ends, which are narrower than the stripe's and cost a page read the first
5872    /// time this is asked about a column. A column with no page here answers `false`, which is the
5873    /// answer a caller got before there were any.
5874    fn outside(&self, place: Place, probe: &Probe) -> bool {
5875        match self.stripe_part_ranges(place.stripe as usize, probe.column) {
5876            Some(ranges) => ranges
5877                .get(place.part as usize)
5878                .is_some_and(|range| range.excludes(probe.op, &probe.value)),
5879            None => false,
5880        }
5881    }
5882
5883    /// The per part ranges of one stripe of one column, read once and kept.
5884    ///
5885    /// `None` when the column has no page in that stripe and when the page is damaged, on the same
5886    /// reasoning as the sieves: this is an index over data that is still there, so a caller that
5887    /// cannot read one reads the rows and gets the right answer slowly.
5888    fn stripe_part_ranges(&self, stripe: usize, column: usize) -> Option<&[Range]> {
5889        let slot = self.part_ranges.get(column)?.get(stripe)?;
5890        if let Some(held) = slot.get() {
5891            return Some(held);
5892        }
5893        let page = self.table.stripes.get(stripe)?.part_ranges.get(column)?;
5894        let mut bytes = vec![0; page.length as usize];
5895        read_at(&self.file, page.offset, &mut bytes).ok()?;
5896        if checksum(&bytes) != page.hash {
5897            return None;
5898        }
5899        let ranges = Arc::new(decode_part_ranges(&bytes).ok()?);
5900        let _ = slot.set(ranges);
5901        slot.get().map(|held| held.as_slice())
5902    }
5903
5904    /// Whether persisted statistics prove that every row of a part matches the predicates.
5905    ///
5906    /// Only the bounds. The sieves say nothing here, because a sieve that holds a value is a sieve
5907    /// that may be holding somebody else's hash, so it can rule a part out and can never wave one
5908    /// through.
5909    ///
5910    /// The stripe first and the part after it, the same two steps and in the same order as
5911    /// [`Self::skips`]. The stripe's bounds are in memory already and its null count covers sixty
5912    /// four parts rather than one, so a stripe that answers is an answer for nothing, and the part's
5913    /// own bounds are only read for the probes it could not settle. Both directions are safe: a
5914    /// stretch where everything passes contains no narrower stretch where something fails, and a
5915    /// stripe with no nulls has no nulls in any of its parts.
5916    ///
5917    /// A string end a part recorded is cut down to its first few bytes, so a part's stretch can be
5918    /// wider than its rows really are as well. That is the same safe direction for the same reason,
5919    /// and it is why this asks the two ends rather than anything `exact` says.
5920    #[must_use]
5921    pub fn certain(&self, part: usize, probes: &[Probe]) -> bool {
5922        let Some(place) = self.places.get(part).copied() else { return false };
5923        let Some(stripe) = self.table.stripes.get(place.stripe as usize) else { return false };
5924        if stripe.zone.certain(probes) {
5925            return true;
5926        }
5927        probes
5928            .iter()
5929            .all(|probe| stripe.zone.certain(slice::from_ref(probe)) || self.inside(place, probe))
5930    }
5931
5932    /// Whether one part's own two ends prove that every row of it passes `probe`.
5933    ///
5934    /// The mirror of [`Self::outside`], reading the same page. `false` for a part whose stripe wrote
5935    /// no range page, which is a stripe of one part, because there the stripe's own bounds are the
5936    /// part's and the caller has already asked them.
5937    fn inside(&self, place: Place, probe: &Probe) -> bool {
5938        match self.stripe_part_ranges(place.stripe as usize, probe.column) {
5939            Some(ranges) => ranges
5940                .get(place.part as usize)
5941                .is_some_and(|range| range.certain(probe.op, &probe.value)),
5942            None => false,
5943        }
5944    }
5945
5946    /// Whether the bounds of one stripe prove that none of its parts can match the predicates.
5947    ///
5948    /// The cheap half of [`Self::skips`], asked about a whole stripe at once. The bounds live in the
5949    /// directory and are already in memory, so this answers without touching the file, and that is
5950    /// the reason it is worth having on its own: a caller that wants to know roughly where the work
5951    /// is before it starts any workers can ask this about sixteen stripes for nothing, where asking
5952    /// [`Self::skips`] about nine hundred parts would read and decode a sieve page per stripe first.
5953    ///
5954    /// It keeps stripes that [`Self::skips`] would rule out part by part, which is the right way for
5955    /// it to be wrong: the parts are still checked when they are read.
5956    #[must_use]
5957    pub fn stripe_skips(&self, stripe: usize, probes: &[Probe]) -> bool {
5958        self.table.stripes.get(stripe).is_some_and(|held| held.zone.skips(probes))
5959    }
5960
5961    /// Whether the sieve of one part rules out one probe.
5962    ///
5963    /// Only equality. An ordered comparison is what the bounds are for and a sieve says nothing
5964    /// about it, and a read that cannot answer keeps the part, which is the answer a caller with no
5965    /// sieve gets anyway.
5966    fn sifted(&self, place: Place, probe: &Probe) -> bool {
5967        if probe.op != Op::Equal {
5968            return false;
5969        }
5970        match self.stripe_sieves(place.stripe as usize, probe.column) {
5971            Some(sieves) => sieves
5972                .get(place.part as usize)
5973                .and_then(Option::as_ref)
5974                .is_some_and(|sieve| sieve.excludes(&probe.value)),
5975            None => false,
5976        }
5977    }
5978
5979    /// The sieves of one stripe of one column, read once and kept.
5980    ///
5981    /// `None` when the column has no sieves in that stripe, when the page is damaged, and when the
5982    /// bytes are not a page this version can read. A sieve is an index over data that is still there
5983    /// and a caller that cannot read one reads the rows, so this is the one place in the file where
5984    /// a bad checksum is a slow query rather than an error.
5985    fn stripe_sieves(&self, stripe: usize, column: usize) -> Option<&[Option<Sieve>]> {
5986        let slot = self.sieves.get(column)?.get(stripe)?;
5987        if let Some(held) = slot.get() {
5988            return Some(held);
5989        }
5990        let page = self.table.stripes.get(stripe)?.sieves.get(column)?;
5991        let mut bytes = vec![0; page.length as usize];
5992        read_at(&self.file, page.offset, &mut bytes).ok()?;
5993        if checksum(&bytes) != page.hash {
5994            return None;
5995        }
5996        let sieves = Arc::new(decode_sieves(&bytes).ok()?);
5997        let _ = slot.set(sieves);
5998        slot.get().map(|held| held.as_slice())
5999    }
6000}
6001
6002/// The value sitting at one position of a dictionary's sorted order.
6003fn text_at_rank(dictionary: &Vector, rank: usize) -> Result<Value> {
6004    let code = dictionary.code_at_rank(rank)? as usize;
6005    let text = dictionary
6006        .try_text_at(code)?
6007        .ok_or_else(|| invalid("global dictionary order names a code it does not have"))?;
6008    Ok(Value::Varchar(text.into()))
6009}
6010
6011/// Writes one span of a file at an offset, without depending on where the cursor is.
6012///
6013/// The writer owns an offset of its own and passes it in here, so that nothing it writes depends on
6014/// a cursor that a read is entitled to move. Both of these can come back short and both loop.
6015#[cfg(unix)]
6016fn write_at(file: &File, mut offset: u64, mut bytes: &[u8]) -> Result<()> {
6017    use std::os::unix::fs::FileExt;
6018    while !bytes.is_empty() {
6019        let written = file.write_at(bytes, offset).map_err(io)?;
6020        if written == 0 {
6021            return Err(invalid("a write to the native file wrote nothing"));
6022        }
6023        offset += written as u64;
6024        bytes = &bytes[written..];
6025    }
6026    Ok(())
6027}
6028
6029/// The same write, on the call Windows spells differently.
6030#[cfg(windows)]
6031fn write_at(file: &File, mut offset: u64, mut bytes: &[u8]) -> Result<()> {
6032    use std::os::windows::fs::FileExt;
6033    while !bytes.is_empty() {
6034        let written = file.seek_write(bytes, offset).map_err(io)?;
6035        if written == 0 {
6036            return Err(invalid("a write to the native file wrote nothing"));
6037        }
6038        offset += written as u64;
6039        bytes = &bytes[written..];
6040    }
6041    Ok(())
6042}
6043
6044/// Somewhere that is neither, where the cursor is all there is.
6045#[cfg(not(any(unix, windows)))]
6046fn write_at(file: &File, offset: u64, bytes: &[u8]) -> Result<()> {
6047    use std::io::Write;
6048    let mut file = file.try_clone().map_err(io)?;
6049    file.seek(SeekFrom::Start(offset)).map_err(io)?;
6050    file.write_all(bytes).map_err(io)
6051}
6052
6053/// Reads one span of a file at an offset, without moving a cursor anybody else can see.
6054///
6055/// Every reader of a table shares one [`File`] behind an [`Arc`], and a grouped aggregate reads its
6056/// pages from several threads at once, so this has to be positional. Seeking and then reading is
6057/// two calls with a gap in the middle, and in that gap another thread's seek lands and the read
6058/// comes back with somebody else's bytes.
6059///
6060/// Both of these can come back short, so both loop. A read of zero bytes before the span is filled
6061/// means the file stops earlier than the directory said it does.
6062#[cfg(unix)]
6063fn read_at(file: &File, mut offset: u64, mut bytes: &mut [u8]) -> Result<()> {
6064    use std::os::unix::fs::FileExt;
6065    while !bytes.is_empty() {
6066        let read = file.read_at(bytes, offset).map_err(io)?;
6067        if read == 0 {
6068            return Err(invalid("column page ends before its declared length"));
6069        }
6070        offset += read as u64;
6071        bytes = &mut bytes[read..];
6072    }
6073    Ok(())
6074}
6075
6076/// The same read, on the call Windows spells differently.
6077///
6078/// `seek_read` is one `ReadFile` carrying the offset with it, so two of them cannot interleave the
6079/// way a seek and a read can. It does leave the shared cursor somewhere afterwards, which is why
6080/// nothing in this file may read that cursor.
6081#[cfg(windows)]
6082fn read_at(file: &File, mut offset: u64, mut bytes: &mut [u8]) -> Result<()> {
6083    use std::os::windows::fs::FileExt;
6084    while !bytes.is_empty() {
6085        let read = file.seek_read(bytes, offset).map_err(io)?;
6086        if read == 0 {
6087            return Err(invalid("column page ends before its declared length"));
6088        }
6089        offset += read as u64;
6090        bytes = &mut bytes[read..];
6091    }
6092    Ok(())
6093}
6094
6095/// Somewhere that is neither, where the cursor is all there is.
6096///
6097/// This one does race, and there is no way to write it so it does not. Nothing we build for runs
6098/// here, so it exists to keep the crate compiling rather than to be correct under threads.
6099#[cfg(not(any(unix, windows)))]
6100fn read_at(file: &File, offset: u64, bytes: &mut [u8]) -> Result<()> {
6101    let mut file = file.try_clone().map_err(io)?;
6102    file.seek(SeekFrom::Start(offset)).map_err(io)?;
6103    file.read_exact(bytes).map_err(io)
6104}
6105
6106/// What a column type is called in the directory.
6107///
6108/// A tag is a number in a file somebody else wrote, so a tag that has been used is used forever and
6109/// the only thing that may happen to this list is that it grows. 1 to 13 are the tags the format
6110/// had when it could store thirteen types, and 14 to 27 are the rest, in the order they were added
6111/// rather than in an order that means anything.
6112fn type_tag(ty: &LogicalType) -> Result<u8> {
6113    match ty {
6114        LogicalType::SmallInt => Ok(1),
6115        LogicalType::Integer => Ok(2),
6116        LogicalType::BigInt => Ok(3),
6117        LogicalType::Varchar => Ok(4),
6118        LogicalType::Date => Ok(5),
6119        LogicalType::Timestamp => Ok(6),
6120        LogicalType::Boolean => Ok(7),
6121        LogicalType::TinyInt => Ok(8),
6122        LogicalType::UTinyInt => Ok(9),
6123        LogicalType::USmallInt => Ok(10),
6124        LogicalType::UInteger => Ok(11),
6125        LogicalType::UBigInt => Ok(12),
6126        LogicalType::Decimal { .. } => Ok(13),
6127        LogicalType::Float => Ok(14),
6128        LogicalType::Double => Ok(15),
6129        LogicalType::HugeInt => Ok(16),
6130        LogicalType::UHugeInt => Ok(17),
6131        LogicalType::Time => Ok(18),
6132        LogicalType::TimeTz => Ok(19),
6133        LogicalType::TimestampTz => Ok(20),
6134        LogicalType::Interval => Ok(21),
6135        LogicalType::Uuid => Ok(22),
6136        LogicalType::Blob => Ok(23),
6137        LogicalType::Bit => Ok(24),
6138        LogicalType::TimestampS => Ok(25),
6139        LogicalType::TimestampMs => Ok(26),
6140        LogicalType::TimestampNs => Ok(27),
6141        _ => Err(Error::not_implemented(format!("native storage for {ty}"))),
6142    }
6143}
6144
6145/// The tag of a column type, and the parameters of the ones that have any.
6146///
6147/// Only `DECIMAL` has parameters today. Width and scale go after the tag rather than into it
6148/// because they are what says how wide a value is on disk, and a reader that guessed would read the
6149/// wrong number of bytes per row rather than the wrong number of digits.
6150fn put_type(out: &mut Vec<u8>, ty: &LogicalType) -> Result<()> {
6151    out.push(type_tag(ty)?);
6152    if let LogicalType::Decimal { width, scale } = ty {
6153        out.push(*width);
6154        out.push(*scale);
6155    }
6156    Ok(())
6157}
6158
6159/// The other half of [`put_type`], reading the parameters the tag says are there.
6160fn read_type(cur: &mut Cursor<'_>) -> Result<LogicalType> {
6161    let tag = cur.u8()?;
6162    if tag == 13 {
6163        let width = cur.u8()?;
6164        let scale = cur.u8()?;
6165        return LogicalType::decimal(width, scale)
6166            .map_err(|_| invalid("decimal column width and scale are not a decimal"));
6167    }
6168    tag_type(tag)
6169}
6170
6171fn tag_type(tag: u8) -> Result<LogicalType> {
6172    match tag {
6173        1 => Ok(LogicalType::SmallInt),
6174        2 => Ok(LogicalType::Integer),
6175        3 => Ok(LogicalType::BigInt),
6176        4 => Ok(LogicalType::Varchar),
6177        5 => Ok(LogicalType::Date),
6178        6 => Ok(LogicalType::Timestamp),
6179        7 => Ok(LogicalType::Boolean),
6180        8 => Ok(LogicalType::TinyInt),
6181        9 => Ok(LogicalType::UTinyInt),
6182        10 => Ok(LogicalType::USmallInt),
6183        11 => Ok(LogicalType::UInteger),
6184        12 => Ok(LogicalType::UBigInt),
6185        14 => Ok(LogicalType::Float),
6186        15 => Ok(LogicalType::Double),
6187        16 => Ok(LogicalType::HugeInt),
6188        17 => Ok(LogicalType::UHugeInt),
6189        18 => Ok(LogicalType::Time),
6190        19 => Ok(LogicalType::TimeTz),
6191        20 => Ok(LogicalType::TimestampTz),
6192        21 => Ok(LogicalType::Interval),
6193        22 => Ok(LogicalType::Uuid),
6194        23 => Ok(LogicalType::Blob),
6195        24 => Ok(LogicalType::Bit),
6196        25 => Ok(LogicalType::TimestampS),
6197        26 => Ok(LogicalType::TimestampMs),
6198        27 => Ok(LogicalType::TimestampNs),
6199        _ => Err(invalid("column type tag is unknown")),
6200    }
6201}
6202
6203fn put_u16(out: &mut Vec<u8>, value: u16) {
6204    out.extend_from_slice(&value.to_le_bytes());
6205}
6206fn put_u32(out: &mut Vec<u8>, value: u32) {
6207    out.extend_from_slice(&value.to_le_bytes());
6208}
6209fn put_u64(out: &mut Vec<u8>, value: u64) {
6210    out.extend_from_slice(&value.to_le_bytes());
6211}
6212fn put_var_u64(out: &mut Vec<u8>, mut value: u64) {
6213    while value >= 0x80 {
6214        out.push((value as u8 & 0x7f) | 0x80);
6215        value >>= 7;
6216    }
6217    out.push(value as u8);
6218}
6219
6220fn frequency_order(left: FrequencyValue, right: FrequencyValue) -> Ordering {
6221    match (left, right) {
6222        (FrequencyValue::Null, FrequencyValue::Null) => Ordering::Equal,
6223        (FrequencyValue::Null, _) => Ordering::Less,
6224        (_, FrequencyValue::Null) => Ordering::Greater,
6225        (FrequencyValue::Integer(left), FrequencyValue::Integer(right)) => left.cmp(&right),
6226        (FrequencyValue::Code(left), FrequencyValue::Code(right)) => left.cmp(&right),
6227        (FrequencyValue::Integer(_), FrequencyValue::Code(_)) => Ordering::Less,
6228        (FrequencyValue::Code(_), FrequencyValue::Integer(_)) => Ordering::Greater,
6229    }
6230}
6231
6232/// Leaves the [`FREQUENCY_ENTRIES`] commonest entries in order and says what the next one counted.
6233///
6234/// There is one entry a distinct value, so on `URL` this is handed two and a quarter million of
6235/// them and keeps five hundred and twelve. Sorting all of them to throw almost all of them away is
6236/// the whole of what counting a dictionary column used to cost, 2.13 seconds of it on `URL` at eight
6237/// million rows against 11.93 for compressing the same column's values.
6238///
6239/// Partitioning answers both questions instead. It puts the five hundred and thirteenth entry where
6240/// it belongs and everything commoner in front of it, which is the entries to keep and the count to
6241/// report as the largest one omitted, and then only the part that survives is sorted. The order that
6242/// comes out is the order the sort gave, because the tie break makes the comparison total: two
6243/// entries never hold the same value.
6244fn keep_most_frequent(entries: &mut Vec<FrequencyEntry>) -> u64 {
6245    let order = |left: &FrequencyEntry, right: &FrequencyEntry| {
6246        right.count.cmp(&left.count).then_with(|| frequency_order(left.value, right.value))
6247    };
6248    let omitted_max = if entries.len() > FREQUENCY_ENTRIES {
6249        let (_, next, _) = entries.select_nth_unstable_by(FREQUENCY_ENTRIES, order);
6250        let omitted_max = next.count;
6251        entries.truncate(FREQUENCY_ENTRIES);
6252        omitted_max
6253    } else {
6254        0
6255    };
6256    entries.sort_unstable_by(order);
6257    omitted_max
6258}
6259
6260fn code_frequency(
6261    dictionary: &GlobalDictionary,
6262    flat: &[u8],
6263    bases: &[u64],
6264) -> Result<(FrequencySummary, Vec<Option<Vec<u8>>>)> {
6265    let mut entries = dictionary
6266        .counts
6267        .iter()
6268        .enumerate()
6269        .filter(|(_, count)| **count != 0)
6270        .map(|(code, &count)| FrequencyEntry { value: FrequencyValue::Code(code as u32), count })
6271        .collect::<Vec<_>>();
6272    if dictionary.nulls != 0 {
6273        entries.push(FrequencyEntry { value: FrequencyValue::Null, count: dictionary.nulls });
6274    }
6275    let omitted_max = keep_most_frequent(&mut entries);
6276    let mut spans = Vec::with_capacity(entries.len());
6277    let mut text_bytes = 0_usize;
6278    for entry in &entries {
6279        let span = match entry.value {
6280            FrequencyValue::Code(code) => {
6281                let span = GlobalDictionary::value_span(&dictionary.ends, bases, code as usize);
6282                let bytes = flat
6283                    .get(span.0..span.1)
6284                    .ok_or_else(|| invalid("a frequency code is outside its dictionary"))?;
6285                text_bytes = text_bytes.saturating_add(bytes.len());
6286                Some(span)
6287            }
6288            FrequencyValue::Null | FrequencyValue::Integer(_) => None,
6289        };
6290        spans.push(span);
6291    }
6292    let texts = if text_bytes > FREQUENCY_TEXT_BUDGET {
6293        Vec::new()
6294    } else {
6295        spans.into_iter().map(|span| span.map(|(from, to)| flat[from..to].to_vec())).collect()
6296    };
6297    Ok((
6298        FrequencySummary {
6299            entries,
6300            omitted_max,
6301            ordinals: Vec::new(),
6302            ordinal_entries: Vec::new(),
6303        },
6304        texts,
6305    ))
6306}
6307
6308fn encode_directory(table: &Table) -> Result<Vec<u8>> {
6309    let mut out = DIRECTORY.to_vec();
6310    let name = table.name.as_bytes();
6311    put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("table name too long"))?);
6312    out.extend_from_slice(name);
6313    put_u16(&mut out, u16::try_from(table.fields.len()).map_err(|_| invalid("too many columns"))?);
6314    for field in &table.fields {
6315        let name = field.name.as_bytes();
6316        put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("column name too long"))?);
6317        out.extend_from_slice(name);
6318        put_type(&mut out, &field.ty)?;
6319        out.push(u8::from(field.not_null));
6320    }
6321    for dictionary in &table.dictionaries {
6322        match dictionary {
6323            None => out.push(0),
6324            Some(page) => {
6325                out.push(1);
6326                put_u64(&mut out, page.offset);
6327                put_u32(&mut out, page.length);
6328                put_u64(&mut out, page.hash);
6329            }
6330        }
6331    }
6332    for distinct in &table.distincts {
6333        match distinct {
6334            None => out.push(0),
6335            Some(count) => {
6336                out.push(1);
6337                put_u64(&mut out, *count);
6338            }
6339        }
6340    }
6341    put_u64(&mut out, u64::try_from(table.rows).map_err(|_| invalid("row count overflow"))?);
6342    put_u32(&mut out, u32::try_from(table.stripes.len()).map_err(|_| invalid("too many stripes"))?);
6343    for stripe in &table.stripes {
6344        put_u32(
6345            &mut out,
6346            u32::try_from(stripe.parts.len()).map_err(|_| invalid("too many parts in a stripe"))?,
6347        );
6348        for &rows in &stripe.parts {
6349            put_u32(&mut out, rows);
6350        }
6351        put_u64(&mut out, stripe.index.offset);
6352        put_u32(&mut out, stripe.index.length);
6353        for page in &stripe.pages {
6354            put_u64(&mut out, page.offset);
6355            put_u32(&mut out, page.length);
6356        }
6357        // A membership index says which of a dictionary's codes a part holds, so a column the writer
6358        // decided against giving a dictionary has nothing for it to be about and writes none. Every
6359        // file written before that decision existed has a dictionary on every varchar column, so
6360        // this reads those files byte for byte the way it always did.
6361        for ((field, dictionary), membership) in
6362            table.fields.iter().zip(&table.dictionaries).zip(stripe.memberships.slots())
6363        {
6364            if field.ty != LogicalType::Varchar || dictionary.is_none() {
6365                continue;
6366            }
6367            let page =
6368                membership.ok_or_else(|| invalid("string page has no code membership index"))?;
6369            put_u64(&mut out, page.offset);
6370            put_u32(&mut out, page.length);
6371            put_u64(&mut out, page.hash);
6372        }
6373        for sieve in stripe.sieves.slots() {
6374            match sieve {
6375                None => out.push(0),
6376                Some(page) => {
6377                    out.push(1);
6378                    put_u64(&mut out, page.offset);
6379                    put_u32(&mut out, page.length);
6380                    put_u64(&mut out, page.hash);
6381                }
6382            }
6383        }
6384        for held in stripe.part_ranges.slots() {
6385            match held {
6386                None => out.push(0),
6387                Some(page) => {
6388                    out.push(1);
6389                    put_u64(&mut out, page.offset);
6390                    put_u32(&mut out, page.length);
6391                    put_u64(&mut out, page.hash);
6392                }
6393            }
6394        }
6395        for range in stripe.zone.columns() {
6396            put_bound(&mut out, range.low.as_ref())?;
6397            put_bound(&mut out, range.high.as_ref())?;
6398            put_u32(
6399                &mut out,
6400                u32::try_from(range.nulls).map_err(|_| invalid("null count overflow"))?,
6401            );
6402            out.push(u8::from(range.exact));
6403            match range.sum {
6404                None => out.push(0),
6405                Some(total) => {
6406                    out.push(1);
6407                    out.extend_from_slice(&total.to_le_bytes());
6408                }
6409            }
6410        }
6411    }
6412    out.extend_from_slice(FREQUENCIES);
6413    put_u16(
6414        &mut out,
6415        u16::try_from(table.frequencies.len())
6416            .map_err(|_| invalid("too many frequency columns"))?,
6417    );
6418    for summary in &table.frequencies {
6419        let summary = match summary {
6420            None => {
6421                out.push(0);
6422                continue;
6423            }
6424            Some(Frequencies::Held(summary)) => summary,
6425            // Only a reader leaves a synopsis in the file, and nothing writes a reader's table back.
6426            Some(Frequencies::Stored { .. }) => {
6427                return Err(invalid("a synopsis left in the file cannot be written back"));
6428            }
6429        };
6430        out.push(1);
6431        put_u64(&mut out, summary.omitted_max);
6432        put_u32(
6433            &mut out,
6434            u32::try_from(summary.entries.len())
6435                .map_err(|_| invalid("too many frequency entries"))?,
6436        );
6437        for entry in &summary.entries {
6438            match entry.value {
6439                FrequencyValue::Null => out.push(0),
6440                FrequencyValue::Integer(value) => {
6441                    out.push(1);
6442                    out.extend_from_slice(&value.to_le_bytes());
6443                }
6444                FrequencyValue::Code(value) => {
6445                    out.push(2);
6446                    put_u32(&mut out, value);
6447                }
6448            }
6449            put_u64(&mut out, entry.count);
6450        }
6451        put_u32(
6452            &mut out,
6453            u32::try_from(summary.ordinals.len())
6454                .map_err(|_| invalid("too many frequency ordinals"))?,
6455        );
6456        let mut previous = 0_u64;
6457        for (at, &ordinal) in summary.ordinals.iter().enumerate() {
6458            let delta = if at == 0 {
6459                ordinal
6460            } else {
6461                ordinal
6462                    .checked_sub(previous)
6463                    .ok_or_else(|| invalid("frequency ordinals are not ordered"))?
6464            };
6465            if at != 0 && delta == 0 {
6466                return Err(invalid("frequency ordinals are not unique"));
6467            }
6468            put_var_u64(&mut out, delta);
6469            previous = ordinal;
6470        }
6471        if summary.ordinal_entries.len() != summary.ordinals.len() {
6472            return Err(invalid("frequency ordinal values have a different length"));
6473        }
6474        for &entry in &summary.ordinal_entries {
6475            if entry as usize >= summary.entries.len() {
6476                return Err(invalid("frequency ordinal value is outside its entries"));
6477            }
6478            put_u16(&mut out, entry);
6479        }
6480    }
6481    if !table.pair_frequencies.is_empty() {
6482        out.extend_from_slice(PAIR_FREQUENCIES);
6483        put_u16(
6484            &mut out,
6485            u16::try_from(table.pair_frequencies.len())
6486                .map_err(|_| invalid("too many pair frequency summaries"))?,
6487        );
6488        for summary in &table.pair_frequencies {
6489            put_u16(&mut out, summary.first);
6490            put_u16(&mut out, summary.second);
6491            put_u64(&mut out, summary.omitted_max);
6492            put_u16(
6493                &mut out,
6494                u16::try_from(summary.entries.len())
6495                    .map_err(|_| invalid("too many pair frequency entries"))?,
6496            );
6497            for entry in &summary.entries {
6498                put_u16(&mut out, entry.first_entry);
6499                match entry.second {
6500                    None => out.push(0),
6501                    Some(code) => {
6502                        out.push(1);
6503                        put_u32(&mut out, code);
6504                    }
6505                }
6506                put_u64(&mut out, entry.count);
6507            }
6508        }
6509    }
6510    let text_columns = table.frequency_texts.iter().filter(|texts| !texts.is_empty()).count();
6511    if text_columns != 0 {
6512        out.extend_from_slice(FREQUENCY_TEXTS);
6513        put_u16(
6514            &mut out,
6515            u16::try_from(text_columns)
6516                .map_err(|_| invalid("too many string frequency columns"))?,
6517        );
6518        for (column, texts) in table.frequency_texts.iter().enumerate() {
6519            if texts.is_empty() {
6520                continue;
6521            }
6522            put_u16(
6523                &mut out,
6524                u16::try_from(column).map_err(|_| invalid("frequency text column overflows"))?,
6525            );
6526            put_u16(
6527                &mut out,
6528                u16::try_from(texts.len())
6529                    .map_err(|_| invalid("too many frequency text entries"))?,
6530            );
6531            for text in texts {
6532                match text {
6533                    None => out.push(0),
6534                    Some(text) => {
6535                        out.push(1);
6536                        put_u32(
6537                            &mut out,
6538                            u32::try_from(text.len())
6539                                .map_err(|_| invalid("frequency text is too long"))?,
6540                        );
6541                        out.extend_from_slice(text);
6542                    }
6543                }
6544            }
6545        }
6546    }
6547    if let Some(summary) = &table.host_groups {
6548        out.extend_from_slice(HOST_GROUPS);
6549        put_u16(
6550            &mut out,
6551            u16::try_from(summary.column).map_err(|_| invalid("host column overflows"))?,
6552        );
6553        put_u64(&mut out, summary.omitted_max);
6554        put_u16(
6555            &mut out,
6556            u16::try_from(summary.entries.len()).map_err(|_| invalid("too many host groups"))?,
6557        );
6558        for entry in &summary.entries {
6559            put_u32(
6560                &mut out,
6561                u32::try_from(entry.host.len()).map_err(|_| invalid("host name is too long"))?,
6562            );
6563            out.extend_from_slice(entry.host.as_bytes());
6564            put_u64(&mut out, entry.count);
6565            out.extend_from_slice(&entry.bytes_sum.to_le_bytes());
6566            put_u32(
6567                &mut out,
6568                u32::try_from(entry.minimum.len())
6569                    .map_err(|_| invalid("host minimum is too long"))?,
6570            );
6571            out.extend_from_slice(entry.minimum.as_bytes());
6572        }
6573    }
6574    // Written only when there is a declaration, so that the common file is the same bytes it was
6575    // and the section is not a byte of zero on every table in the world that never asked for one.
6576    if let Some(clustering) = &table.clustering {
6577        out.extend_from_slice(CLUSTERING);
6578        out.push(clustering.width().tag());
6579        put_u16(
6580            &mut out,
6581            u16::try_from(clustering.columns().len())
6582                .map_err(|_| invalid("too many clustering columns"))?,
6583        );
6584        for &column in clustering.columns() {
6585            put_u16(
6586                &mut out,
6587                u16::try_from(column).map_err(|_| invalid("clustering column index overflow"))?,
6588            );
6589        }
6590    }
6591    // The section table, last, behind its own magic, for the same reason the frequency block is
6592    // behind its own: a reader that stops before it gets a table with no sections, and a table with
6593    // no sections is a correct table. The one difference from the blocks before it is that this one
6594    // is written even when it is empty, so that a file written by this build always says which
6595    // sections it has rather than leaving a reader to infer it from where the bytes ran out.
6596    out.extend_from_slice(SECTIONS);
6597    put_u64(&mut out, table.generation);
6598    put_u16(
6599        &mut out,
6600        u16::try_from(table.sections.len()).map_err(|_| invalid("too many sections"))?,
6601    );
6602    for held in &table.sections {
6603        held.encode(&mut out)?;
6604    }
6605    if table.dictionary_payloads.iter().any(|&bytes| bytes != 0) {
6606        out.extend_from_slice(DICTIONARY_PAYLOADS);
6607        put_u16(
6608            &mut out,
6609            u16::try_from(table.fields.len()).map_err(|_| invalid("too many columns"))?,
6610        );
6611        for at in 0..table.fields.len() {
6612            put_u64(&mut out, table.dictionary_payloads.get(at).copied().unwrap_or(0));
6613        }
6614    }
6615    Ok(out)
6616}
6617
6618/// The small level of the directory, naming every table in the file.
6619///
6620/// This is what a footer slot points at. Each entry carries its own checksum over its table
6621/// directory, so a table whose directory is torn is found when that table is first touched rather
6622/// than being trusted because the catalog around it checksummed.
6623///
6624/// The views go after the tables and are whole here, since a view is text and a column list and has
6625/// no pages for a second level to point at.
6626fn table_nonzero_counts(table: &Table) -> Vec<Option<u64>> {
6627    table
6628        .fields
6629        .iter()
6630        .enumerate()
6631        .map(|(column, field)| {
6632            if !matches!(
6633                field.ty,
6634                LogicalType::TinyInt
6635                    | LogicalType::SmallInt
6636                    | LogicalType::Integer
6637                    | LogicalType::BigInt
6638                    | LogicalType::UTinyInt
6639                    | LogicalType::USmallInt
6640                    | LogicalType::UInteger
6641                    | LogicalType::UBigInt
6642            ) {
6643                return None;
6644            }
6645            let Some(Frequencies::Held(summary)) = &table.frequencies[column] else {
6646                return None;
6647            };
6648            let zero = summary
6649                .entries
6650                .iter()
6651                .find(|entry| entry.value == FrequencyValue::Integer(0))
6652                .map(|entry| entry.count)
6653                .or_else(|| (summary.omitted_max == 0).then_some(0))?;
6654            let nulls = table.stripes.iter().try_fold(0_u64, |count, stripe| {
6655                count.checked_add(stripe.zone.column(column)?.nulls as u64)
6656            })?;
6657            (table.rows as u64).checked_sub(nulls)?.checked_sub(zero)
6658        })
6659        .collect()
6660}
6661
6662fn signed_integer(ty: &LogicalType) -> bool {
6663    matches!(
6664        ty,
6665        LogicalType::TinyInt | LogicalType::SmallInt | LogicalType::Integer | LogicalType::BigInt
6666    )
6667}
6668
6669fn table_exact_sum(table: &Table, column: usize) -> Option<(i128, u64)> {
6670    table.stripes.iter().try_fold((0_i128, 0_u64), |(sum, count), stripe| {
6671        let range = stripe.zone.column(column)?;
6672        let sum = sum.checked_add(range.sum?)?;
6673        let nonnull = (stripe.rows as u64).checked_sub(range.nulls as u64)?;
6674        Some((sum, count.checked_add(nonnull)?))
6675    })
6676}
6677
6678fn table_aggregate_sums(table: &Table) -> Vec<Option<(i128, u64)>> {
6679    table
6680        .fields
6681        .iter()
6682        .enumerate()
6683        .map(|(column, field)| {
6684            signed_integer(&field.ty).then(|| table_exact_sum(table, column)).flatten()
6685        })
6686        .collect()
6687}
6688
6689fn reader_nonzero_counts(reader: &Reader) -> Result<Vec<Option<u64>>> {
6690    reader
6691        .table
6692        .fields
6693        .iter()
6694        .enumerate()
6695        .map(|(column, field)| {
6696            if !matches!(
6697                field.ty,
6698                LogicalType::TinyInt
6699                    | LogicalType::SmallInt
6700                    | LogicalType::Integer
6701                    | LogicalType::BigInt
6702                    | LogicalType::UTinyInt
6703                    | LogicalType::USmallInt
6704                    | LogicalType::UInteger
6705                    | LogicalType::UBigInt
6706            ) {
6707                return Ok(None);
6708            }
6709            let Some(summary) = reader.frequency_summary(column)? else {
6710                return Ok(None);
6711            };
6712            let zero = summary
6713                .entries
6714                .iter()
6715                .find(|entry| entry.value == FrequencyValue::Integer(0))
6716                .map(|entry| entry.count)
6717                .or_else(|| (summary.omitted_max == 0).then_some(0));
6718            let Some(zero) = zero else { return Ok(None) };
6719            let nulls = reader.null_count(column)?;
6720            Ok((reader.table.rows as u64)
6721                .checked_sub(nulls)
6722                .and_then(|count| count.checked_sub(zero)))
6723        })
6724        .collect()
6725}
6726
6727fn reader_aggregate_sums(reader: &Reader) -> Result<Vec<Option<(i128, u64)>>> {
6728    reader
6729        .table
6730        .fields
6731        .iter()
6732        .enumerate()
6733        .map(
6734            |(column, field)| {
6735                if signed_integer(&field.ty) { reader.exact_sum(column) } else { Ok(None) }
6736            },
6737        )
6738        .collect()
6739}
6740
6741fn encode_catalog(entries: &[Entry], views: &[ViewEntry]) -> Result<Vec<u8>> {
6742    let mut out = CATALOG.to_vec();
6743    put_u32(&mut out, u32::try_from(entries.len()).map_err(|_| invalid("too many tables"))?);
6744    for entry in entries {
6745        let name = entry.name.as_bytes();
6746        put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("table name too long"))?);
6747        out.extend_from_slice(name);
6748        put_u64(&mut out, u64::try_from(entry.rows).map_err(|_| invalid("row count overflow"))?);
6749        put_u16(
6750            &mut out,
6751            u16::try_from(entry.fields.len()).map_err(|_| invalid("too many columns"))?,
6752        );
6753        for field in &entry.fields {
6754            let name = field.name.as_bytes();
6755            put_u16(
6756                &mut out,
6757                u16::try_from(name.len()).map_err(|_| invalid("column name too long"))?,
6758            );
6759            out.extend_from_slice(name);
6760            put_type(&mut out, &field.ty)?;
6761            out.push(u8::from(field.not_null));
6762        }
6763        put_u64(&mut out, entry.directory.offset);
6764        put_u32(&mut out, entry.directory.length);
6765        put_u64(&mut out, entry.directory.hash);
6766    }
6767    put_u32(&mut out, u32::try_from(views.len()).map_err(|_| invalid("too many views"))?);
6768    for view in views {
6769        let name = view.name.as_bytes();
6770        put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("view name too long"))?);
6771        out.extend_from_slice(name);
6772        put_long_text(&mut out, &view.sql, "view body")?;
6773        put_long_text(&mut out, &view.statement, "view statement")?;
6774        put_u16(
6775            &mut out,
6776            u16::try_from(view.aliases.len()).map_err(|_| invalid("too many aliases"))?,
6777        );
6778        for alias in &view.aliases {
6779            let alias = alias.as_bytes();
6780            put_u16(
6781                &mut out,
6782                u16::try_from(alias.len()).map_err(|_| invalid("alias name too long"))?,
6783            );
6784            out.extend_from_slice(alias);
6785        }
6786        put_u16(
6787            &mut out,
6788            u16::try_from(view.columns.len()).map_err(|_| invalid("too many columns"))?,
6789        );
6790        for field in &view.columns {
6791            let name = field.name.as_bytes();
6792            put_u16(
6793                &mut out,
6794                u16::try_from(name.len()).map_err(|_| invalid("column name too long"))?,
6795            );
6796            out.extend_from_slice(name);
6797            put_type(&mut out, &field.ty)?;
6798            out.push(u8::from(field.not_null));
6799        }
6800    }
6801    out.extend_from_slice(NONZERO_COUNTS);
6802    for entry in entries {
6803        if entry.nonzero.len() != entry.fields.len() {
6804            return Err(invalid("nonzero count width differs from schema"));
6805        }
6806        for count in &entry.nonzero {
6807            match count {
6808                None => out.push(0),
6809                Some(count) => {
6810                    out.push(1);
6811                    put_u64(&mut out, *count);
6812                }
6813            }
6814        }
6815    }
6816    out.extend_from_slice(AGGREGATE_SUMS);
6817    for entry in entries {
6818        if entry.aggregates.len() != entry.fields.len() {
6819            return Err(invalid("aggregate sum width differs from schema"));
6820        }
6821        for summary in &entry.aggregates {
6822            match summary {
6823                None => out.push(0),
6824                Some((sum, count)) => {
6825                    out.push(1);
6826                    out.extend_from_slice(&sum.to_le_bytes());
6827                    put_u64(&mut out, *count);
6828                }
6829            }
6830        }
6831    }
6832    Ok(out)
6833}
6834
6835/// A length and that many bytes, for text that is allowed to be longer than a name.
6836fn put_long_text(out: &mut Vec<u8>, text: &str, what: &str) -> Result<()> {
6837    let bytes = text.as_bytes();
6838    put_u32(out, u32::try_from(bytes.len()).map_err(|_| invalid(&format!("{what} too long")))?);
6839    out.extend_from_slice(bytes);
6840    Ok(())
6841}
6842
6843/// Reads the catalog directory back, checking every span against the file before anything is
6844/// allocated for it.
6845fn decode_catalog(bytes: &[u8], size: u64) -> Result<(Vec<Entry>, Vec<ViewEntry>)> {
6846    let mut cur = Cursor::new(bytes);
6847    if cur.take(8)? != CATALOG {
6848        return Err(invalid("catalog magic differs"));
6849    }
6850    let count = cur.u32()? as usize;
6851    let mut entries: Vec<Entry> = Vec::with_capacity(count.min(1024));
6852    for _ in 0..count {
6853        let name = cur.text()?;
6854        let rows = usize::try_from(cur.u64()?).map_err(|_| invalid("row count does not fit"))?;
6855        let width = cur.u16()? as usize;
6856        let mut fields = Vec::with_capacity(width);
6857        for _ in 0..width {
6858            let name = cur.text()?;
6859            let ty = read_type(&mut cur)?;
6860            let not_null = match cur.u8()? {
6861                0 => false,
6862                1 => true,
6863                _ => return Err(invalid("nullability flag differs")),
6864            };
6865            fields.push(Field { name, ty, not_null });
6866        }
6867        let directory = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
6868        let end = directory
6869            .offset
6870            .checked_add(u64::from(directory.length))
6871            .ok_or_else(|| invalid("table directory offset overflow"))?;
6872        if directory.offset < HEADER
6873            || end > size
6874            || directory.length as usize > MAX_DIRECTORY
6875            || directory.length == 0
6876        {
6877            return Err(invalid("table directory range is outside the file"));
6878        }
6879        if entries.iter().any(|held| held.name == name) {
6880            return Err(invalid("two tables in the catalog have the same name"));
6881        }
6882        let nonzero = vec![None; fields.len()];
6883        let aggregates = vec![None; fields.len()];
6884        entries.push(Entry { name, fields, rows, directory, nonzero, aggregates });
6885    }
6886    // A catalog that ends where the tables end is a catalog with no views in it, which is every
6887    // file written before format 25. That is why the count is allowed to be missing rather than
6888    // read as a zero that has to be there: an older file has nothing after the last table entry at
6889    // all, and [`READABLE`] says those files still open.
6890    let count = if cur.done() { 0 } else { cur.u32()? as usize };
6891    let mut views: Vec<ViewEntry> = Vec::with_capacity(count.min(1024));
6892    for _ in 0..count {
6893        let name = cur.text()?;
6894        let sql = cur.long_text()?;
6895        let statement = cur.long_text()?;
6896        let width = cur.u16()? as usize;
6897        let mut aliases = Vec::with_capacity(width);
6898        for _ in 0..width {
6899            aliases.push(cur.text()?);
6900        }
6901        let width = cur.u16()? as usize;
6902        let mut columns = Vec::with_capacity(width);
6903        for _ in 0..width {
6904            let name = cur.text()?;
6905            let ty = read_type(&mut cur)?;
6906            let not_null = match cur.u8()? {
6907                0 => false,
6908                1 => true,
6909                _ => return Err(invalid("nullability flag differs")),
6910            };
6911            columns.push(Field { name, ty, not_null });
6912        }
6913        // The same rule the tables above get, and for the same reason. Two entries under one name
6914        // is a catalog nothing can answer a lookup from, and finding that out here is better than
6915        // finding it out from whichever of the two a search happened to reach first.
6916        if views.iter().any(|held| held.name == name) {
6917            return Err(invalid("two views in the catalog have the same name"));
6918        }
6919        if entries.iter().any(|held| held.name == name) {
6920            return Err(invalid("a table and a view in the catalog have the same name"));
6921        }
6922        views.push(ViewEntry { name, sql, statement, aliases, columns });
6923    }
6924    if !cur.done() {
6925        if cur.take(8)? != NONZERO_COUNTS {
6926            return Err(invalid("catalog extension magic differs"));
6927        }
6928        for entry in &mut entries {
6929            for (field, count) in entry.fields.iter().zip(&mut entry.nonzero) {
6930                *count = match cur.u8()? {
6931                    0 => None,
6932                    1 if matches!(
6933                        field.ty,
6934                        LogicalType::TinyInt
6935                            | LogicalType::SmallInt
6936                            | LogicalType::Integer
6937                            | LogicalType::BigInt
6938                            | LogicalType::UTinyInt
6939                            | LogicalType::USmallInt
6940                            | LogicalType::UInteger
6941                            | LogicalType::UBigInt
6942                    ) =>
6943                    {
6944                        let value = cur.u64()?;
6945                        if value > entry.rows as u64 {
6946                            return Err(invalid("nonzero count exceeds rows"));
6947                        }
6948                        Some(value)
6949                    }
6950                    _ => return Err(invalid("nonzero count tag or column type differs")),
6951                };
6952            }
6953        }
6954    }
6955    if !cur.done() {
6956        if cur.take(8)? != AGGREGATE_SUMS {
6957            return Err(invalid("aggregate catalog extension magic differs"));
6958        }
6959        for entry in &mut entries {
6960            for (field, summary) in entry.fields.iter().zip(&mut entry.aggregates) {
6961                *summary = match cur.u8()? {
6962                    0 => None,
6963                    1 if signed_integer(&field.ty) => {
6964                        let sum = i128::from_le_bytes(
6965                            cur.take(16)?
6966                                .try_into()
6967                                .map_err(|_| invalid("aggregate sum is truncated"))?,
6968                        );
6969                        let count = cur.u64()?;
6970                        if count > entry.rows as u64 {
6971                            return Err(invalid("aggregate count exceeds table rows"));
6972                        }
6973                        Some((sum, count))
6974                    }
6975                    _ => return Err(invalid("aggregate sum tag or column type differs")),
6976                };
6977            }
6978        }
6979    }
6980    if !cur.done() {
6981        return Err(invalid("catalog has trailing bytes"));
6982    }
6983    Ok((entries, views))
6984}
6985
6986/// Reads the fields of a directory or a catalog in order, off bytes in memory or out of the file.
6987///
6988/// A catalog is small and is read whole. A table directory is not: at ten million rows of `hits` it
6989/// is nearly a megabyte, and holding that buffer while the table it describes is built out of it
6990/// put both at the peak of every query. Out of the file, the cursor holds one window of
6991/// [`DIRECTORY_WINDOW`] bytes and moves it forward as the fields are read, so what a directory
6992/// costs at open is what it decodes into and not that plus its own bytes.
6993struct Cursor<'a> {
6994    bytes: &'a [u8],
6995    at: usize,
6996    window: Option<Window<'a>>,
6997}
6998
6999/// The part of a directory in the file that a [`Cursor`] has read in.
7000struct Window<'a> {
7001    file: &'a File,
7002    offset: u64,
7003    length: usize,
7004    /// Where `held` starts, counted from the start of the directory.
7005    start: usize,
7006    held: Vec<u8>,
7007    /// How much to read at once, which is [`DIRECTORY_WINDOW`] outside the tests.
7008    size: usize,
7009}
7010
7011/// How much of a directory a cursor reading one out of the file holds at once.
7012const DIRECTORY_WINDOW: usize = 64 << 10;
7013
7014impl<'a> Cursor<'a> {
7015    fn new(bytes: &'a [u8]) -> Self {
7016        Self { bytes, at: 0, window: None }
7017    }
7018
7019    /// A cursor over `length` bytes of `file` from `offset`, which it reads a window at a time.
7020    fn over(file: &'a File, offset: u64, length: usize) -> Self {
7021        let window =
7022            Window { file, offset, length, start: 0, held: Vec::new(), size: DIRECTORY_WINDOW };
7023        Self { bytes: &[], at: 0, window: Some(window) }
7024    }
7025
7026    /// How many bytes the cursor walks in all.
7027    fn len(&self) -> usize {
7028        self.window.as_ref().map_or(self.bytes.len(), |window| window.length)
7029    }
7030
7031    /// Makes sure the next `len` bytes are in memory.
7032    fn ensure(&mut self, len: usize) -> Result<()> {
7033        let end = self.at.checked_add(len).ok_or_else(|| invalid("directory offset overflow"))?;
7034        if end > self.len() {
7035            return Err(invalid("directory is truncated"));
7036        }
7037        let Some(window) = &mut self.window else { return Ok(()) };
7038        if self.at < window.start || end > window.start + window.held.len() {
7039            let want = len.max(window.size).min(window.length - self.at);
7040            window.start = self.at;
7041            window.held.resize(want, 0);
7042            read_at(window.file, window.offset + self.at as u64, &mut window.held)?;
7043        }
7044        Ok(())
7045    }
7046
7047    /// `len` bytes from `at`, which [`Self::ensure`] has already brought in.
7048    fn held(&self, at: usize, len: usize) -> &[u8] {
7049        match &self.window {
7050            Some(window) => &window.held[at - window.start..at - window.start + len],
7051            None => &self.bytes[at..at + len],
7052        }
7053    }
7054
7055    /// The next `len` bytes, without moving past them.
7056    #[inline]
7057    fn peek(&mut self, len: usize) -> Result<&[u8]> {
7058        if self.window.is_none() {
7059            let bytes = self.bytes;
7060            return Ok(&bytes[self.at..self.end(len)?]);
7061        }
7062        self.ensure(len)?;
7063        Ok(self.held(self.at, len))
7064    }
7065
7066    /// The next `len` bytes, moving past them.
7067    ///
7068    /// Every data page is decoded through this, a byte or a word at a time, so a cursor over bytes
7069    /// already in memory takes them here and never reaches [`Self::ensure`]. With the window check
7070    /// on every call, q06 on TPC-H spent a seventh of its instructions in it.
7071    #[inline]
7072    fn take(&mut self, len: usize) -> Result<&[u8]> {
7073        if self.window.is_none() {
7074            let bytes = self.bytes;
7075            let (at, end) = (self.at, self.end(len)?);
7076            self.at = end;
7077            return Ok(&bytes[at..end]);
7078        }
7079        self.take_windowed(len)
7080    }
7081
7082    /// Moves over a checked field without reading its payload from a windowed directory.
7083    fn skip(&mut self, len: usize) -> Result<()> {
7084        let end = self.at.checked_add(len).ok_or_else(|| invalid("directory offset overflow"))?;
7085        if end > self.len() {
7086            return Err(invalid("directory is truncated"));
7087        }
7088        self.at = end;
7089        Ok(())
7090    }
7091
7092    fn skip_bound(&mut self) -> Result<()> {
7093        match self.u8()? {
7094            0 => Ok(()),
7095            1 => self.skip(16),
7096            2 => self.skip(8),
7097            3 => {
7098                let length = self.u32()? as usize;
7099                self.skip(length)
7100            }
7101            4 => self.skip(17),
7102            _ => Err(invalid("a stored bound has an unknown tag")),
7103        }
7104    }
7105
7106    /// Where `len` bytes from here end, when they end inside the bytes.
7107    #[inline]
7108    fn end(&self, len: usize) -> Result<usize> {
7109        let end = self.at.checked_add(len).ok_or_else(|| invalid("directory offset overflow"))?;
7110        if end > self.bytes.len() {
7111            return Err(invalid("directory is truncated"));
7112        }
7113        Ok(end)
7114    }
7115
7116    /// [`Self::take`] out of the file, a window at a time.
7117    #[inline(never)]
7118    fn take_windowed(&mut self, len: usize) -> Result<&[u8]> {
7119        self.ensure(len)?;
7120        self.at += len;
7121        Ok(self.held(self.at - len, len))
7122    }
7123    #[inline]
7124    fn u8(&mut self) -> Result<u8> {
7125        Ok(self.take(1)?[0])
7126    }
7127    #[inline]
7128    fn u16(&mut self) -> Result<u16> {
7129        Ok(u16::from_le_bytes(self.take(2)?.try_into().expect("two bytes")))
7130    }
7131    #[inline]
7132    fn u32(&mut self) -> Result<u32> {
7133        Ok(u32::from_le_bytes(self.take(4)?.try_into().expect("four bytes")))
7134    }
7135    #[inline]
7136    fn u64(&mut self) -> Result<u64> {
7137        Ok(u64::from_le_bytes(self.take(8)?.try_into().expect("eight bytes")))
7138    }
7139    fn var_u64(&mut self) -> Result<u64> {
7140        let mut value = 0_u64;
7141        for shift in (0..=63).step_by(7) {
7142            let byte = self.u8()?;
7143            let part = u64::from(byte & 0x7f);
7144            if shift == 63 && part > 1 {
7145                return Err(invalid("frequency ordinal varint overflows"));
7146            }
7147            value |= part << shift;
7148            if byte & 0x80 == 0 {
7149                return Ok(value);
7150            }
7151        }
7152        Err(invalid("frequency ordinal varint is too long"))
7153    }
7154    /// A zone map's end, in the layout `rudb_common::bounds` defines.
7155    ///
7156    /// The bytes are the ones this directory has written since format 10 and the codec moved to
7157    /// rank zero rather than being copied, because a column summary now writes the same two ends
7158    /// and two encodings of one type is how the two quietly stop agreeing.
7159    ///
7160    /// A bound's length is in the bound, so out of the file the cursor offers the codec a few bytes
7161    /// and offers it twice as many whenever it runs out before the directory does.
7162    fn bound(&mut self) -> Result<Option<Bound>> {
7163        let rest = self.len().saturating_sub(self.at);
7164        let mut want = 32;
7165        loop {
7166            let offered = self.peek(want.min(rest))?;
7167            let mut used = 0;
7168            match bounds::get(offered, &mut used) {
7169                Ok(bound) => {
7170                    self.at += used;
7171                    return Ok(bound);
7172                }
7173                Err(_) if want < rest => want *= 2,
7174                Err(error) => return Err(error),
7175            }
7176        }
7177    }
7178    fn text(&mut self) -> Result<String> {
7179        let len = self.u16()? as usize;
7180        String::from_utf8(self.take(len)?.to_vec()).map_err(|_| invalid("name is not UTF-8"))
7181    }
7182    /// Whether everything has been read, which is how a section that an older file does not have at
7183    /// all is told from one that is there and empty.
7184    fn done(&self) -> bool {
7185        self.at >= self.len()
7186    }
7187    /// The same, for text that is a query rather than a name.
7188    ///
7189    /// A name fits in sixteen bits of length and a view body does not have to. Nobody writes a 64
7190    /// kilobyte identifier by accident and people do write generated queries that long, and a view
7191    /// that could not be written down because its body was too big would be a limit invented here
7192    /// rather than one anything else in the engine has.
7193    fn long_text(&mut self) -> Result<String> {
7194        let len = self.u32()? as usize;
7195        String::from_utf8(self.take(len)?.to_vec()).map_err(|_| invalid("text is not UTF-8"))
7196    }
7197}
7198
7199/// One column's frequency synopsis, or `None` for a column that has none, checked against the column.
7200fn decode_summary(
7201    cur: &mut Cursor<'_>,
7202    field: &Field,
7203    rows: usize,
7204    values: bool,
7205) -> Result<Option<FrequencySummary>> {
7206    Ok(match cur.u8()? {
7207        0 => None,
7208        1 => {
7209            let omitted_max = cur.u64()?;
7210            let count = cur.u32()? as usize;
7211            if count > FREQUENCY_ENTRIES {
7212                return Err(invalid("frequency entry count exceeds its bound"));
7213            }
7214            let mut entries = Vec::with_capacity(count);
7215            // row at a time: directory decoding validates each persisted bounded frequency entry.
7216            for _ in 0..count {
7217                let value = match cur.u8()? {
7218                    0 => FrequencyValue::Null,
7219                    1 => FrequencyValue::Integer(i128::from_le_bytes(
7220                        cur.take(16)?.try_into().expect("sixteen bytes"),
7221                    )),
7222                    2 => FrequencyValue::Code(cur.u32()?),
7223                    _ => return Err(invalid("frequency value tag differs")),
7224                };
7225                let valid = matches!(
7226                    (&field.ty, value),
7227                    (_, FrequencyValue::Null)
7228                        | (LogicalType::Varchar, FrequencyValue::Code(_))
7229                        | (
7230                            LogicalType::TinyInt
7231                                | LogicalType::SmallInt
7232                                | LogicalType::Integer
7233                                | LogicalType::BigInt
7234                                | LogicalType::UTinyInt
7235                                | LogicalType::USmallInt
7236                                | LogicalType::UInteger
7237                                | LogicalType::UBigInt
7238                                | LogicalType::Date
7239                                | LogicalType::Timestamp,
7240                            FrequencyValue::Integer(_),
7241                        )
7242                );
7243                if !valid {
7244                    return Err(invalid("frequency value does not match its column"));
7245                }
7246                let count = cur.u64()?;
7247                if count == 0 || count > rows as u64 {
7248                    return Err(invalid("frequency count is outside the table"));
7249                }
7250                entries.push(FrequencyEntry { value, count });
7251            }
7252            if entries.windows(2).any(|pair| pair[0].count < pair[1].count) {
7253                return Err(invalid("frequency entries are not descending"));
7254            }
7255            let ordinals = {
7256                let ordinal_count = cur.u32()? as usize;
7257                if ordinal_count > FREQUENCY_ORDINALS || ordinal_count > rows {
7258                    return Err(invalid("frequency ordinal count exceeds its bound"));
7259                }
7260                let mut ordinals = Vec::with_capacity(ordinal_count);
7261                let mut previous = 0_u64;
7262                for at in 0..ordinal_count {
7263                    let delta = cur.var_u64()?;
7264                    if at != 0 && delta == 0 {
7265                        return Err(invalid("frequency ordinals are not increasing"));
7266                    }
7267                    let ordinal = if at == 0 {
7268                        delta
7269                    } else {
7270                        previous
7271                            .checked_add(delta)
7272                            .ok_or_else(|| invalid("frequency ordinal overflows"))?
7273                    };
7274                    if ordinal >= rows as u64 {
7275                        return Err(invalid("frequency ordinal is outside the table"));
7276                    }
7277                    ordinals.push(ordinal);
7278                    previous = ordinal;
7279                }
7280                ordinals
7281            };
7282            let ordinal_entries = if values {
7283                let mut ordinal_entries = Vec::with_capacity(ordinals.len());
7284                for _ in 0..ordinals.len() {
7285                    let entry = cur.u16()?;
7286                    if entry as usize >= entries.len() {
7287                        return Err(invalid("frequency ordinal value is outside its entries"));
7288                    }
7289                    ordinal_entries.push(entry);
7290                }
7291                ordinal_entries
7292            } else {
7293                Vec::new()
7294            };
7295            Some(FrequencySummary { entries, omitted_max, ordinals, ordinal_entries })
7296        }
7297        _ => return Err(invalid("frequency summary tag differs")),
7298    })
7299}
7300
7301/// Skips a synopsis whose column the caller does not need. The directory checksum was checked
7302/// before this walk, and the fields still need their lengths and tags checked to find the next one.
7303fn skip_summary(cur: &mut Cursor<'_>, values: bool, rows: usize) -> Result<()> {
7304    match cur.u8()? {
7305        0 => Ok(()),
7306        1 => {
7307            cur.skip(8)?;
7308            let entries = cur.u32()? as usize;
7309            if entries > FREQUENCY_ENTRIES {
7310                return Err(invalid("frequency entry count exceeds its bound"));
7311            }
7312            for _ in 0..entries {
7313                match cur.u8()? {
7314                    0 => {}
7315                    1 => cur.skip(16)?,
7316                    2 => cur.skip(4)?,
7317                    _ => return Err(invalid("frequency value tag differs")),
7318                }
7319                cur.skip(8)?;
7320            }
7321            let ordinals = cur.u32()? as usize;
7322            if ordinals > FREQUENCY_ORDINALS || ordinals > rows {
7323                return Err(invalid("frequency ordinal count exceeds its bound"));
7324            }
7325            for _ in 0..ordinals {
7326                cur.var_u64()?;
7327            }
7328            if values {
7329                cur.skip(ordinals * 2)?;
7330            }
7331            Ok(())
7332        }
7333        _ => Err(invalid("frequency summary tag differs")),
7334    }
7335}
7336
7337/// Reads only the catalog, stripe null counts, and one frequency synopsis. This is the cold path
7338/// for a summary-backed count; constructing every page descriptor and zone map would make it cost
7339/// the size of the table directory even when no row is read.
7340fn quick_nonzero(
7341    mut cur: Cursor<'_>,
7342    name: &str,
7343    fields: &[Field],
7344    rows: usize,
7345    wanted: usize,
7346) -> Result<Option<u64>> {
7347    if cur.take(8)? != DIRECTORY || cur.text()? != name {
7348        return Err(invalid("table directory differs from the catalog"));
7349    }
7350    let width = cur.u16()? as usize;
7351    if width != fields.len() {
7352        return Err(invalid("table directory width differs from the catalog"));
7353    }
7354    for field in fields {
7355        let stored =
7356            Field { name: cur.text()?, ty: read_type(&mut cur)?, not_null: cur.u8()? != 0 };
7357        if &stored != field {
7358            return Err(invalid("table directory schema differs from the catalog"));
7359        }
7360    }
7361    let mut dictionaries = Vec::with_capacity(width);
7362    for _ in 0..width {
7363        let held = match cur.u8()? {
7364            0 => false,
7365            1 => {
7366                cur.skip(20)?;
7367                true
7368            }
7369            _ => return Err(invalid("dictionary page tag differs")),
7370        };
7371        dictionaries.push(held);
7372    }
7373    for _ in 0..width {
7374        match cur.u8()? {
7375            0 => {}
7376            1 => cur.skip(8)?,
7377            _ => return Err(invalid("distinct count tag differs")),
7378        }
7379    }
7380    if cur.u64()? != rows as u64 {
7381        return Err(invalid("table row count differs from the catalog"));
7382    }
7383    let stripes = cur.u32()? as usize;
7384    let mut total = 0_usize;
7385    let mut nulls = 0_u64;
7386    for _ in 0..stripes {
7387        let parts = cur.u32()? as usize;
7388        if parts == 0 || parts > STRIPE_PARTS {
7389            return Err(invalid("stripe part count is outside its bound"));
7390        }
7391        let mut stripe_rows = 0_usize;
7392        for _ in 0..parts {
7393            stripe_rows = stripe_rows
7394                .checked_add(cur.u32()? as usize)
7395                .ok_or_else(|| invalid("stripe row count overflow"))?;
7396        }
7397        total =
7398            total.checked_add(stripe_rows).ok_or_else(|| invalid("stripe row count overflow"))?;
7399        cur.skip(12 + width * 12)?;
7400        for (field, held) in fields.iter().zip(&dictionaries) {
7401            if field.ty == LogicalType::Varchar && *held {
7402                cur.skip(20)?;
7403            }
7404        }
7405        for _ in 0..width * 2 {
7406            match cur.u8()? {
7407                0 => {}
7408                1 => cur.skip(20)?,
7409                _ => return Err(invalid("stripe page tag differs")),
7410            }
7411        }
7412        for column in 0..width {
7413            cur.skip_bound()?;
7414            cur.skip_bound()?;
7415            let count = cur.u32()? as u64;
7416            if count > stripe_rows as u64 {
7417                return Err(invalid("null count exceeds stripe rows"));
7418            }
7419            if column == wanted {
7420                nulls = nulls.checked_add(count).ok_or_else(|| invalid("null count overflow"))?;
7421            }
7422            cur.skip(1)?;
7423            match cur.u8()? {
7424                0 => {}
7425                1 => cur.skip(16)?,
7426                _ => return Err(invalid("a stripe sum has an unknown tag")),
7427            }
7428        }
7429    }
7430    if total != rows {
7431        return Err(invalid("table row count differs from stripes"));
7432    }
7433    if cur.done() {
7434        return Ok(None);
7435    }
7436    let magic = cur.take(8)?;
7437    let values = magic == FREQUENCIES;
7438    if !values && magic != FREQUENCIES_V2 {
7439        return Err(invalid("directory extension magic differs"));
7440    }
7441    if cur.u16()? as usize != width {
7442        return Err(invalid("frequency column count differs"));
7443    }
7444    for _ in 0..wanted {
7445        skip_summary(&mut cur, values, rows)?;
7446    }
7447    let Some(summary) = decode_summary(&mut cur, &fields[wanted], rows, values)? else {
7448        return Ok(None);
7449    };
7450    let zero = summary
7451        .entries
7452        .iter()
7453        .find(|entry| entry.value == FrequencyValue::Integer(0))
7454        .map(|entry| entry.count)
7455        .or_else(|| (summary.omitted_max == 0).then_some(0));
7456    Ok(zero.and_then(|zero| (rows as u64).checked_sub(nulls)?.checked_sub(zero)))
7457}
7458
7459fn decode_directory(bytes: &[u8], size: u64) -> Result<Table> {
7460    read_directory(Cursor::new(bytes), size, None)
7461}
7462
7463/// A directory out of `cur`, which is a whole one in memory or one being read out of the file.
7464///
7465/// `stored_at` is where the directory starts in the file when it is being read out of it, and then
7466/// every frequency synopsis is checked and left there, as [`Frequencies::Stored`].
7467fn read_directory(mut cur: Cursor<'_>, size: u64, stored_at: Option<u64>) -> Result<Table> {
7468    if cur.take(8)? != DIRECTORY {
7469        return Err(invalid("directory magic differs"));
7470    }
7471    let name = cur.text()?;
7472    let width = cur.u16()? as usize;
7473    let mut fields = Vec::with_capacity(width);
7474    for _ in 0..width {
7475        let name = cur.text()?;
7476        let ty = read_type(&mut cur)?;
7477        let not_null = match cur.u8()? {
7478            0 => false,
7479            1 => true,
7480            _ => return Err(invalid("nullability flag differs")),
7481        };
7482        fields.push(Field { name, ty, not_null });
7483    }
7484    let mut dictionaries = Vec::with_capacity(width);
7485    for _ in 0..width {
7486        dictionaries.push(match cur.u8()? {
7487            0 => None,
7488            1 => {
7489                let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
7490                let end = page
7491                    .offset
7492                    .checked_add(u64::from(page.length))
7493                    .ok_or_else(|| invalid("dictionary page offset overflow"))?;
7494                // A global dictionary covers a whole column, not one bounded stripe. Its lazy
7495                // payload is intentionally allowed to grow past `MAX_PAGE`; only ordinary column
7496                // pages are capped there. `Writer::finish` has already bounded this length by the
7497                // on-disk `u32`, and the range check below keeps it inside the file.
7498                if page.offset < HEADER || end > size {
7499                    return Err(invalid("dictionary page range is outside the file"));
7500                }
7501                Some(page)
7502            }
7503            _ => return Err(invalid("dictionary page tag differs")),
7504        });
7505    }
7506    let mut distincts = Vec::with_capacity(width);
7507    for _ in 0..width {
7508        distincts.push(match cur.u8()? {
7509            0 => None,
7510            1 => Some(cur.u64()?),
7511            _ => return Err(invalid("distinct count tag differs")),
7512        });
7513    }
7514    let rows = usize::try_from(cur.u64()?).map_err(|_| invalid("row count does not fit"))?;
7515    let count = cur.u32()? as usize;
7516    let mut stripes = Vec::with_capacity(count);
7517    let mut total = 0_usize;
7518    for _ in 0..count {
7519        let count = cur.u32()? as usize;
7520        if count == 0 || count > STRIPE_PARTS {
7521            return Err(invalid("stripe part count is outside its bound"));
7522        }
7523        let mut parts = Vec::with_capacity(count);
7524        let mut stripe_rows = 0_usize;
7525        for _ in 0..count {
7526            let rows = cur.u32()?;
7527            if rows == 0 {
7528                return Err(invalid("empty part"));
7529            }
7530            parts.push(rows);
7531            stripe_rows = stripe_rows
7532                .checked_add(rows as usize)
7533                .ok_or_else(|| invalid("stripe row count overflow"))?;
7534        }
7535        total =
7536            total.checked_add(stripe_rows).ok_or_else(|| invalid("stripe row count overflow"))?;
7537        let index = Span { offset: cur.u64()?, length: cur.u32()? };
7538        let section = index_section(count)?;
7539        let wanted = section
7540            .checked_mul(width)
7541            .and_then(|bytes| u32::try_from(bytes).ok())
7542            .ok_or_else(|| invalid("index page length overflow"))?;
7543        let end = index
7544            .offset
7545            .checked_add(u64::from(index.length))
7546            .ok_or_else(|| invalid("index page offset overflow"))?;
7547        if index.offset < HEADER || end > size || index.length != wanted {
7548            return Err(invalid("index page range is outside the file"));
7549        }
7550        let mut pages = Vec::with_capacity(width);
7551        for _ in 0..width {
7552            let offset = cur.u64()?;
7553            let length = cur.u32()?;
7554            let end = offset
7555                .checked_add(u64::from(length))
7556                .ok_or_else(|| invalid("page offset overflow"))?;
7557            if offset < HEADER || end > size || length as usize > MAX_PAGE {
7558                return Err(invalid("page range is outside the file"));
7559            }
7560            pages.push(Span { offset, length });
7561        }
7562        let mut memberships = vec![None; width];
7563        for (column, field) in fields.iter().enumerate() {
7564            if field.ty != LogicalType::Varchar || dictionaries[column].is_none() {
7565                continue;
7566            }
7567            let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
7568            let end = page
7569                .offset
7570                .checked_add(u64::from(page.length))
7571                .ok_or_else(|| invalid("membership page offset overflow"))?;
7572            if page.offset < HEADER || end > size || page.length as usize > MAX_PAGE {
7573                return Err(invalid("membership page range is outside the file"));
7574            }
7575            memberships[column] = Some(page);
7576        }
7577        let mut sieves = vec![None; width];
7578        for sieve in sieves.iter_mut().take(width) {
7579            match cur.u8()? {
7580                0 => continue,
7581                1 => {}
7582                _ => return Err(invalid("a sieve page has an unknown tag")),
7583            }
7584            let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
7585            let end = page
7586                .offset
7587                .checked_add(u64::from(page.length))
7588                .ok_or_else(|| invalid("sieve page offset overflow"))?;
7589            if page.offset < HEADER || end > size || page.length as usize > MAX_PAGE {
7590                return Err(invalid("sieve page range is outside the file"));
7591            }
7592            *sieve = Some(page);
7593        }
7594        let mut part_ranges = vec![None; width];
7595        for held in part_ranges.iter_mut().take(width) {
7596            match cur.u8()? {
7597                0 => continue,
7598                1 => {}
7599                _ => return Err(invalid("a part range page has an unknown tag")),
7600            }
7601            let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
7602            let end = page
7603                .offset
7604                .checked_add(u64::from(page.length))
7605                .ok_or_else(|| invalid("part range page offset overflow"))?;
7606            if page.offset < HEADER || end > size || page.length as usize > MAX_PAGE {
7607                return Err(invalid("part range page range is outside the file"));
7608            }
7609            *held = Some(page);
7610        }
7611        let mut ranges = Vec::with_capacity(width);
7612        for column in 0..width {
7613            let low = cur.bound()?;
7614            let high = cur.bound()?;
7615            let nulls = cur.u32()? as usize;
7616            if nulls > stripe_rows {
7617                return Err(invalid("null count exceeds stripe rows"));
7618            }
7619            let exact = cur.u8()? != 0;
7620            let sum = match cur.u8()? {
7621                0 => None,
7622                1 => Some(i128::from_le_bytes(
7623                    cur.take(16)?.try_into().map_err(|_| invalid("a stripe sum is truncated"))?,
7624                )),
7625                _ => return Err(invalid("a stripe sum has an unknown tag")),
7626            };
7627            // Files written before the ends of a decimal or a timestamp column carried their power
7628            // of ten hold a bare integer here, and that integer is the one the column holds, which
7629            // is what the power is over. So the type puts it back on the way in and an old file
7630            // prunes as well as a new one. A file that already wrote the power keeps it, because
7631            // this leaves anything that is not an integer alone.
7632            let ty = &fields.get(column).ok_or_else(|| invalid("a stripe range has no column"))?.ty;
7633            let low = low.map(|bound| scaled_as(bound, ty));
7634            let high = high.map(|bound| scaled_as(bound, ty));
7635            ranges.push(Range { low, high, nulls, exact, sum });
7636        }
7637        stripes.push(Stripe {
7638            rows: stripe_rows,
7639            parts,
7640            index,
7641            pages,
7642            memberships: Pages::from_slots(memberships)?,
7643            sieves: Pages::from_slots(sieves)?,
7644            part_ranges: Pages::from_slots(part_ranges)?,
7645            zone: Zone::from_ranges(ranges),
7646        });
7647    }
7648    if total != rows {
7649        return Err(invalid("table row count differs from stripes"));
7650    }
7651    // How many entries each column's synopsis lists, which is all a pair summary is checked against,
7652    // kept apart because the synopses themselves may be left in the file.
7653    let mut entry_counts = vec![0; width];
7654    let frequencies = if cur.done() {
7655        vec![None; width]
7656    } else {
7657        let frequency_magic = cur.take(8)?;
7658        let frequency_values = frequency_magic == FREQUENCIES;
7659        if !frequency_values && frequency_magic != FREQUENCIES_V2 {
7660            return Err(invalid("directory extension magic differs"));
7661        }
7662        if cur.u16()? as usize != width {
7663            return Err(invalid("frequency column count differs"));
7664        }
7665        let mut frequencies = Vec::with_capacity(width);
7666        for (field, entry_count) in fields.iter().zip(&mut entry_counts) {
7667            let start = cur.at;
7668            let summary = decode_summary(&mut cur, field, rows, frequency_values)?;
7669            *entry_count = summary.as_ref().map_or(0, |summary| summary.entries.len());
7670            frequencies.push(match (summary, stored_at) {
7671                (None, _) => None,
7672                (Some(summary), None) => Some(Frequencies::Held(summary)),
7673                (Some(_), Some(offset)) => Some(Frequencies::Stored {
7674                    span: Span {
7675                        offset: offset + start as u64,
7676                        length: u32::try_from(cur.at - start)
7677                            .map_err(|_| invalid("a frequency synopsis is too long"))?,
7678                    },
7679                    values: frequency_values,
7680                }),
7681            });
7682        }
7683        frequencies
7684    };
7685    // There are two optional trailing blocks now rather than one, so the reader dispatches on the
7686    // magic it finds rather than on where the bytes ran out. That is what lets the two arrive
7687    // independently: a format 22 directory ends here and has neither, a directory written before
7688    // the section table has only the clustering declaration, and each one still opens without a
7689    // rewrite. It is the G1 exit criterion, which is that a reader that knows about sections opens
7690    // a file that predates them and answers every query, only without the graph path.
7691    //
7692    // A repeated block is refused rather than allowed to win, because two clustering declarations
7693    // in one directory is a torn directory and the only question is which of them is the lie.
7694    let mut clustering = None;
7695    let mut sections = Vec::new();
7696    let mut pair_frequencies = Vec::new();
7697    let mut seen_pair_frequencies = false;
7698    let mut frequency_texts = vec![Vec::new(); width];
7699    let mut seen_frequency_texts = false;
7700    let mut host_groups = None;
7701    let mut seen_sections = false;
7702    let mut dictionary_payloads = Vec::new();
7703    let mut seen_payloads = false;
7704    // Zero until a section table says otherwise, which is what a format 22 table gets and what
7705    // makes every section stamp fail to match on one, because real generations start at one.
7706    let mut generation = 0;
7707    while !cur.done() {
7708        let mut tag = [0u8; 8];
7709        tag.copy_from_slice(cur.take(8)?);
7710        if &tag == PAIR_FREQUENCIES {
7711            if seen_pair_frequencies {
7712                return Err(invalid("directory names two pair frequency blocks"));
7713            }
7714            seen_pair_frequencies = true;
7715            let count = cur.u16()? as usize;
7716            if count > MAX_PAIR_FREQUENCIES {
7717                return Err(invalid("pair frequency count exceeds its bound"));
7718            }
7719            pair_frequencies = Vec::with_capacity(count);
7720            for _ in 0..count {
7721                let first = cur.u16()?;
7722                let second = cur.u16()?;
7723                let first_at = first as usize;
7724                let second_at = second as usize;
7725                if frequencies.get(first_at).and_then(Option::as_ref).is_none() {
7726                    return Err(invalid("pair frequency first column has no synopsis"));
7727                }
7728                let first_entries = entry_counts[first_at];
7729                if !matches!(fields.get(second_at), Some(field) if field.ty == LogicalType::Varchar)
7730                    || dictionaries.get(second_at).copied().flatten().is_none()
7731                {
7732                    return Err(invalid("pair frequency second column has no stable dictionary"));
7733                }
7734                if pair_frequencies
7735                    .iter()
7736                    .any(|held: &PairFrequencySummary| held.first == first && held.second == second)
7737                {
7738                    return Err(invalid("directory repeats a pair frequency summary"));
7739                }
7740                let omitted_max = cur.u64()?;
7741                if omitted_max > rows as u64 {
7742                    return Err(invalid("pair frequency omitted count exceeds the table"));
7743                }
7744                let entries_count = cur.u16()? as usize;
7745                if entries_count > FREQUENCY_ENTRIES {
7746                    return Err(invalid("pair frequency entry count exceeds its bound"));
7747                }
7748                let mut entries = Vec::with_capacity(entries_count);
7749                for _ in 0..entries_count {
7750                    let first_entry = cur.u16()?;
7751                    if first_entry as usize >= first_entries {
7752                        return Err(invalid("pair frequency anchor is outside its synopsis"));
7753                    }
7754                    let second = match cur.u8()? {
7755                        0 => None,
7756                        1 => Some(cur.u32()?),
7757                        _ => return Err(invalid("pair frequency string tag differs")),
7758                    };
7759                    let count = cur.u64()?;
7760                    if count == 0 || count > rows as u64 {
7761                        return Err(invalid("pair frequency count is outside the table"));
7762                    }
7763                    entries.push(PairFrequencyEntry { first_entry, second, count });
7764                }
7765                if entries.windows(2).any(|pair| pair[0].count < pair[1].count) {
7766                    return Err(invalid("pair frequency entries are not descending"));
7767                }
7768                pair_frequencies.push(PairFrequencySummary { first, second, entries, omitted_max });
7769            }
7770        } else if &tag == FREQUENCY_TEXTS {
7771            if seen_frequency_texts {
7772                return Err(invalid("directory names two frequency text blocks"));
7773            }
7774            seen_frequency_texts = true;
7775            let columns = cur.u16()? as usize;
7776            if columns > width {
7777                return Err(invalid("frequency text column count exceeds the schema"));
7778            }
7779            for _ in 0..columns {
7780                let column = cur.u16()? as usize;
7781                if !frequency_texts.get(column).is_some_and(Vec::is_empty) {
7782                    return Err(invalid("frequency text column is repeated or out of range"));
7783                }
7784                if !matches!(fields.get(column), Some(field) if field.ty == LogicalType::Varchar)
7785                    || dictionaries.get(column).copied().flatten().is_none()
7786                    || frequencies.get(column).and_then(Option::as_ref).is_none()
7787                {
7788                    return Err(invalid("frequency texts belong to a non-string synopsis"));
7789                }
7790                let count = cur.u16()? as usize;
7791                if count == 0 || count != entry_counts[column] {
7792                    return Err(invalid("frequency text count differs from its synopsis"));
7793                }
7794                let mut texts = Vec::with_capacity(count);
7795                for _ in 0..count {
7796                    texts.push(match cur.u8()? {
7797                        0 => None,
7798                        1 => {
7799                            let length = cur.u32()? as usize;
7800                            let bytes = cur.take(length)?.to_vec();
7801                            std::str::from_utf8(&bytes)
7802                                .map_err(|_| invalid("frequency text is not UTF-8"))?;
7803                            Some(bytes)
7804                        }
7805                        _ => return Err(invalid("frequency text tag differs")),
7806                    });
7807                }
7808                frequency_texts[column] = texts;
7809            }
7810        } else if &tag == HOST_GROUPS {
7811            if host_groups.is_some() {
7812                return Err(invalid("directory names two host group blocks"));
7813            }
7814            let column = cur.u16()? as usize;
7815            if !matches!(fields.get(column), Some(field) if field.ty == LogicalType::Varchar)
7816                || dictionaries.get(column).copied().flatten().is_none()
7817            {
7818                return Err(invalid("host groups belong to a non-string dictionary"));
7819            }
7820            let omitted_max = cur.u64()?;
7821            if omitted_max > rows as u64 {
7822                return Err(invalid("host group bound exceeds the table"));
7823            }
7824            let count = cur.u16()? as usize;
7825            if count > host::CAPACITY {
7826                return Err(invalid("host group count exceeds its bound"));
7827            }
7828            let mut entries = Vec::with_capacity(count);
7829            let mut bytes = 0_usize;
7830            for _ in 0..count {
7831                let host_len = cur.u32()? as usize;
7832                bytes =
7833                    bytes.checked_add(host_len).ok_or_else(|| invalid("host bytes overflow"))?;
7834                if bytes > host::BYTE_BUDGET {
7835                    return Err(invalid("host groups exceed their byte budget"));
7836                }
7837                let host = std::str::from_utf8(cur.take(host_len)?)
7838                    .map_err(|_| invalid("host is not UTF-8"))?
7839                    .to_owned();
7840                let count = cur.u64()?;
7841                if count == 0 || count > rows as u64 {
7842                    return Err(invalid("host group count exceeds the table"));
7843                }
7844                let bytes_sum = i128::from_le_bytes(
7845                    cur.take(16)?
7846                        .try_into()
7847                        .map_err(|_| invalid("host length sum is truncated"))?,
7848                );
7849                if bytes_sum < 0 {
7850                    return Err(invalid("host length sum is negative"));
7851                }
7852                let minimum_len = cur.u32()? as usize;
7853                bytes =
7854                    bytes.checked_add(minimum_len).ok_or_else(|| invalid("host bytes overflow"))?;
7855                if bytes > host::BYTE_BUDGET {
7856                    return Err(invalid("host groups exceed their byte budget"));
7857                }
7858                let minimum = std::str::from_utf8(cur.take(minimum_len)?)
7859                    .map_err(|_| invalid("host minimum is not UTF-8"))?
7860                    .to_owned();
7861                entries.push(host::HostEntry { host, count, bytes_sum, minimum });
7862            }
7863            if entries.windows(2).any(|pair| pair[0].count < pair[1].count)
7864                || entries.iter().any(|entry| entry.host.is_empty() || entry.minimum.is_empty())
7865            {
7866                return Err(invalid("host groups are not in certified order"));
7867            }
7868            host_groups = Some(host::HostSummary { column, omitted_max, entries });
7869        } else if &tag == CLUSTERING {
7870            if clustering.is_some() {
7871                return Err(invalid("directory names two clustering declarations"));
7872            }
7873            let bucket = Width::from_tag(cur.u8()?)
7874                .ok_or_else(|| invalid("clustering width tag differs"))?;
7875            let count = cur.u16()? as usize;
7876            let mut columns = Vec::with_capacity(count.min(fields.len()));
7877            for _ in 0..count {
7878                columns.push(u32::from(cur.u16()?));
7879            }
7880            // Through the constructor and not built by hand, so that a file claiming a column the
7881            // table does not have is caught at open rather than at the first scan that trusted it.
7882            clustering = Some(Clustering::new(columns, bucket, &fields).map_err(|_| {
7883                invalid("stored clustering declaration does not match the table it is on")
7884            })?);
7885        } else if &tag == SECTIONS {
7886            if seen_sections {
7887                return Err(invalid("directory names two section tables"));
7888            }
7889            seen_sections = true;
7890            generation = cur.u64()?;
7891            let count = cur.u16()? as usize;
7892            if count > MAX_SECTIONS {
7893                return Err(invalid("section count exceeds its bound"));
7894            }
7895            sections = Vec::with_capacity(count);
7896            // entry at a time: a malformed section entry is refused rather than turned into an
7897            // offset.
7898            for _ in 0..count {
7899                sections.push(Section::decode(cur.take(section::ENTRY_BYTES)?)?);
7900            }
7901            for held in &sections {
7902                let Some(end) = held.extent_page.checked_add(u64::from(held.extent_bytes)) else {
7903                    return Err(invalid("a section's extent table overflows the file"));
7904                };
7905                // The bound check is here and not in `section`, because only the caller knows how
7906                // big the file is. A section pointing past the end is a torn directory, and reading
7907                // the payload it names would be reading whatever else is at that offset.
7908                if held.extent_bytes != 0 && (held.extent_page < HEADER || end > size) {
7909                    return Err(invalid("a section's extent table is outside the file"));
7910                }
7911                if held.extents == 0 && held.extent_bytes != 0 {
7912                    return Err(invalid("a section with no extents names an extent table"));
7913                }
7914            }
7915        } else if &tag == DICTIONARY_PAYLOADS {
7916            if seen_payloads {
7917                return Err(invalid("directory names two dictionary payload blocks"));
7918            }
7919            seen_payloads = true;
7920            let count = cur.u16()? as usize;
7921            if count != fields.len() {
7922                return Err(invalid("dictionary payload block does not match the table's columns"));
7923            }
7924            dictionary_payloads = Vec::with_capacity(count);
7925            for _ in 0..count {
7926                let bytes = cur.u64()?;
7927                if bytes > size {
7928                    return Err(invalid("a dictionary payload is larger than the file"));
7929                }
7930                dictionary_payloads.push(bytes);
7931            }
7932        } else {
7933            return Err(invalid("directory extension magic differs"));
7934        }
7935    }
7936    if !cur.done() {
7937        return Err(invalid("directory has trailing bytes"));
7938    }
7939    Ok(Table {
7940        name,
7941        fields,
7942        stripes,
7943        rows,
7944        dictionaries,
7945        dictionary_payloads,
7946        distincts,
7947        frequencies,
7948        pair_frequencies,
7949        frequency_texts,
7950        host_groups,
7951        clustering,
7952        generation,
7953        sections,
7954    })
7955}
7956
7957/// A zone map's end, in the layout `rudb_common::bounds` defines. See [`Cursor::bound`].
7958fn put_bound(out: &mut Vec<u8>, bound: Option<&Bound>) -> Result<()> {
7959    bounds::put(out, bound)
7960}
7961
7962/// Which cascades are worth trying on a run of dictionary codes.
7963///
7964/// The exhaustive chooser encodes every candidate at every level of a cascade three deep and keeps
7965/// the smallest, which on a part of 1024 codes is around a hundred full encodes to decide something
7966/// three candidates were always going to win. It is the right default for a crate that does not
7967/// know what it is looking at. Here we do know. Codes are counted from zero in the order the values
7968/// were first seen, so a part of them is one value, or a narrow band, or a few long runs, and those
7969/// are constant, frame of reference and run length. Nothing else has ever come first on this data.
7970///
7971/// A dictionary of dictionary codes is the one candidate that can never pay, because the codes are
7972/// already the dictionary, and it is also the most expensive one to try. Below the top level the
7973/// streams are an RLE's run values and run lengths, which are integers in their own right with no
7974/// runs left in them, so only the two flat candidates go down there.
7975///
7976/// This is size given up for time on purpose, and the ablation is this chooser against
7977/// [`chooser::EXHAUSTIVE`] on the same file.
7978#[derive(Debug)]
7979struct Codes;
7980
7981impl chooser::Chooser for Codes {
7982    fn name(&self) -> &'static str {
7983        "codes"
7984    }
7985
7986    fn narrow_strings(
7987        &self,
7988        _values: &[&[u8]],
7989        offered: &[string::Kind],
7990        _depth: u8,
7991    ) -> Vec<string::Kind> {
7992        // Never reached, because nothing here encodes strings through the cascade. The trait asks
7993        // for it and the honest answer to a question we have no opinion on is the whole list.
7994        offered.to_vec()
7995    }
7996
7997    fn narrow_integers(
7998        &self,
7999        _values: &[i64],
8000        offered: &[integer::Kind],
8001        depth: u8,
8002    ) -> Vec<integer::Kind> {
8003        // The contract is a non empty subset, and a chunk that offers none of the three is a chunk
8004        // this has no opinion about rather than one that cannot be written.
8005        narrowed_to(Codes::keep(depth), offered)
8006    }
8007
8008    fn considers_integer(&self, kind: integer::Kind, depth: u8) -> bool {
8009        Codes::keep(depth).contains(&kind)
8010    }
8011}
8012
8013impl Codes {
8014    fn keep(depth: u8) -> &'static [integer::Kind] {
8015        if depth == 0 {
8016            &[integer::Kind::Constant, integer::Kind::Packed, integer::Kind::Rle]
8017        } else {
8018            &[integer::Kind::Constant, integer::Kind::Packed]
8019        }
8020    }
8021}
8022
8023/// The kinds of `offered` that are in `keep`, or all of `offered` when none of them are.
8024///
8025/// `Packed` applies to every chunk and both choosers keep it, so the fallback is never taken on a
8026/// chunk the cascade offers. It is there because the contract is a non empty subset and a chooser
8027/// that returned nothing would be a chunk that cannot be written. It is also why saying no to a kind
8028/// in `considers_integer` is safe: a kind that is never offered could only have been kept through
8029/// this fallback, and the fallback is never reached.
8030fn narrowed_to(keep: &[integer::Kind], offered: &[integer::Kind]) -> Vec<integer::Kind> {
8031    let narrowed: Vec<integer::Kind> =
8032        offered.iter().copied().filter(|kind| keep.contains(kind)).collect();
8033    if narrowed.is_empty() { offered.to_vec() } else { narrowed }
8034}
8035
8036/// Which cascades are worth trying on a part of plain integers.
8037///
8038/// Wider than [`Codes`] because the values are not codes and carry whatever shape the column has.
8039/// A timestamp column climbs, so delta is the one that matters and is the reason this exists at
8040/// all: three timestamp columns in ClickBench were coming out at exactly eight bytes a row with
8041/// nothing asked of them. The same three columns are why the stride is here, since a timestamp
8042/// loaded from a source that recorded whole seconds is microseconds with twenty zero bits under
8043/// every value. A column that is one value with a handful of exceptions is sparse. What is still
8044/// left out is the dictionary, for the same reason as in [`Codes`]: it is the most
8045/// expensive candidate to try and this file already puts the columns that want one through a
8046/// dictionary of their own before they ever reach here.
8047#[derive(Debug)]
8048struct Fixed;
8049
8050impl chooser::Chooser for Fixed {
8051    fn name(&self) -> &'static str {
8052        "fixed"
8053    }
8054
8055    fn narrow_strings(
8056        &self,
8057        _values: &[&[u8]],
8058        offered: &[string::Kind],
8059        _depth: u8,
8060    ) -> Vec<string::Kind> {
8061        offered.to_vec()
8062    }
8063
8064    fn narrow_integers(
8065        &self,
8066        _values: &[i64],
8067        offered: &[integer::Kind],
8068        depth: u8,
8069    ) -> Vec<integer::Kind> {
8070        narrowed_to(Fixed::keep(depth), offered)
8071    }
8072
8073    fn considers_integer(&self, kind: integer::Kind, depth: u8) -> bool {
8074        Fixed::keep(depth).contains(&kind)
8075    }
8076}
8077
8078impl Fixed {
8079    fn keep(depth: u8) -> &'static [integer::Kind] {
8080        if depth == 0 {
8081            &[
8082                integer::Kind::Constant,
8083                integer::Kind::Packed,
8084                integer::Kind::Delta,
8085                integer::Kind::Rle,
8086                integer::Kind::Sparse,
8087                integer::Kind::Strided,
8088            ]
8089        } else {
8090            &[integer::Kind::Constant, integer::Kind::Packed, integer::Kind::Delta]
8091        }
8092    }
8093}
8094
8095/// Every value of an integer part as an `i64`, or `None` for a part this cannot widen without
8096/// losing one.
8097///
8098/// `UBIGINT` is the only integer type left out, because half its range does not fit and a page that
8099/// silently wrapped would be worse than a page that stays plain. Booleans and strings are not
8100/// integers and have their own ways of being small.
8101fn widened(data: &Data) -> Option<Vec<i64>> {
8102    match data {
8103        Data::Int8(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
8104        Data::UInt8(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
8105        Data::Int16(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
8106        Data::UInt16(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
8107        Data::Int32(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
8108        Data::UInt32(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
8109        Data::Int64(values) => Some(values.to_vec()),
8110        _ => None,
8111    }
8112}
8113
8114/// An integer type a cascaded page can be read back into, and how to tell whether a value fits.
8115///
8116/// This exists so that the check and the conversion can be two loops instead of one. `TryFrom` puts
8117/// them together, which is the right shape for one value and the wrong one for a page: a fallible
8118/// conversion a value at a time is a branch a value at a time, the branch decides whether the loop
8119/// keeps going, and a loop like that is one no compiler will widen.
8120trait Narrow: Copy {
8121    /// How wide this type is, and what to add to a value to put its range at the bottom of a `u64`.
8122    ///
8123    /// Half the width for a signed type, which is what moves its smallest value to zero, and nothing
8124    /// for an unsigned one, whose smallest value is already there.
8125    const BIASED: (u32, u64);
8126
8127    /// The value narrowed, which the caller has already shown fits.
8128    fn narrow(value: i64) -> Self;
8129}
8130
8131/// The bits of `value` a `T` cannot hold, and zero when the value fits.
8132///
8133/// The question is asked this way round because the answers or together. A page fits when every
8134/// residue in it is zero, so the loop is an or into an accumulator and the decision is one test
8135/// after it, where asking whether each value is between a floor and a ceiling gives an answer that
8136/// does not combine and turns into a running minimum and maximum.
8137///
8138/// Biasing and shifting is what the answer is made of, rather than anything that reads more like the
8139/// question, because those are the operations a machine has four of. A 64 bit integer minimum is
8140/// AVX-512. So is a 64 bit arithmetic shift right, which is how the sign extension this could be
8141/// written as would have to be done. An add and a logical shift right are AVX2 and are on every
8142/// machine this runs on, so this is the form that gets four values a cycle instead of one.
8143///
8144/// Adding the bias moves the type's range to `0..=2^bits`, wrapping, so everything in range shifts
8145/// away to nothing and everything outside it leaves something behind. A negative value under an
8146/// unsigned type is caught by the same shift, because a negative `i64` read as a `u64` is enormous.
8147#[allow(clippy::cast_sign_loss, reason = "a residue is a bit pattern and not a number")]
8148fn residue<T: Narrow>(value: i64) -> u64 {
8149    let (bits, bias) = T::BIASED;
8150    (value as u64).wrapping_add(bias) >> bits
8151}
8152
8153/// Says a primitive integer narrows with `as`, and where the bottom of its range is.
8154///
8155/// `as` is a truncation and is the right operation here only because [`fit`] has already found every
8156/// residue zero, and it is what makes the second loop a narrowing store with no branch in it.
8157macro_rules! narrows {
8158    ($($ty:ty => $bias:expr),* $(,)?) => {$(
8159        impl Narrow for $ty {
8160            const BIASED: (u32, u64) = (<$ty>::BITS, $bias);
8161
8162            #[allow(
8163                clippy::cast_possible_truncation,
8164                clippy::cast_sign_loss,
8165                reason = "the caller has checked the bits this truncates away"
8166            )]
8167            fn narrow(value: i64) -> Self {
8168                value as Self
8169            }
8170        }
8171    )*};
8172}
8173
8174narrows! {
8175    i8 => 1 << 7,
8176    u8 => 0,
8177    i16 => 1 << 15,
8178    u16 => 0,
8179    i32 => 1 << 31,
8180    u32 => 0,
8181}
8182
8183/// Narrows a page's values, refusing the page if any of them does not fit.
8184///
8185/// The check first and the conversion second, rather than a fallible conversion a value at a time.
8186/// Both loops here are ones a compiler widens: [`residue`] is three instructions a lane and a
8187/// narrowing store is one. The version before this was a `TryFrom` and a `collect` into a `Result`,
8188/// which is a compare, a branch and a short circuit a value at a time, and on ClickBench 39 it was
8189/// seven percent of the query. The version after that kept a running minimum and maximum, which is
8190/// the obvious way to ask and needs a 64 bit integer minimum that AVX2 does not have, so it stayed
8191/// a value at a time and was still ten percent of the same query.
8192///
8193/// An empty page has nothing to refuse, which falls out of the accumulator starting at zero rather
8194/// than needing a case of its own.
8195fn fit<T: Narrow>(values: &[i64]) -> Result<Vec<T>> {
8196    let mut spilled = 0u64;
8197    for value in values {
8198        spilled |= residue::<T>(*value);
8199    }
8200    if spilled != 0 {
8201        return Err(invalid("page value is not of its type"));
8202    }
8203    Ok(values.iter().map(|value| T::narrow(*value)).collect())
8204}
8205
8206/// The same values back in the width the column is declared at.
8207///
8208/// A value that does not fit is a page that disagrees with the directory about what the column is,
8209/// which is a damaged file rather than a caller error, so it is refused rather than truncated.
8210fn narrowed(ty: &LogicalType, values: Vec<i64>) -> Result<Data> {
8211    Ok(match ty {
8212        LogicalType::TinyInt => Data::Int8(fit::<i8>(&values)?.into()),
8213        LogicalType::UTinyInt => Data::UInt8(fit::<u8>(&values)?.into()),
8214        LogicalType::SmallInt => Data::Int16(fit::<i16>(&values)?.into()),
8215        LogicalType::USmallInt => Data::UInt16(fit::<u16>(&values)?.into()),
8216        LogicalType::Integer | LogicalType::Date => Data::Int32(fit::<i32>(&values)?.into()),
8217        LogicalType::UInteger => Data::UInt32(fit::<u32>(&values)?.into()),
8218        LogicalType::BigInt
8219        | LogicalType::Timestamp
8220        | LogicalType::Time
8221        | LogicalType::TimeTz
8222        | LogicalType::TimestampTz
8223        | LogicalType::TimestampS
8224        | LogicalType::TimestampMs
8225        | LogicalType::TimestampNs => Data::Int64(values.into()),
8226        // A decimal is an integer of unscaled units, so the cascade reads back into whichever
8227        // integer the declared width says the column is stored as.
8228        LogicalType::Decimal { .. } => match ty.physical() {
8229            PhysicalType::Int16 => Data::Int16(fit::<i16>(&values)?.into()),
8230            PhysicalType::Int32 => Data::Int32(fit::<i32>(&values)?.into()),
8231            PhysicalType::Int64 => Data::Int64(values.into()),
8232            _ => return Err(invalid("cascade codec belongs to a decimal that is not an integer")),
8233        },
8234        _ => return Err(invalid("cascade codec belongs to a page that is not integers")),
8235    })
8236}
8237
8238/// How many bytes a part of this type costs written out plainly, which is what the cascade has to
8239/// beat before it is worth the decode.
8240fn plain_width(ty: &LogicalType) -> Option<usize> {
8241    Some(match ty {
8242        LogicalType::TinyInt | LogicalType::UTinyInt => 1,
8243        LogicalType::SmallInt | LogicalType::USmallInt => 2,
8244        LogicalType::Integer | LogicalType::UInteger | LogicalType::Date => 4,
8245        LogicalType::BigInt
8246        | LogicalType::Timestamp
8247        | LogicalType::Time
8248        | LogicalType::TimeTz
8249        | LogicalType::TimestampTz
8250        | LogicalType::TimestampS
8251        | LogicalType::TimestampMs
8252        | LogicalType::TimestampNs => 8,
8253        LogicalType::Decimal { .. } => match ty.physical() {
8254            PhysicalType::Int16 => 2,
8255            PhysicalType::Int32 => 4,
8256            PhysicalType::Int64 => 8,
8257            // The widest decimals are stored as `i128`, which the cascade does not widen into, so
8258            // they take the plain path and there is nothing here to compare against.
8259            _ => return None,
8260        },
8261        _ => return None,
8262    })
8263}
8264
8265/// A part's plain integers through the cascade, or `None` when nothing it offers is worth it.
8266///
8267/// What it has to beat is whatever the page would otherwise have cost, which is the bit packed form
8268/// where there is one and the plain width where there is not. Both are cheaper to decode than a
8269/// cascade, so a tie goes to them.
8270fn cascaded(
8271    flat: &Vector,
8272    ty: &LogicalType,
8273    packed: Option<&Packed<'_>>,
8274    settling: &mut Settling,
8275) -> Result<Option<Vec<u8>>> {
8276    let (Some(width), Some(data)) = (plain_width(ty), flat.data()) else { return Ok(None) };
8277    let Some(values) = widened(data) else { return Ok(None) };
8278    let plain = values.len().saturating_mul(width);
8279    let best = match packed {
8280        // The tag, the base, the word count and the words, which is what the codec 2 branch writes.
8281        Some(packed) => plain.min(21 + size_of_val(packed.words())),
8282        None => plain,
8283    };
8284    let out = settling.encode(&values)?;
8285    Ok((out.len() < best).then_some(out))
8286}
8287
8288/// How often the parts of one column in one stripe search the cascade again, in parts.
8289///
8290/// A stripe is 64 parts, so this is four searches a stripe where there were 64. The search is
8291/// what the cascade costs: on ClickBench `hits` the integer cascade was about a tenth of the load's
8292/// CPU and nearly all of it under `encode_pages`, trying six trees on every part to keep the one
8293/// the part before had kept.
8294const SEARCH_EVERY: usize = 16;
8295
8296/// What the parts of one column in one stripe have settled on in the integer cascade.
8297///
8298/// One of these per column per stripe, used in part order, so what a part comes out as depends on
8299/// the stripe and not on which thread wrote it or on how many there were.
8300#[derive(Debug, Default)]
8301struct Settling {
8302    /// The shape of the last part that was searched, with what its top level offered, its length
8303    /// and its row count, which is the size a replay is held to.
8304    shape: Option<Shape>,
8305    /// Parts replayed since that search.
8306    since: usize,
8307}
8308
8309impl Settling {
8310    /// A part's integers through the cascade, replaying the settled shape where there is one.
8311    ///
8312    /// The replay is kept when it held and came out no more than a quarter bigger a row than the
8313    /// part the shape was searched on. Past that the column has changed under it and the part is
8314    /// searched. A replay that stopped fitting partway has already searched from where it stopped,
8315    /// so its shape is taken as the new one rather than searched a second time.
8316    fn encode(&mut self, values: &[i64]) -> Result<Vec<u8>> {
8317        if let Some(shape) = self.shape.as_ref().filter(|_| self.since < SEARCH_EVERY) {
8318            let replay = chooser::Replay::new(&shape.kinds, &Fixed).expecting(&shape.offered);
8319            let out = integer::encode_with(values, &replay)?;
8320            if !replay.held() {
8321                self.settle(&out, values.len(), replay.first_offered())?;
8322                return Ok(out);
8323            }
8324            let grown = (out.len() as u128) * (shape.rows as u128) * 4;
8325            if grown <= (shape.len as u128) * (values.len() as u128) * 5 {
8326                self.since += 1;
8327                return Ok(out);
8328            }
8329        }
8330        // A replay of nothing is the search, and says what the top level offered on the way.
8331        let search = chooser::Replay::new(&[], &Fixed);
8332        let out = integer::encode_with(values, &search)?;
8333        self.settle(&out, values.len(), search.first_offered())?;
8334        Ok(out)
8335    }
8336
8337    fn settle(&mut self, out: &[u8], rows: usize, offered: Vec<integer::Kind>) -> Result<()> {
8338        let kinds = integer::shape(out)?;
8339        self.shape = Some(Shape { kinds, offered, len: out.len().max(1), rows: rows.max(1) });
8340        self.since = 0;
8341        Ok(())
8342    }
8343}
8344
8345/// A searched part's cascade, what its top level was offered, and what it came to.
8346#[derive(Debug)]
8347struct Shape {
8348    kinds: Vec<integer::Kind>,
8349    offered: Vec<integer::Kind>,
8350    len: usize,
8351    rows: usize,
8352}
8353
8354/// A part's dictionary codes through the integer cascade, or `None` when the cascade did not pay.
8355///
8356/// Until now this stream was a `u32` a row with nothing asked of it, and on ClickBench that was
8357/// 400,185,326 bytes for every one of the 28 varchar columns, the same count for `URL` as for a
8358/// column holding the empty string in nearly every row. Codes are dense integers counted from zero
8359/// and a part holds 1024 of them, which is the shape frame of reference is best at, and a column
8360/// with one value everywhere comes back a constant costing nothing per row rather than four bytes.
8361///
8362/// The result is taken only when it is smaller than the plain form. A cascade is allowed to come
8363/// out larger on a part whose codes are genuinely wide, `URL` has about sixty million distinct
8364/// values, and there is no reason to pay for the decode when it does.
8365/// A varchar page as one FSST layer, or `None` when it did not pay.
8366///
8367/// Until now a varchar page that neither the global dictionary nor the per page dictionary claimed
8368/// was written out raw: four bytes of offset a row and then the bytes. That is the right answer for
8369/// a page of values with nothing in common and the wrong one for a page of English, and a column of
8370/// comments is the case this exists for.
8371///
8372/// One layer and not the full string cascade, which is what the payload blocks of a global
8373/// dictionary go through. The cascade is a search: it encodes the page under every candidate it has
8374/// and recurses into the integer cascade for the lengths of each one, and on TPC-H `orders` that
8375/// took the write from 6.9 s to 48.3 s. It reads back no faster than the dictionary it replaced
8376/// either, 1.807 G instructions against 1.810 G for `select o_comment from orders`, because
8377/// unpicking a nest of layers a value at a time costs what the dictionary's payload block decode
8378/// cost. Raw pages of the same column read in 0.686 G, which says the whole of the difference is
8379/// what the page has to be put back together from.
8380///
8381/// FSST alone keeps most of what the cascade found and gives all of that back. Decoding it is one
8382/// pass over the payload into one buffer, the values are laid end to end in it the way the raw form
8383/// already lays them out, and what the reader hands a chunk is views over that buffer.
8384///
8385/// The page dictionary gets first refusal because it is cheaper still, and it wins on a page whose
8386/// values repeat. What is left for this is the page whose values mostly do not, which is exactly the
8387/// page that was being written raw.
8388///
8389/// Taken only when it comes out smaller than the raw form, so a page of incompressible values pays
8390/// nothing at read time for having been offered.
8391fn text_compressed(flat: &Vector) -> Result<Option<Vec<u8>>> {
8392    let mut values: Vec<&[u8]> = Vec::with_capacity(flat.len());
8393    let mut payload = 0_usize;
8394    for row in 0..flat.len() {
8395        let text = flat.text_at(row).unwrap_or("").as_bytes();
8396        payload = payload.saturating_add(text.len());
8397        values.push(text);
8398    }
8399    // What codec 0 writes for a varchar page: an offset a row and one more, then the payload.
8400    let plain = (flat.len() + 1).saturating_mul(4).saturating_add(payload);
8401    let Some(out) = string::encode_only(string::Kind::Fsst, &values)? else {
8402        return Ok(None);
8403    };
8404    Ok((out.len() < plain).then_some(out))
8405}
8406
8407fn encoded_codes(codes: &[u32]) -> Result<Option<Vec<u8>>> {
8408    let wide: Vec<i64> = codes.iter().map(|code| i64::from(*code)).collect();
8409    let coded = integer::encode_with(&wide, &Codes)?;
8410    let plain = codes.len().saturating_mul(size_of::<u32>());
8411    Ok((coded.len() < plain).then_some(coded))
8412}
8413
8414/// The validity of a page, which is a flag and then, when some rows are null and some are not, a
8415/// bit a row with the valid ones set.
8416fn push_validity(out: &mut Vec<u8>, flat: &Vector) {
8417    let flag = match flat.validity() {
8418        Validity::AllValid => 0,
8419        Validity::AllInvalid => 1,
8420        Validity::Mask(_) => 2,
8421    };
8422    out.push(flag);
8423    if flag == 2 {
8424        for group in (0..flat.len()).step_by(8) {
8425            let mut bits = 0_u8;
8426            for bit in 0..8 {
8427                if group + bit < flat.len() && !flat.is_null_at(group + bit) {
8428                    bits |= 1 << bit;
8429                }
8430            }
8431            out.push(bits);
8432        }
8433    }
8434}
8435
8436/// One part of a column coded against its global dictionary as a page, from the codes and the
8437/// validity [`push_validity`] wrote for it.
8438///
8439/// The codes go through the integer cascade when that comes out smaller than four bytes a code,
8440/// which on a column that repeats itself it nearly always does, and are written as they are when it
8441/// does not.
8442fn coded_page(codes: &[u32], validity: &[u8]) -> Result<Vec<u8>> {
8443    let coded = encoded_codes(codes)?;
8444    let mut out = Vec::with_capacity(
8445        1 + validity.len() + coded.as_ref().map_or(size_of_val(codes), Vec::len),
8446    );
8447    out.push(if coded.is_some() { 4 } else { 3 });
8448    out.extend_from_slice(validity);
8449    match coded {
8450        Some(coded) => out.extend_from_slice(&coded),
8451        None => {
8452            for &code in codes {
8453                put_u32(&mut out, code);
8454            }
8455        }
8456    }
8457    Ok(out)
8458}
8459
8460/// One part of one column as a page, for every column that is not coded against a global
8461/// dictionary. Those are built by [`coded_page`] from codes [`prepare`] handed out.
8462fn encode(vector: &Vector, settling: &mut Settling) -> Result<Vec<u8>> {
8463    let ty = vector.logical_type();
8464    // flatten: the file writer needs a uniform scalar page and does it once per loaded chunk.
8465    let flat = vector.flatten()?;
8466    let mut out = Vec::new();
8467    let dictionary = if ty == &LogicalType::Varchar { string_dictionary(&flat)? } else { None };
8468    let compressed_text = if dictionary.is_none() && ty == &LogicalType::Varchar {
8469        text_compressed(&flat)?
8470    } else {
8471        None
8472    };
8473    let packed_vector = if dictionary.is_none() { Some(flat.bit_packed()?) } else { None };
8474    let packed = packed_vector.as_ref().and_then(Vector::packed_parts);
8475    // Only where nothing else has claimed the page, which is the plain integer case. A packed part
8476    // is still on the table because the cascade has to beat it too: the bit pack takes a part only
8477    // when it halves it, so a column that shrinks by a third was coming out whole.
8478    let cascade =
8479        if dictionary.is_none() { cascaded(&flat, ty, packed.as_ref(), settling)? } else { None };
8480    out.push(if cascade.is_some() {
8481        5
8482    } else if dictionary.is_some() {
8483        1
8484    } else if compressed_text.is_some() {
8485        6
8486    } else if packed.is_some() {
8487        2
8488    } else {
8489        0
8490    });
8491    push_validity(&mut out, &flat);
8492    if let Some(cascade) = cascade {
8493        out.extend_from_slice(&cascade);
8494        return Ok(out);
8495    }
8496    if let Some(dictionary) = dictionary {
8497        out.extend_from_slice(&dictionary);
8498        return Ok(out);
8499    }
8500    if let Some(compressed_text) = compressed_text {
8501        out.extend_from_slice(&compressed_text);
8502        return Ok(out);
8503    }
8504    if let Some(packed) = packed {
8505        if packed.offset() != 0 {
8506            return Err(invalid("writer received a sliced packed vector"));
8507        }
8508        out.push(u8::try_from(packed.width()).map_err(|_| invalid("packed width overflow"))?);
8509        out.extend_from_slice(&packed.base().to_le_bytes());
8510        put_u32(
8511            &mut out,
8512            u32::try_from(packed.words().len()).map_err(|_| invalid("too many packed words"))?,
8513        );
8514        for word in packed.words() {
8515            put_u64(&mut out, *word);
8516        }
8517        return Ok(out);
8518    }
8519    let data = flat.data().ok_or_else(|| invalid("scalar column did not flatten"))?;
8520    match (ty, data) {
8521        (LogicalType::TinyInt, Data::Int8(values)) => {
8522            for value in &**values {
8523                out.extend_from_slice(&value.to_le_bytes());
8524            }
8525        }
8526        (LogicalType::UTinyInt, Data::UInt8(values)) => {
8527            for value in &**values {
8528                out.extend_from_slice(&value.to_le_bytes());
8529            }
8530        }
8531        (LogicalType::SmallInt, Data::Int16(values)) => {
8532            for value in &**values {
8533                out.extend_from_slice(&value.to_le_bytes());
8534            }
8535        }
8536        (LogicalType::USmallInt, Data::UInt16(values)) => {
8537            for value in &**values {
8538                out.extend_from_slice(&value.to_le_bytes());
8539            }
8540        }
8541        (LogicalType::UInteger, Data::UInt32(values)) => {
8542            for value in &**values {
8543                out.extend_from_slice(&value.to_le_bytes());
8544            }
8545        }
8546        (LogicalType::UBigInt, Data::UInt64(values)) => {
8547            for value in &**values {
8548                out.extend_from_slice(&value.to_le_bytes());
8549            }
8550        }
8551        (LogicalType::Integer | LogicalType::Date, Data::Int32(values)) => {
8552            for value in &**values {
8553                out.extend_from_slice(&value.to_le_bytes());
8554            }
8555        }
8556        (
8557            LogicalType::BigInt
8558            | LogicalType::Timestamp
8559            | LogicalType::Time
8560            | LogicalType::TimeTz
8561            | LogicalType::TimestampTz
8562            | LogicalType::TimestampS
8563            | LogicalType::TimestampMs
8564            | LogicalType::TimestampNs,
8565            Data::Int64(values),
8566        ) => {
8567            for value in &**values {
8568                out.extend_from_slice(&value.to_le_bytes());
8569            }
8570        }
8571        // A hugeint and a uuid are both the 128 bit lane, and a uuid's bits are the ones the rest of
8572        // the engine already carries it in, so nothing about the value changes on the way down.
8573        (LogicalType::HugeInt | LogicalType::Uuid, Data::Int128(values)) => {
8574            for value in &**values {
8575                out.extend_from_slice(&value.to_le_bytes());
8576            }
8577        }
8578        (LogicalType::UHugeInt, Data::UInt128(values)) => {
8579            for value in &**values {
8580                out.extend_from_slice(&value.to_le_bytes());
8581            }
8582        }
8583        // Plainly, in the IEEE bytes. The integer encodings do not apply to a float and none of the
8584        // float codecs is worth having before somebody has measured a corpus of them.
8585        (LogicalType::Float, Data::Float32(values)) => {
8586            for value in &**values {
8587                out.extend_from_slice(&value.to_le_bytes());
8588            }
8589        }
8590        (LogicalType::Double, Data::Float64(values)) => {
8591            for value in &**values {
8592                out.extend_from_slice(&value.to_le_bytes());
8593            }
8594        }
8595        // Three counts and not one number. Months, days and microseconds stay apart on disk because
8596        // they are apart in the value: a month is not a fixed number of days and a day is not a
8597        // fixed number of microseconds, which is the whole reason the type has three fields.
8598        (LogicalType::Interval, Data::Interval(values)) => {
8599            for (months, days, micros) in &**values {
8600                out.extend_from_slice(&months.to_le_bytes());
8601                out.extend_from_slice(&days.to_le_bytes());
8602                out.extend_from_slice(&micros.to_le_bytes());
8603            }
8604        }
8605        (LogicalType::Boolean, Data::Bool(values)) => {
8606            for value in &**values {
8607                out.push(u8::from(*value));
8608            }
8609        }
8610        // The unscaled integer and nothing else. Scale is a property of the column and it is in the
8611        // directory already, so writing it a value at a time would be paying for it twice.
8612        (LogicalType::Decimal { .. }, Data::Int16(values)) => {
8613            for value in &**values {
8614                out.extend_from_slice(&value.to_le_bytes());
8615            }
8616        }
8617        (LogicalType::Decimal { .. }, Data::Int32(values)) => {
8618            for value in &**values {
8619                out.extend_from_slice(&value.to_le_bytes());
8620            }
8621        }
8622        (LogicalType::Decimal { .. }, Data::Int64(values)) => {
8623            for value in &**values {
8624                out.extend_from_slice(&value.to_le_bytes());
8625            }
8626        }
8627        (LogicalType::Decimal { .. }, Data::Int128(values)) => {
8628            for value in &**values {
8629                out.extend_from_slice(&value.to_le_bytes());
8630            }
8631        }
8632        // A blob and a bit string go down the way a varchar does, because the layout is the same
8633        // one: an offset a value and then the bytes. What is not the same is that nothing here may
8634        // read the payload as text, which is why this arm asks the column for bytes rather than for
8635        // a string, and why the codecs above that do read text are all asked of a varchar by name.
8636        (LogicalType::Varchar | LogicalType::Blob | LogicalType::Bit, Data::Varlen(values)) => {
8637            let mut bytes = Vec::new();
8638            put_u32(&mut out, 0);
8639            for row in 0..vector.len() {
8640                let value = values.bytes(row).ok_or_else(|| invalid("string view is invalid"))?;
8641                bytes.extend_from_slice(value);
8642                put_u32(
8643                    &mut out,
8644                    u32::try_from(bytes.len())
8645                        .map_err(|_| invalid("string payload exceeds 4GiB"))?,
8646                );
8647            }
8648            out.extend_from_slice(&bytes);
8649        }
8650        _ => return Err(Error::not_implemented(format!("native page for {ty}"))),
8651    }
8652    Ok(out)
8653}
8654
8655fn put_varint(out: &mut Vec<u8>, mut value: u32) {
8656    while value >= 0x80 {
8657        out.push((value as u8 & 0x7f) | 0x80);
8658        value >>= 7;
8659    }
8660    out.push(value as u8);
8661}
8662
8663/// The distinct codes of one part, which is what a stripe's membership index is merged from.
8664fn unique_codes(codes: &[u32]) -> Vec<u32> {
8665    let mut unique = codes.to_vec();
8666    unique.sort_unstable();
8667    unique.dedup();
8668    unique
8669}
8670
8671/// The union of the sorted distinct codes of every part in a stripe.
8672///
8673/// Pairwise up a tree rather than one long list concatenated and sorted. Both are the same order of
8674/// work on paper and the tree is the one that does not sort what is already in order: sixty four
8675/// sorted lists become one in six passes over the values.
8676fn merged_codes(lists: Vec<Vec<u32>>) -> Vec<u32> {
8677    let mut lists = lists;
8678    while lists.len() > 1 {
8679        let mut next = Vec::with_capacity(lists.len().div_ceil(2));
8680        for pair in lists.chunks(2) {
8681            match pair {
8682                [left, right] => next.push(merged_pair(left, right)),
8683                [only] => next.push(only.clone()),
8684                _ => {}
8685            }
8686        }
8687        lists = next;
8688    }
8689    lists.pop().unwrap_or_default()
8690}
8691
8692fn merged_pair(left: &[u32], right: &[u32]) -> Vec<u32> {
8693    let mut out = Vec::with_capacity(left.len().saturating_add(right.len()));
8694    let mut at = 0;
8695    let mut to = 0;
8696    while at < left.len() && to < right.len() {
8697        match left[at].cmp(&right[to]) {
8698            Ordering::Less => {
8699                out.push(left[at]);
8700                at += 1;
8701            }
8702            Ordering::Greater => {
8703                out.push(right[to]);
8704                to += 1;
8705            }
8706            Ordering::Equal => {
8707                out.push(left[at]);
8708                at += 1;
8709                to += 1;
8710            }
8711        }
8712    }
8713    out.extend_from_slice(&left[at..]);
8714    out.extend_from_slice(&right[to..]);
8715    out
8716}
8717
8718/// The widest bounds and the total null count of a stripe, from the bounds of its parts.
8719///
8720/// A bound that is missing from any part is missing from the stripe, because a missing bound means
8721/// nothing is known and a stripe that holds an unknown cannot claim one.
8722fn merged_range(ranges: impl Iterator<Item = Range>) -> Range {
8723    let mut merged = Range::default();
8724    let mut first = true;
8725    for range in ranges {
8726        merged.nulls = merged.nulls.saturating_add(range.nulls);
8727        // Both of these have to survive every part, so one part that could not say anything makes
8728        // the stripe unable to say it either. A sum is dropped on overflow rather than wrapped,
8729        // which leaves the stripe with exact ends and no total, which is a true thing to say.
8730        merged.sum = match (merged.sum.take(), range.sum) {
8731            (Some(held), Some(next)) if !first => held.checked_add(next),
8732            (_, next) if first => next,
8733            _ => None,
8734        };
8735        merged.exact = if first { range.exact } else { merged.exact && range.exact };
8736        if first {
8737            merged.low = range.low;
8738            merged.high = range.high;
8739            first = false;
8740            continue;
8741        }
8742        merged.low = match (merged.low.take(), range.low) {
8743            (Some(held), Some(next)) => Some(held.smaller(next)),
8744            _ => None,
8745        };
8746        merged.high = match (merged.high.take(), range.high) {
8747            (Some(held), Some(next)) => Some(held.larger(next)),
8748            _ => None,
8749        };
8750    }
8751    merged
8752}
8753
8754/// One stripe's sieves for one column: the part count, a length for each part, then their bytes.
8755///
8756/// One page for the whole stripe rather than one per part, because a part's sieve is a few hundred
8757/// bytes and sixty four of those are sixty four directory entries and sixty four reads for something
8758/// a scan walks straight through. A part with no sieve writes a length of zero and costs four bytes.
8759/// `bound` cut down to [`PART_BOUND_BYTES`], still a bound of the side it was.
8760///
8761/// A prefix of a string sorts at or before the string, so cutting one down leaves a low end that is
8762/// still a low end. A high end has to go the other way, so the cut prefix is stepped up at the last
8763/// byte that can carry it, and a prefix of nothing but `0xFF` has no such byte and gives up the
8764/// bound rather than claiming one that is too small. Anything that is not a string is already a
8765/// fixed width and is left alone.
8766fn shortened(bound: Option<Bound>, high: bool) -> Option<Bound> {
8767    match bound {
8768        Some(Bound::Bytes(mut value)) if value.len() > PART_BOUND_BYTES => {
8769            value.truncate(PART_BOUND_BYTES);
8770            if !high {
8771                return Some(Bound::Bytes(value));
8772            }
8773            while let Some(last) = value.pop() {
8774                if last < u8::MAX {
8775                    value.push(last + 1);
8776                    return Some(Bound::Bytes(value));
8777                }
8778            }
8779            None
8780        }
8781        other => other,
8782    }
8783}
8784
8785/// The ranges of one column's parts of one stripe, as a page.
8786///
8787/// The two ends and the null count, and not `exact` or the total. Those two answer a `MIN` or a
8788/// `SUM` out of the directory, and the directory already answers those per stripe, where the same
8789/// number costs sixty times less to keep. What a part range is for is skipping the part, and
8790/// skipping needs the ends. So a range read back from here says it is not exact, which is true of a
8791/// string end that was cut down anyway.
8792fn encode_part_ranges(ranges: &[Range]) -> Result<Vec<u8>> {
8793    let mut out = Vec::new();
8794    put_u32(
8795        &mut out,
8796        u32::try_from(ranges.len()).map_err(|_| invalid("too many parts in a stripe"))?,
8797    );
8798    for range in ranges {
8799        put_bound(&mut out, shortened(range.low.clone(), false).as_ref())?;
8800        put_bound(&mut out, shortened(range.high.clone(), true).as_ref())?;
8801        put_u32(&mut out, u32::try_from(range.nulls).map_err(|_| invalid("null count overflow"))?);
8802    }
8803    Ok(out)
8804}
8805
8806/// The ranges one encoded page holds, one entry per part of the stripe.
8807fn decode_part_ranges(bytes: &[u8]) -> Result<Vec<Range>> {
8808    let mut cur = Cursor::new(bytes);
8809    let parts = cur.u32()? as usize;
8810    let mut out = Vec::new();
8811    for _ in 0..parts {
8812        let low = cur.bound()?;
8813        let high = cur.bound()?;
8814        let nulls = cur.u32()? as usize;
8815        out.push(Range { low, high, nulls, exact: false, sum: None });
8816    }
8817    Ok(out)
8818}
8819
8820fn encode_sieves<'a>(sieves: impl Iterator<Item = &'a Option<Sieve>>) -> Result<Vec<u8>> {
8821    let held: Vec<&Option<Sieve>> = sieves.collect();
8822    let mut out = Vec::new();
8823    put_u32(
8824        &mut out,
8825        u32::try_from(held.len()).map_err(|_| invalid("too many parts in a stripe"))?,
8826    );
8827    for sieve in &held {
8828        let length = sieve.as_ref().map_or(0, Sieve::len);
8829        put_u32(&mut out, u32::try_from(length).map_err(|_| invalid("sieve length overflow"))?);
8830    }
8831    // flatten: a part with no sieve wrote a length of zero above and contributes no bytes here.
8832    for sieve in held.into_iter().flatten() {
8833        out.extend_from_slice(&sieve.to_bytes());
8834    }
8835    Ok(out)
8836}
8837
8838/// The sieves one encoded page holds, one entry per part of the stripe.
8839///
8840/// A part whose bytes are not a sieve this version understands comes back as `None`, which is a part
8841/// that gets read. That is how a file written by a later version of the sieve stays readable rather
8842/// than being a corrupt page.
8843fn decode_sieves(bytes: &[u8]) -> Result<Vec<Option<Sieve>>> {
8844    let parts = u32::from_le_bytes(
8845        bytes
8846            .get(..4)
8847            .ok_or_else(|| invalid("sieve page is truncated"))?
8848            .try_into()
8849            .map_err(|_| invalid("sieve page is truncated"))?,
8850    ) as usize;
8851    let mut lengths = Vec::with_capacity(parts);
8852    for part in 0..parts {
8853        let at = 4 + part * 4;
8854        let field = bytes.get(at..at + 4).ok_or_else(|| invalid("sieve page is truncated"))?;
8855        lengths.push(u32::from_le_bytes(
8856            field.try_into().map_err(|_| invalid("sieve page is truncated"))?,
8857        ) as usize);
8858    }
8859    let mut at = 4 + parts * 4;
8860    let mut out = Vec::with_capacity(parts);
8861    for length in lengths {
8862        if length == 0 {
8863            out.push(None);
8864            continue;
8865        }
8866        let end = at.checked_add(length).ok_or_else(|| invalid("sieve page is truncated"))?;
8867        let field = bytes.get(at..end).ok_or_else(|| invalid("sieve page is truncated"))?;
8868        out.push(Sieve::from_bytes(field));
8869        at = end;
8870    }
8871    if at != bytes.len() {
8872        return Err(invalid("sieve page has trailing bytes"));
8873    }
8874    Ok(out)
8875}
8876
8877/// One stripe's membership index: the code count and then the codes as ascending deltas.
8878///
8879/// The codes have to be sorted and distinct already, which is what [`unique_codes`] and
8880/// [`merged_codes`] hand over. Anything else decodes as different codes, so neither of those two is
8881/// a step a caller can skip.
8882fn encode_membership(unique: &[u32]) -> Vec<u8> {
8883    let mut out = Vec::with_capacity(unique.len().saturating_mul(2).saturating_add(5));
8884    put_varint(&mut out, u32::try_from(unique.len()).unwrap_or(u32::MAX));
8885    let mut previous = 0;
8886    for (at, &code) in unique.iter().enumerate() {
8887        put_varint(&mut out, if at == 0 { code } else { code - previous });
8888        previous = code;
8889    }
8890    out
8891}
8892
8893fn take_varint(bytes: &[u8], at: &mut usize) -> Result<u32> {
8894    let mut value = 0_u32;
8895    for shift in (0..35).step_by(7) {
8896        let byte = *bytes.get(*at).ok_or_else(|| invalid("membership varint is truncated"))?;
8897        *at += 1;
8898        let part = u32::from(byte & 0x7f);
8899        if shift == 28 && part > 0x0f {
8900            return Err(invalid("membership varint overflow"));
8901        }
8902        value = value
8903            .checked_add(
8904                part.checked_shl(shift).ok_or_else(|| invalid("membership varint overflow"))?,
8905            )
8906            .ok_or_else(|| invalid("membership varint overflow"))?;
8907        if byte & 0x80 == 0 {
8908            return Ok(value);
8909        }
8910    }
8911    Err(invalid("membership varint is too long"))
8912}
8913
8914fn decode_membership(bytes: &[u8]) -> Result<Vec<u32>> {
8915    let mut at = 0;
8916    let count = take_varint(bytes, &mut at)? as usize;
8917    let mut codes = Vec::with_capacity(count);
8918    let mut previous = 0_u32;
8919    for index in 0..count {
8920        let delta = take_varint(bytes, &mut at)?;
8921        let code = if index == 0 {
8922            delta
8923        } else {
8924            previous.checked_add(delta).ok_or_else(|| invalid("membership code overflow"))?
8925        };
8926        if index > 0 && code <= previous {
8927            return Err(invalid("membership codes are not increasing"));
8928        }
8929        codes.push(code);
8930        previous = code;
8931    }
8932    if at != bytes.len() {
8933        return Err(invalid("membership page has trailing bytes"));
8934    }
8935    Ok(codes)
8936}
8937
8938fn string_dictionary(vector: &Vector) -> Result<Option<Vec<u8>>> {
8939    let mut by_text = HashMap::new();
8940    let mut values = Vec::new();
8941    let mut codes = Vec::with_capacity(vector.len());
8942    let mut plain_bytes = 0_usize;
8943    for row in 0..vector.len() {
8944        let text = vector.text_at(row).unwrap_or("");
8945        plain_bytes = plain_bytes.saturating_add(text.len());
8946        let code = match by_text.get(text) {
8947            Some(&code) => code,
8948            None => {
8949                let code = u32::try_from(values.len())
8950                    .map_err(|_| invalid("too many dictionary values"))?;
8951                by_text.insert(text, code);
8952                values.push(text);
8953                code
8954            }
8955        };
8956        codes.push(code);
8957    }
8958    let dictionary_bytes = values.iter().map(|value| value.len()).sum::<usize>();
8959    let encoded = 8_usize
8960        .saturating_add((values.len() + 1).saturating_mul(4))
8961        .saturating_add(dictionary_bytes)
8962        .saturating_add(codes.len().saturating_mul(4));
8963    let plain = (vector.len() + 1).saturating_mul(4).saturating_add(plain_bytes);
8964    if encoded >= plain {
8965        return Ok(None);
8966    }
8967    let mut out = Vec::with_capacity(encoded);
8968    put_u32(
8969        &mut out,
8970        u32::try_from(values.len()).map_err(|_| invalid("too many dictionary values"))?,
8971    );
8972    put_u32(
8973        &mut out,
8974        u32::try_from(dictionary_bytes).map_err(|_| invalid("dictionary payload exceeds 4GiB"))?,
8975    );
8976    let mut offset = 0_u32;
8977    put_u32(&mut out, offset);
8978    for value in &values {
8979        offset = offset
8980            .checked_add(
8981                u32::try_from(value.len()).map_err(|_| invalid("dictionary value is too long"))?,
8982            )
8983            .ok_or_else(|| invalid("dictionary payload exceeds 4GiB"))?;
8984        put_u32(&mut out, offset);
8985    }
8986    for value in values {
8987        out.extend_from_slice(value.as_bytes());
8988    }
8989    for code in codes {
8990        put_u32(&mut out, code);
8991    }
8992    Ok(Some(out))
8993}
8994
8995struct EncodedDictionary {
8996    index: Vec<u8>,
8997    ranks: Vec<u8>,
8998    grams: Vec<u8>,
8999}
9000
9001/// Sorts codes into the byte order of the values they name, eight bytes of depth at a time.
9002///
9003/// # What the shape of the data does to a comparison sort
9004///
9005/// Distinct values against distinct prefixes, on the eight million row `hits`:
9006///
9007/// ```text
9008///   distinct   first 8   first 16   first 32   column
9009///  2,266,417        50      8,892    232,630   URL
9010///  2,346,025        49      8,534    204,060   Referer
9011///  1,357,764    81,362    348,340    861,579   Title
9012/// ```
9013///
9014/// Two and a quarter million URLs have fifty distinct first eight bytes between them, because they
9015/// all begin `http://` and then a host and there are not many hosts. So a sort that leads with
9016/// those eight bytes settles almost nothing on `URL` and `Referer`, whatever the comment on it used
9017/// to say, and almost every pair falls through to a comparison of whole values that agree for most
9018/// of their length. `Title` is free text and separates at eight bytes, which is why the design
9019/// looked right when it was written.
9020///
9021/// # What is done about it
9022///
9023/// Sort on eight bytes of the value at the current depth, held beside the code, and then take each
9024/// run that those eight bytes leave tied and sort it again on the next eight. A value is fetched
9025/// from the payload once per eight bytes of depth rather than once per comparison, and the sort
9026/// itself runs over an array of integers that is in cache rather than over pointers into a payload
9027/// that is hundreds of megabytes.
9028///
9029/// That is the whole trick, and it matters because the payload touch is the expensive part. The
9030/// bytes themselves are nearly free once the line is in cache, so reading eight at a time and
9031/// throwing away the ones that were not needed beats going back for each one.
9032///
9033/// # Why the length has to be carried
9034///
9035/// The eight bytes are padded with zero when the value has fewer than eight left, and a zero byte
9036/// can appear in a value, so equal keys do not mean equal bytes. What is true is that a value which
9037/// ran out inside the window is a prefix of any other value with the same key, and a prefix sorts
9038/// first, so how many of the eight bytes were real is the tie break and nothing further is needed.
9039/// A run is only worth another pass when all eight were real, because otherwise the run is one
9040/// value: a dictionary holds a value once.
9041fn sort_by_value<'a>(codes: &mut [u32], values: impl Fn(u32) -> &'a [u8]) {
9042    let mut work = vec![(0, codes.len(), 0)];
9043    let mut keyed: Vec<(u64, u8, u32)> = Vec::new();
9044    while let Some((from, to, depth)) = work.pop() {
9045        let part = &mut codes[from..to];
9046        keyed.clear();
9047        keyed.extend(part.iter().map(|&code| {
9048            let value = values(code);
9049            let rest = value.get(depth..).unwrap_or_default();
9050            (head(rest), rest.len().min(8) as u8, code)
9051        }));
9052        keyed.sort_unstable();
9053        for (slot, entry) in part.iter_mut().zip(keyed.iter()) {
9054            *slot = entry.2;
9055        }
9056        let mut start = 0;
9057        while start < keyed.len() {
9058            let (key, taken, _) = keyed[start];
9059            let mut end = start + 1;
9060            while end < keyed.len() && keyed[end].0 == key && keyed[end].1 == taken {
9061                end += 1;
9062            }
9063            if taken == 8 && end - start > 1 {
9064                work.push((from + start, from + end, depth + 8));
9065            }
9066            start = end;
9067        }
9068    }
9069}
9070
9071/// How few codes are worth sorting on more than one thread.
9072const PARALLEL_SORT_MIN: usize = 1 << 16;
9073
9074/// How many buckets a thread gets in [`sort_by_value_across`], so that a thread that drew a slow
9075/// bucket is not what the others wait for.
9076const BUCKETS_PER_WORKER: usize = 4;
9077
9078/// How many sampled codes stand for each bucket when the splitters are picked.
9079const SAMPLES_PER_BUCKET: usize = 32;
9080
9081/// [`sort_by_value`] over `workers` threads, with the same answer.
9082///
9083/// A sample sort. A sample of the codes is sorted and cut into as many equal runs as there are
9084/// buckets, and the values at the cuts are the splitters. Every code goes to the bucket its value
9085/// falls in by a binary search of the splitters, the buckets are laid end to end in splitter order,
9086/// and each bucket is then sorted on its own by whichever thread takes it. Every value in a bucket
9087/// sorts after every value in the bucket before, so the buckets sorted one by one are the codes
9088/// sorted.
9089///
9090/// The answer is the one [`sort_by_value`] gives down to the order of equal values, not only the
9091/// order of different ones. A global dictionary holds each value once, so there are none, but the
9092/// sort does not rely on it: equal values land in the same bucket in code order, which is the order
9093/// [`sort_by_value`] leaves them in, since the code is the last thing it sorts on.
9094///
9095/// On the 10m ClickBench sample the close sorts five columns of one to three and a half million
9096/// distinct values, one column at a time, and until this each sort ran on one thread while the
9097/// other thirty one waited for it.
9098fn sort_by_value_across<'a>(
9099    codes: &mut [u32],
9100    values: impl Fn(u32) -> &'a [u8] + Sync,
9101    workers: usize,
9102) {
9103    if workers <= 1 || codes.len() < PARALLEL_SORT_MIN {
9104        sort_by_value(codes, values);
9105        return;
9106    }
9107    let buckets = workers * BUCKETS_PER_WORKER;
9108    let wanted = buckets * SAMPLES_PER_BUCKET;
9109    let mut sample = (0..wanted).map(|at| codes[at * codes.len() / wanted]).collect::<Vec<_>>();
9110    sort_by_value(&mut sample, &values);
9111    let splitters =
9112        (1..buckets).map(|cut| values(sample[cut * sample.len() / buckets])).collect::<Vec<_>>();
9113    let values = &values;
9114    let splitters = &splitters;
9115    let per = codes.len().div_ceil(workers);
9116    // Which bucket each code goes to, a run of the codes per thread.
9117    let places = std::thread::scope(|scope| {
9118        codes
9119            .chunks(per)
9120            .map(|run| {
9121                scope.spawn(move || {
9122                    run.iter()
9123                        .map(|&code| {
9124                            let value = values(code);
9125                            splitters.partition_point(|splitter| *splitter <= value) as u32
9126                        })
9127                        .collect::<Vec<_>>()
9128                })
9129            })
9130            .collect::<Vec<_>>()
9131            .into_iter()
9132            .flat_map(|handle| {
9133                handle.join().unwrap_or_else(|panic| std::panic::resume_unwind(panic))
9134            })
9135            .collect::<Vec<_>>()
9136    });
9137    let mut starts = vec![0_usize; buckets + 1];
9138    for &place in &places {
9139        starts[place as usize + 1] += 1;
9140    }
9141    for bucket in 0..buckets {
9142        starts[bucket + 1] += starts[bucket];
9143    }
9144    let mut laid = vec![0_u32; codes.len()];
9145    let mut next = starts.clone();
9146    for (&code, &place) in codes.iter().zip(&places) {
9147        laid[next[place as usize]] = code;
9148        next[place as usize] += 1;
9149    }
9150    drop(places);
9151    let mut runs = Vec::with_capacity(buckets);
9152    let mut rest = laid.as_mut_slice();
9153    for bucket in 0..buckets {
9154        let (run, after) = rest.split_at_mut(starts[bucket + 1] - starts[bucket]);
9155        runs.push(run);
9156        rest = after;
9157    }
9158    // The largest buckets first, since they are taken from the back.
9159    runs.sort_by_key(|run| run.len());
9160    let queue = Mutex::new(runs);
9161    std::thread::scope(|scope| {
9162        for _ in 0..workers {
9163            scope.spawn(|| {
9164                loop {
9165                    let taken =
9166                        queue.lock().unwrap_or_else(std::sync::PoisonError::into_inner).pop();
9167                    let Some(run) = taken else { break };
9168                    sort_by_value(run, values);
9169                }
9170            });
9171        }
9172    });
9173    codes.copy_from_slice(&laid);
9174}
9175
9176/// The first eight bytes of a value as an integer that sorts the way the bytes sort.
9177fn head(bytes: &[u8]) -> u64 {
9178    let mut word = [0; 8];
9179    let take = bytes.len().min(8);
9180    word[..take].copy_from_slice(&bytes[..take]);
9181    u64::from_be_bytes(word)
9182}
9183
9184/// One column's dictionary page, which is its index and its sorted order.
9185///
9186/// The payload is not in it. Its blocks are in the file already, written as each was encoded, and
9187/// `places` says where, in block order. With `scattered` set the index records each block's start
9188/// and length, so a reader can find one wherever it went.
9189///
9190/// `scattered` false lays the blocks out the way a file written before format 26 has them, one
9191/// behind the next with only the ends recorded. Nothing in the writer asks for that any more. It is
9192/// kept because [`open_global_dictionary`] still reads those files and a reading path that nothing
9193/// can produce is a reading path nothing tests.
9194fn encode_global_dictionary(
9195    dictionary: &GlobalDictionary,
9196    order: &[(u64, u32)],
9197    places: &[Placed],
9198    scattered: bool,
9199) -> Result<EncodedDictionary> {
9200    let values = dictionary.values();
9201    if order.len() != values {
9202        return Err(invalid("global dictionary order does not cover its values"));
9203    }
9204    let blocks = values.div_ceil(TEXT_PAYLOAD_VALUES);
9205    if places.len() != blocks {
9206        return Err(invalid("global dictionary payload is not the blocks it says it is"));
9207    }
9208    if dictionary.grams.len() != blocks {
9209        return Err(invalid("global dictionary signatures do not cover its blocks"));
9210    }
9211    let (ranks, rank_ends) = encode_ranks(order, code_width(values))?;
9212    let rank_blocks = values.div_ceil(TEXT_RANK_BLOCK);
9213    let offset_bits = offset_width(&dictionary.ends);
9214    let payload_words = if scattered { 3 } else { 2 };
9215    let index_len = DICTIONARY_HEADER
9216        .checked_add(offset_bytes(values, offset_bits))
9217        .and_then(|len| len.checked_add(blocks.checked_mul(payload_words * 8)?))
9218        .and_then(|len| len.checked_add(rank_blocks.checked_mul(16)?))
9219        .and_then(|len| len.checked_add(8))
9220        .ok_or_else(|| invalid("global dictionary index length overflow"))?;
9221    let mut index = Vec::with_capacity(index_len);
9222    put_u32(
9223        &mut index,
9224        u32::try_from(values).map_err(|_| invalid("global dictionary has too many values"))?,
9225    );
9226    put_u32(&mut index, TEXT_PAYLOAD_VALUES as u32);
9227    put_u32(
9228        &mut index,
9229        u32::try_from(blocks).map_err(|_| invalid("global dictionary has too many blocks"))?,
9230    );
9231    let flag = (if scattered { DICTIONARY_SCATTERED } else { 0 }) | DICTIONARY_GRAMS;
9232    put_u32(&mut index, offset_bits as u32 | flag);
9233    encode_offsets(&dictionary.ends, offset_bits, &mut index)?;
9234    // Where each block is and how long it is, so a reader can find one. The stored blocks are
9235    // shorter than the decoded ones and by a different amount each, so their lengths are the one
9236    // thing the offsets above no longer say, and where they start is no longer arithmetic on the
9237    // block before once a block is written the moment it is encoded.
9238    let mut end = 0_u64;
9239    for place in places {
9240        if scattered {
9241            put_u64(&mut index, place.start);
9242            put_u64(&mut index, place.length);
9243        } else {
9244            end = end
9245                .checked_add(place.length)
9246                .ok_or_else(|| invalid("global dictionary payload overflow"))?;
9247            put_u64(&mut index, end);
9248        }
9249    }
9250    for place in places {
9251        put_u64(&mut index, place.hash);
9252    }
9253    // The same two lists for the sorted order. A rank block is packed at whatever width its own
9254    // heads need, so where one ends is no longer arithmetic on the block number.
9255    if rank_ends.len() != rank_blocks {
9256        return Err(invalid("global dictionary order is not the blocks it says it is"));
9257    }
9258    for end in &rank_ends {
9259        put_u64(&mut index, *end);
9260    }
9261    let mut at = 0_usize;
9262    for end in &rank_ends {
9263        let end = usize::try_from(*end).map_err(|_| invalid("global dictionary order overflow"))?;
9264        put_u64(&mut index, checksum(&ranks[at..end]));
9265        at = end;
9266    }
9267    let gram_len = blocks
9268        .checked_mul(TEXT_GRAM_BYTES)
9269        .ok_or_else(|| invalid("global dictionary signature count overflow"))?;
9270    let mut grams = Vec::with_capacity(gram_len);
9271    for block in &dictionary.grams {
9272        grams.extend_from_slice(block);
9273    }
9274    put_u64(&mut index, checksum(&grams));
9275    if index.len() != index_len {
9276        return Err(invalid("global dictionary index is not the length it was laid out for"));
9277    }
9278    Ok(EncodedDictionary { index, ranks, grams })
9279}
9280
9281/// How many blocks of the payload the shape is settled on.
9282///
9283/// Eight blocks is 8,192 values, which is the sample `chooser::Sampled` draws and is that size for
9284/// the same reason. They are spread across the dictionary rather than taken off the front, because
9285/// a dictionary is in the order values were first seen and the front of it is the first morsel of
9286/// the load.
9287const PAYLOAD_SAMPLE_BLOCKS: usize = 8;
9288
9289/// The shapes the payload encoder picks between.
9290///
9291/// Narrow on purpose. The exhaustive search encodes every candidate at every level and runs at two
9292/// to six megabytes a second on this data, which over the twelve gigabytes of dictionary `hits`
9293/// carries is about an hour of processor time, so it cannot be what a load does. Each of these
9294/// settles the outer level and the one below it, which is where almost all of that hour goes, and
9295/// leaves the levels under them to the exhaustive search where the chunks are small enough for it
9296/// to cost nothing.
9297///
9298/// Measured on the five ClickBench columns that have a dictionary worth the name, at 1,024 values a
9299/// block, against the exhaustive search over the same blocks:
9300///
9301/// | column | exhaustive | FRONT then LZ | LZ then FSST | LZ then PLAIN |
9302/// |---|---|---|---|---|
9303/// | 2 | 2.923 at 4.3 MB/s | 2.587 at 21.2 | 2.593 at 36.1 | 2.538 at 53.6 |
9304/// | 13 | 3.093 at 3.1 | 3.029 at 36.4 | 2.921 at 35.7 | 2.770 at 82.9 |
9305/// | 14 | 2.330 at 2.1 | 2.283 at 24.3 | 2.213 at 23.5 | 2.113 at 67.6 |
9306/// | 39 | 2.459 at 5.3 | 2.147 at 10.6 | 2.145 at 29.3 | 2.088 at 43.1 |
9307/// | 56 | 4.694 at 6.3 | 4.381 at 51.0 | 4.172 at 50.6 | 3.983 at 86.8 |
9308///
9309/// The best of the three per column is 98 percent of the exhaustive ratio for a tenth of the time.
9310/// `FSST` and `PLAIN` on their own are in the list as a floor rather than to win. `FSST` is the
9311/// right answer for text that does not share prefixes with its neighbours, and `PLAIN` is there so
9312/// that a column nothing compresses is found out in the sample and written at a gigabyte a second
9313/// rather than searched for an answer that does not exist.
9314fn payload_shapes() -> Vec<chooser::Settled> {
9315    let integers = vec![integer::Kind::Packed];
9316    [
9317        vec![string::Kind::Front, string::Kind::Lz],
9318        vec![string::Kind::Lz, string::Kind::Fsst],
9319        vec![string::Kind::Lz, string::Kind::Plain],
9320        vec![string::Kind::Fsst],
9321        vec![string::Kind::Plain],
9322    ]
9323    .into_iter()
9324    .map(|strings| chooser::Settled::new(strings, integers.clone()))
9325    .collect()
9326}
9327
9328/// Encodes every payload block that filled during the stripe just written, across threads.
9329///
9330/// This is where the load's dictionary work happens, and where it happens matters more than it
9331/// looks. It used to happen in [`Writer::close`], over every block of a column at once, which meant
9332/// the raw bytes of every block had to still exist when the load ended. Doing it a block at a time
9333/// inside the stripe encode instead was measured at 2.2 times the wall clock for the same user time:
9334/// a stripe is a barrier the parquet reader waits on, one column's blocks are one thread, and two of
9335/// `hits`'s five dictionary columns hold most of the distinct values, so the whole load ran at less
9336/// than one core.
9337///
9338/// Here is neither. The stripe's columns have all been encoded and handed back by the time this
9339/// runs, so every waiting block of every column is one flat queue and it fans out over all of them
9340/// the way [`Writer::close`] used to fan out over one column's. The work and the parallelism are
9341/// what they were. It happens sixty times during the load rather than once at the end of it, and the
9342/// raw bytes go as it goes.
9343/// Syncs the file, and counts the sync and how long it took as a publish wait when a load is being
9344/// profiled.
9345///
9346/// A wait rather than time, because the time is already in the publish span around it. What the
9347/// wait columns add is how much of publish was the device, which on the WSL2 disk of the gaming PC
9348/// is most of it: a sync there costs about two milliseconds (see `rudb_device_card`).
9349fn synced(file: &File, profile: Option<&LoadProfile>) -> Result<()> {
9350    let started = profile.map(|_| std::time::Instant::now());
9351    file.sync_all().map_err(io)?;
9352    if let (Some(profile), Some(started)) = (profile, started) {
9353        profile.waited(
9354            Stage::Publish,
9355            u64::try_from(started.elapsed().as_nanos()).unwrap_or(u64::MAX),
9356        );
9357    }
9358    Ok(())
9359}
9360
9361fn encode_ready(dictionaries: &mut [Option<GlobalDictionary>]) -> Result<()> {
9362    for dictionary in dictionaries.iter_mut().flatten() {
9363        dictionary.settle()?;
9364    }
9365    // A column with no shape yet is a column with fewer blocks than the sample wants, so its blocks
9366    // wait. There are at most `PAYLOAD_SAMPLE_BLOCKS` of them and they are about to be encoded one
9367    // way or the other, and encoding them now would be encoding them without having looked at the
9368    // column.
9369    encode_waiting(dictionaries, false)
9370}
9371
9372/// Encodes every block still raw at the end of a load: the part block each column ends on and,
9373/// for a column too small to have settled a shape, every block it has.
9374///
9375/// Across threads, the way [`encode_ready`] does it. This ran one column at a time on the thread
9376/// closing the table, and a column that never settled a shape encodes each block by trying every
9377/// candidate, so on a million rows of `hits` it was most of the load's CPU on one core.
9378fn finish_dictionaries(dictionaries: &mut [Option<GlobalDictionary>]) -> Result<()> {
9379    for dictionary in dictionaries.iter_mut().flatten() {
9380        dictionary.seal_rest();
9381    }
9382    encode_waiting(dictionaries, true)
9383}
9384
9385/// Encodes the waiting blocks of every dictionary with a shape, or of every dictionary when
9386/// `closing`, across threads, and appends them to their columns in order.
9387fn encode_waiting(dictionaries: &mut [Option<GlobalDictionary>], closing: bool) -> Result<()> {
9388    let jobs = dictionaries
9389        .iter()
9390        .enumerate()
9391        .filter(|(_, held)| held.as_ref().is_some_and(|held| closing || held.shape.is_some()))
9392        .flat_map(|(column, held)| {
9393            (0..held.as_ref().map_or(0, |held| held.waiting.len())).map(move |at| (column, at))
9394        })
9395        .collect::<Vec<_>>();
9396    if jobs.is_empty() {
9397        return Ok(());
9398    }
9399    let one = |column: usize, at: usize| -> Result<(usize, usize, Vec<u8>)> {
9400        let held = dictionaries[column].as_ref().ok_or_else(|| Error::internal("no dictionary"))?;
9401        Ok((column, at, held.encode_waiting(at)?))
9402    };
9403    let workers = std::thread::available_parallelism()
9404        .map_or(1, usize::from)
9405        .min(MAX_FREQUENCY_WORKERS)
9406        .min(jobs.len());
9407    let made = if workers <= 1 {
9408        jobs.iter().map(|&(column, at)| one(column, at)).collect::<Result<Vec<_>>>()?
9409    } else {
9410        let next = AtomicUsize::new(0);
9411        let jobs = &jobs;
9412        let pieces = std::thread::scope(|scope| {
9413            (0..workers)
9414                .map(|_| {
9415                    scope.spawn(|| {
9416                        let mut mine = Vec::new();
9417                        loop {
9418                            let job = next.fetch_add(1, Atomic::Relaxed);
9419                            let Some(&(column, at)) = jobs.get(job) else { break };
9420                            mine.push(one(column, at)?);
9421                        }
9422                        Ok(mine)
9423                    })
9424                })
9425                .collect::<Vec<_>>()
9426                .into_iter()
9427                .map(|handle| {
9428                    handle
9429                        .join()
9430                        .map_err(|_| Error::internal("a dictionary encode worker panicked"))?
9431                })
9432                .collect::<Result<Vec<_>>>()
9433        })?;
9434        pieces.into_iter().flatten().collect()
9435    };
9436    let mut done: Vec<Vec<(usize, Vec<u8>)>> =
9437        (0..dictionaries.len()).map(|_| Vec::new()).collect();
9438    for (column, at, bytes) in made {
9439        done[column].push((at, bytes));
9440    }
9441    for (column, mut made) in done.into_iter().enumerate() {
9442        if made.is_empty() {
9443            continue;
9444        }
9445        let Some(held) = dictionaries[column].as_mut() else { continue };
9446        made.sort_by_key(|(at, _)| *at);
9447        let waiting = std::mem::take(&mut held.waiting);
9448        for ((block, _), (_, bytes)) in waiting.into_iter().zip(made) {
9449            if held.encoded() != block {
9450                return Err(Error::internal("a dictionary block was encoded out of order"));
9451            }
9452            held.blocks.push(bytes);
9453        }
9454    }
9455    Ok(())
9456}
9457
9458/// Which of [`payload_shapes`] comes out smallest over a sample of the blocks.
9459///
9460/// Every shape is encoded over the same sample and the smallest wins, which is the exhaustive
9461/// search moved up a level: over shapes of a column rather than over candidates of a chunk. The
9462/// sample is spread across the dictionary so that the first and last blocks are both in it, because
9463/// a dictionary written in first seen order has its common values at the front and its long tail at
9464/// the back, and those do not compress alike. Which blocks those are is
9465/// [`GlobalDictionary::seal`]'s to decide, because by the time this is called the rest of them have
9466/// been encoded and the raw bytes are gone.
9467fn settle_shape(sample: &[Vec<&[u8]>]) -> Result<chooser::Settled> {
9468    let mut best: Option<(chooser::Settled, usize)> = None;
9469    for shape in payload_shapes() {
9470        let mut size = 0;
9471        for block in sample {
9472            size += string::encode_with(block, &shape)?.len();
9473        }
9474        if best.as_ref().is_none_or(|(_, smallest)| size < *smallest) {
9475            best = Some((shape, size));
9476        }
9477    }
9478    best.map(|(shape, _)| shape)
9479        .ok_or_else(|| invalid("no shape applies to a global dictionary payload"))
9480}
9481
9482/// The sorted order laid out the way a reader reads it, in blocks of [`TEXT_RANK_BLOCK`] entries.
9483///
9484/// Each block holds its heads first and then its codes, rather than pairing them, because a search
9485/// asks for a head at every probe and for a code about once a search. Keeping the heads together
9486/// means a probe touches eight bytes of a block rather than twelve spread over it, and the last few
9487/// probes of a search, which are the ones that land in the same block, touch the same cache line.
9488fn encode_ranks(order: &[(u64, u32)], code_bits: usize) -> Result<(Vec<u8>, Vec<u64>)> {
9489    let mut out = Vec::with_capacity(order.len() * 4);
9490    let mut ends = Vec::with_capacity(order.len().div_ceil(TEXT_RANK_BLOCK));
9491    let mut heads = Vec::with_capacity(TEXT_RANK_BLOCK);
9492    let mut codes = Vec::with_capacity(TEXT_RANK_BLOCK);
9493    for block in order.chunks(TEXT_RANK_BLOCK) {
9494        // The order is sorted by value and a head is a prefix of a value, so the heads of a block
9495        // rise, the smallest is the first and the largest is the last.
9496        let base = block.first().map_or(0, |&(head, _)| head);
9497        let span = block.last().map_or(0, |&(head, _)| head.wrapping_sub(base));
9498        let width = (u64::BITS - span.leading_zeros()) as usize;
9499        heads.clear();
9500        codes.clear();
9501        for &(head, code) in block {
9502            heads.push(head.wrapping_sub(base));
9503            codes.push(u64::from(code));
9504        }
9505        put_u64(&mut out, base);
9506        out.push(width as u8);
9507        bitpack::pack_tail(&heads, width, &mut out)
9508            .map_err(|_| invalid("global dictionary heads do not pack"))?;
9509        bitpack::pack_tail(&codes, code_bits, &mut out)
9510            .map_err(|_| invalid("global dictionary codes do not pack"))?;
9511        ends.push(out.len() as u64);
9512    }
9513    Ok((out, ends))
9514}
9515
9516/// Opens a column's global dictionary, which reads its index and none of its payload.
9517///
9518/// `keep_budget` is how many decoded payload bytes this dictionary may hold on to, and every
9519/// caller bar the test of the ceiling passes [`TEXT_KEEP_BUDGET`]. It is a parameter rather than
9520/// the constant read where it is used because a test of a ceiling that cannot be moved has to build
9521/// a quarter of a gigabyte of dictionary to reach it.
9522fn open_global_dictionary(
9523    file: Arc<File>,
9524    page: Page,
9525    ty: &LogicalType,
9526    keep_budget: usize,
9527) -> Result<Vector> {
9528    if ty != &LogicalType::Varchar {
9529        return Err(invalid("global dictionary belongs to a non-string column"));
9530    }
9531    let mut header = [0; DICTIONARY_HEADER];
9532    read_at(&file, page.offset, &mut header)?;
9533    let count = u32::from_le_bytes(header[0..4].try_into().expect("four bytes")) as usize;
9534    let per_block = u32::from_le_bytes(header[4..8].try_into().expect("four bytes")) as usize;
9535    let blocks = u32::from_le_bytes(header[8..12].try_into().expect("four bytes")) as usize;
9536    let width = u32::from_le_bytes(header[12..16].try_into().expect("four bytes"));
9537    let scattered = width & DICTIONARY_SCATTERED != 0;
9538    let has_grams = width & DICTIONARY_GRAMS != 0;
9539    let offset_bits = (width & !(DICTIONARY_SCATTERED | DICTIONARY_GRAMS)) as usize;
9540    if per_block != TEXT_PAYLOAD_VALUES {
9541        return Err(invalid("global dictionary block width differs"));
9542    }
9543    if blocks != count.div_ceil(TEXT_PAYLOAD_VALUES) {
9544        return Err(invalid("global dictionary block count differs from its value count"));
9545    }
9546    if offset_bits > u32::BITS as usize {
9547        return Err(invalid("global dictionary packs offsets past a payload"));
9548    }
9549    let offset_len = offset_bytes(count, offset_bits);
9550    // The sorted order is kept out of the index on purpose. The index is read and checksummed in
9551    // full the moment the column is first touched, and the order is half again the size of the
9552    // offsets, so putting it there would make every query that reads a string column pay for a
9553    // search that most of them never make.
9554    let ranks = count;
9555    let rank_blocks = ranks.div_ceil(TEXT_RANK_BLOCK);
9556    // Three words a payload block, for where it starts, how long it is and what it hashes to, or
9557    // two of them on a file that has the blocks back to back and needs no start. Two a rank block
9558    // either way, since those are still one run.
9559    let payload_words = if scattered { 3 } else { 2 };
9560    let hash_len = blocks
9561        .checked_mul(payload_words * 8)
9562        .and_then(|len| len.checked_add(rank_blocks.checked_mul(16)?))
9563        .and_then(|len| len.checked_add(usize::from(has_grams) * 8))
9564        .ok_or_else(|| invalid("global dictionary block count overflow"))?;
9565    let gram_len = if has_grams {
9566        blocks
9567            .checked_mul(TEXT_GRAM_BYTES)
9568            .ok_or_else(|| invalid("global dictionary signature count overflow"))?
9569    } else {
9570        0
9571    };
9572    let index_len = DICTIONARY_HEADER
9573        .checked_add(offset_len)
9574        .and_then(|len| len.checked_add(hash_len))
9575        .ok_or_else(|| invalid("global dictionary header overflow"))?;
9576    if index_len > page.length as usize {
9577        return Err(invalid("global dictionary offset index exceeds its page"));
9578    }
9579    let mut index = vec![0; index_len];
9580    index[..DICTIONARY_HEADER].copy_from_slice(&header);
9581    read_at(&file, page.offset + DICTIONARY_HEADER as u64, &mut index[DICTIONARY_HEADER..])?;
9582    if checksum(&index) != page.hash {
9583        return Err(invalid("global dictionary index checksum differs"));
9584    }
9585    let offsets = index[DICTIONARY_HEADER..DICTIONARY_HEADER + offset_len].to_vec();
9586    let word_end = index_len - usize::from(has_grams) * 8;
9587    let gram_hash = has_grams
9588        .then(|| u64::from_le_bytes(index[word_end..index_len].try_into().expect("eight bytes")));
9589    let mut words = index[DICTIONARY_HEADER + offset_len..word_end]
9590        .chunks_exact(8)
9591        .map(|part| u64::from_le_bytes(part.try_into().expect("eight bytes")))
9592        .collect::<Vec<_>>();
9593    let mut rest = words.split_off(blocks * payload_words);
9594    let rank_hashes = rest.split_off(rank_blocks);
9595    let rank_ends = rest;
9596    // A rank block packs its heads at whatever width its own values need, so its length is no longer
9597    // arithmetic on the block number and the reader has to be told where each one ends.
9598    if rank_ends.windows(2).any(|pair| pair[0] >= pair[1]) {
9599        return Err(invalid("global dictionary order blocks do not rise"));
9600    }
9601    let rank_len = usize::try_from(rank_ends.last().copied().unwrap_or_default())
9602        .map_err(|_| invalid("global dictionary rank overflow"))?;
9603    let body_len = index_len
9604        .checked_add(rank_len)
9605        .ok_or_else(|| invalid("global dictionary header overflow"))?;
9606    if body_len > page.length as usize {
9607        return Err(invalid("global dictionary order exceeds its page"));
9608    }
9609    let gram_end = body_len
9610        .checked_add(gram_len)
9611        .ok_or_else(|| invalid("global dictionary signature length overflow"))?;
9612    if gram_end > page.length as usize {
9613        return Err(invalid("global dictionary signatures exceed their page"));
9614    }
9615    let grams = gram_hash.map(|hash| NativeGrams {
9616        start: page.offset + body_len as u64,
9617        length: gram_len,
9618        hash,
9619        loaded: OnceLock::new(),
9620    });
9621    let hashes = words.split_off(blocks * (payload_words - 1));
9622    let (starts, lengths) = if scattered {
9623        let mut starts = Vec::with_capacity(blocks);
9624        let mut lengths = Vec::with_capacity(blocks);
9625        for pair in words.chunks_exact(2) {
9626            starts.push(pair[0]);
9627            lengths.push(pair[1]);
9628        }
9629        (starts, lengths)
9630    } else {
9631        // A file written before the blocks said where they were has them behind one another at the
9632        // end of the page, so the base is where the sorted order stops and each end is the start of
9633        // the one after it. Turning them round here is what lets everything below take one shape.
9634        let base = page.offset + gram_end as u64;
9635        let mut starts = Vec::with_capacity(blocks);
9636        let mut lengths = Vec::with_capacity(blocks);
9637        let mut at = 0_u64;
9638        for &end in &words {
9639            let len = end
9640                .checked_sub(at)
9641                .ok_or_else(|| invalid("global dictionary block ends before it starts"))?;
9642            starts.push(base + at);
9643            lengths.push(len);
9644            at = end;
9645        }
9646        (starts, lengths)
9647    };
9648    // What the offsets bound is the decoded payload, and what the page length counts is the stored
9649    // one, so on a format 26 file the block lengths adding up to the rest of the page is the one
9650    // thing that ties the index to the page. From format 27 the blocks are written during the load
9651    // and the page is only the index and the order, so there the most that can be said is that
9652    // every block is somewhere in the file past its header.
9653    let stored_len = page.length as u64 - gram_end as u64;
9654    if scattered && stored_len == 0 {
9655        let size = file.metadata().map_err(io)?.len();
9656        let inside = starts.iter().zip(&lengths).all(|(&start, &len)| {
9657            start >= HEADER && start.checked_add(len).is_some_and(|end| end <= size)
9658        });
9659        if !inside {
9660            return Err(invalid("global dictionary block lies outside the file"));
9661        }
9662    } else if lengths.iter().try_fold(0_u64, |sum, len| sum.checked_add(*len)) != Some(stored_len) {
9663        return Err(invalid("global dictionary blocks do not bound the payload"));
9664    }
9665    Vector::external_text(
9666        LogicalType::Varchar,
9667        Arc::new(NativeText {
9668            file,
9669            values: count,
9670            offsets,
9671            offset_bits,
9672            value_ends: OnceLock::new(),
9673            value_lens: OnceLock::new(),
9674            ends_asked: AtomicUsize::new(0),
9675            ranks,
9676            rank_at: page.offset + index_len as u64,
9677            rank_ends,
9678            rank_hashes,
9679            rank_blocks: (0..rank_blocks).map(|_| OnceLock::new()).collect(),
9680            code_bits: code_width(count),
9681            code_ranks: OnceLock::new(),
9682            starts,
9683            lengths,
9684            hashes,
9685            grams,
9686            blocks: (0..blocks).map(|_| OnceLock::new()).collect(),
9687            keep_budget,
9688            payload_kept: AtomicUsize::new(0),
9689            swept: (0..blocks).map(|_| AtomicBool::new(false)).collect(),
9690            searched: Mutex::new(HashMap::new()),
9691        }),
9692    )
9693}
9694
9695/// What a stored page is, without decoding a value out of it.
9696///
9697/// Two layers, and both of them belong in the answer. The codec byte at the front of every page is
9698/// the format's own choice, and it is what says whether the column came back as codes into a table
9699/// wide dictionary, as a bit packed page, as an encoding cascade or as the bytes themselves. Under
9700/// the cascade codecs there is a second choice the encoder made per chunk, and that is what
9701/// [`integer::describe`] and [`string::describe`] already write out as `DICT(PACKED, PACKED)`.
9702///
9703/// This mirrors the tags [`decode`] reads and has to be kept beside it. A page whose header this
9704/// cannot walk comes back as text rather than as an error, because a caller asking what a file
9705/// looks like is usually asking because something is wrong with it, and a report that stops at the
9706/// first bad page is a report that says nothing about the other nine hundred.
9707fn page_encoding(ty: &LogicalType, rows: usize, bytes: &[u8]) -> String {
9708    /// The page header is the codec, the validity tag and, for a page that stores a mask, the mask.
9709    fn cascade_at(rows: usize, bytes: &[u8]) -> Result<(u8, usize)> {
9710        let mut cur = Cursor::new(bytes);
9711        let codec = cur.u8()?;
9712        if cur.u8()? == 2 {
9713            cur.take(rows.div_ceil(8))?;
9714        }
9715        Ok((codec, cur.at))
9716    }
9717    let Ok((codec, at)) = cascade_at(rows, bytes) else {
9718        return "UNREADABLE".to_string();
9719    };
9720    let tail = &bytes[at..];
9721    let described = |described: Result<String>| described.unwrap_or_else(|_| "UNREADABLE".into());
9722    match codec {
9723        0 => match ty {
9724            LogicalType::Varchar | LogicalType::Blob => "PLAIN".to_string(),
9725            _ => "FIXED".to_string(),
9726        },
9727        1 => "DICT(PLAIN)".to_string(),
9728        2 => "FOR+BITPACK".to_string(),
9729        3 => "TABLE DICT".to_string(),
9730        4 => format!("TABLE DICT({})", described(integer::describe(tail))),
9731        5 => described(integer::describe(tail)),
9732        6 => described(string::describe(tail)),
9733        other => format!("CODEC {other}"),
9734    }
9735}
9736
9737/// Selected stable dictionary codes from one page.
9738///
9739/// Pair-frequency construction needs at most the bounded heavy-hitter rows. Reading those code
9740/// positions directly avoids materializing every code in each part that contains a candidate.
9741fn decode_selected_stable_codes(
9742    rows: usize,
9743    bytes: &[u8],
9744    positions: &[usize],
9745    out: &mut Vec<Option<u32>>,
9746) -> Result<bool> {
9747    if positions.windows(2).any(|pair| pair[0] >= pair[1])
9748        || positions.last().is_some_and(|&position| position >= rows)
9749    {
9750        return Err(invalid("selected code positions are not sorted and in range"));
9751    }
9752    let mut cur = Cursor::new(bytes);
9753    let codec = cur.u8()?;
9754    if codec != 3 && codec != 4 {
9755        return Ok(false);
9756    }
9757    let flag = cur.u8()?;
9758    let mask = match flag {
9759        0 | 1 => None,
9760        2 => {
9761            let at = cur.at;
9762            let len = rows.div_ceil(8);
9763            cur.take(len)?;
9764            Some((at, len))
9765        }
9766        _ => return Err(invalid("page validity tag differs")),
9767    };
9768    let valid = |row: usize| match flag {
9769        0 => true,
9770        1 => false,
9771        2 => mask.is_some_and(|(at, _)| bytes[at + row / 8] >> (row % 8) & 1 == 1),
9772        _ => unreachable!("the validity tag was checked"),
9773    };
9774    if codec == 4 {
9775        let wide = integer::decode_selected(&bytes[cur.at..], positions)?;
9776        for (&row, code) in positions.iter().zip(wide) {
9777            let code = u32::try_from(code).map_err(|_| invalid("code is not a code"))?;
9778            out.push(valid(row).then_some(code));
9779        }
9780        return Ok(true);
9781    }
9782    let codes_at = cur.at;
9783    let codes_len = rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?;
9784    cur.take(codes_len)?;
9785    if cur.at != bytes.len() {
9786        return Err(invalid("global code page has trailing bytes"));
9787    }
9788    let codes = &bytes[codes_at..codes_at + codes_len];
9789    for &row in positions {
9790        let at = row.checked_mul(4).ok_or_else(|| invalid("dictionary code offset overflow"))?;
9791        let code = u32::from_le_bytes(
9792            codes[at..at + 4].try_into().map_err(|_| invalid("dictionary code is truncated"))?,
9793        );
9794        out.push(valid(row).then_some(code));
9795    }
9796    Ok(true)
9797}
9798
9799fn decode(
9800    ty: &LogicalType,
9801    rows: usize,
9802    bytes: &[u8],
9803    global: Option<Arc<Vector>>,
9804) -> Result<Vector> {
9805    let mut cur = Cursor::new(bytes);
9806    let codec = cur.u8()?;
9807    let flag = cur.u8()?;
9808    let validity = match flag {
9809        0 => Validity::AllValid,
9810        1 => Validity::AllInvalid,
9811        2 => {
9812            let mask = cur.take(rows.div_ceil(8))?;
9813            Validity::from_iter(rows, |row| mask[row / 8] >> (row % 8) & 1 == 1)
9814        }
9815        _ => return Err(invalid("page validity tag differs")),
9816    };
9817    if codec == 1 {
9818        if ty != &LogicalType::Varchar {
9819            return Err(invalid("dictionary codec belongs to a non-string page"));
9820        }
9821        let count = cur.u32()? as usize;
9822        let payload_len = cur.u32()? as usize;
9823        let offset_bytes = cur.take(
9824            (count + 1)
9825                .checked_mul(4)
9826                .ok_or_else(|| invalid("dictionary offset count overflow"))?,
9827        )?;
9828        let offsets = offset_bytes
9829            .chunks_exact(4)
9830            .map(|part| u32::from_le_bytes(part.try_into().expect("four bytes")))
9831            .collect::<Vec<_>>();
9832        let payload = cur.take(payload_len)?.to_vec();
9833        if offsets.first() != Some(&0)
9834            || offsets.last().copied().map(|last| last as usize) != Some(payload.len())
9835            || offsets.windows(2).any(|pair| pair[0] > pair[1])
9836        {
9837            return Err(invalid("dictionary offsets do not bound the payload"));
9838        }
9839        // A page, because every chunk cut out of this dictionary points at the same payload and a
9840        // page is what lets a cut be the views and nothing else.
9841        let mut strings = StringColumn::over(Buffer::from_vec(payload).into_page());
9842        for pair in offsets.windows(2) {
9843            strings.push_in_place(pair[0] as usize, (pair[1] - pair[0]) as usize)?;
9844        }
9845        let mut codes = Vec::with_capacity(rows);
9846        for _ in 0..rows {
9847            codes.push(cur.u32()?);
9848        }
9849        if codes.iter().any(|code| *code as usize >= count) {
9850            return Err(invalid("dictionary code is out of range"));
9851        }
9852        if cur.at != bytes.len() {
9853            return Err(invalid("dictionary page has trailing bytes"));
9854        }
9855        let dictionary = Vector::flat(LogicalType::Varchar, Data::Varlen(strings))?;
9856        return Ok(Vector::dictionary(codes, dictionary)?.with_validity(validity));
9857    }
9858    if codec == 3 || codec == 4 {
9859        let dictionary = global.ok_or_else(|| invalid("global code page has no dictionary"))?;
9860        let codes = if codec == 4 {
9861            // The cascade holds the whole tail of the page and says how long it is itself, so the
9862            // check that nothing is left over is the one the decoder already makes.
9863            let wide = integer::decode(&bytes[cur.at..])?;
9864            if wide.len() != rows {
9865                return Err(invalid("encoded code page holds the wrong number of rows"));
9866            }
9867            // Checked once for the page rather than a fallible conversion per code. Every code a
9868            // file holds is inside a `u32` or the file is corrupt, so or the codes together and the
9869            // answer has a bit set above the low thirty two, or the sign bit, exactly when one of
9870            // them did. The or and the narrowing are two passes because each is then a vector
9871            // loop. As one loop with a `push` a code, the length check and the store kept it scalar,
9872            // and it was sixteen instructions a row on the two flag columns of q1.
9873            let seen = wide.iter().fold(0_i64, |seen, &code| seen | code);
9874            if seen < 0 || seen > i64::from(u32::MAX) {
9875                return Err(invalid("code is not a code"));
9876            }
9877            wide.iter().map(|&code| code as u32).collect()
9878        } else {
9879            let mut codes = Vec::with_capacity(rows);
9880            for _ in 0..rows {
9881                codes.push(cur.u32()?);
9882            }
9883            if cur.at != bytes.len() {
9884                return Err(invalid("global code page has trailing bytes"));
9885            }
9886            codes
9887        };
9888        let highest = codes.iter().copied().max();
9889        return Ok(Vector::stable_dictionary_validated(codes, dictionary, highest)?
9890            .with_validity(validity));
9891    }
9892    if codec == 6 {
9893        if ty != &LogicalType::Varchar {
9894            return Err(invalid("compressed text codec belongs to a non-string page"));
9895        }
9896        // As codec 5, the layer holds the whole tail of the page and says how long it is itself.
9897        // It comes back as one buffer with the values laid end to end and where each one ends, which
9898        // is the raw form's layout, so what is left to do here is what codec 0 does.
9899        let (payload, ends) = string::decode_flat(&bytes[cur.at..])?.into_parts();
9900        if ends.len() != rows {
9901            return Err(invalid("compressed text page holds the wrong number of rows"));
9902        }
9903        // A page, because this is read once and handed out a chunk at a time, and a cut of a paged
9904        // payload moves views rather than bytes.
9905        let mut values = StringColumn::over(Buffer::from_vec(payload).into_page());
9906        let mut start = 0;
9907        for end in ends {
9908            let len = end
9909                .checked_sub(start)
9910                .ok_or_else(|| invalid("compressed text value ends before it starts"))?;
9911            values.push_in_place(start, len)?;
9912            start = end;
9913        }
9914        return Ok(Vector::flat(ty.clone(), Data::Varlen(values))?.with_validity(validity));
9915    }
9916    if codec == 5 {
9917        // The cascade holds the whole tail of the page and says how long it is itself.
9918        let values = integer::decode(&bytes[cur.at..])?;
9919        if values.len() != rows {
9920            return Err(invalid("cascade page holds the wrong number of rows"));
9921        }
9922        let data = narrowed(ty, values)?;
9923        return Ok(Vector::flat(ty.clone(), data)?.with_validity(validity));
9924    }
9925    if codec == 2 {
9926        let width = u32::from(cur.u8()?);
9927        let base = i128::from_le_bytes(cur.take(16)?.try_into().expect("sixteen bytes"));
9928        let count = cur.u32()? as usize;
9929        let length = count.checked_mul(8).ok_or_else(|| invalid("packed page is too long"))?;
9930        let words: Vec<u64> = cur
9931            .take(length)?
9932            .chunks_exact(8)
9933            .map(|word| u64::from_le_bytes(word.try_into().expect("eight bytes")))
9934            .collect();
9935        if cur.at != bytes.len() {
9936            return Err(invalid("packed page has trailing bytes"));
9937        }
9938        return Ok(Vector::packed(ty.clone(), words, width, base, rows)?.with_validity(validity));
9939    }
9940    if codec != 0 {
9941        return Err(invalid("page codec is unknown"));
9942    }
9943    let data = match ty {
9944        LogicalType::TinyInt => {
9945            let values = cur.take(rows)?;
9946            Data::Int8(values.iter().map(|item| *item as i8).collect::<Vec<_>>().into())
9947        }
9948        LogicalType::UTinyInt => Data::UInt8(cur.take(rows)?.to_vec().into()),
9949        LogicalType::SmallInt => {
9950            let values =
9951                cur.take(rows.checked_mul(2).ok_or_else(|| invalid("page size overflow"))?)?;
9952            Data::Int16(
9953                values
9954                    .chunks_exact(2)
9955                    .map(|item| i16::from_le_bytes(item.try_into().expect("two bytes")))
9956                    .collect::<Vec<_>>()
9957                    .into(),
9958            )
9959        }
9960        LogicalType::USmallInt => {
9961            let values =
9962                cur.take(rows.checked_mul(2).ok_or_else(|| invalid("page size overflow"))?)?;
9963            Data::UInt16(
9964                values
9965                    .chunks_exact(2)
9966                    .map(|item| u16::from_le_bytes(item.try_into().expect("two bytes")))
9967                    .collect::<Vec<_>>()
9968                    .into(),
9969            )
9970        }
9971        LogicalType::UInteger => {
9972            let values =
9973                cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
9974            Data::UInt32(
9975                values
9976                    .chunks_exact(4)
9977                    .map(|item| u32::from_le_bytes(item.try_into().expect("four bytes")))
9978                    .collect::<Vec<_>>()
9979                    .into(),
9980            )
9981        }
9982        LogicalType::UBigInt => {
9983            let values =
9984                cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
9985            Data::UInt64(
9986                values
9987                    .chunks_exact(8)
9988                    .map(|item| u64::from_le_bytes(item.try_into().expect("eight bytes")))
9989                    .collect::<Vec<_>>()
9990                    .into(),
9991            )
9992        }
9993        LogicalType::Integer | LogicalType::Date => {
9994            let values =
9995                cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
9996            Data::Int32(
9997                values
9998                    .chunks_exact(4)
9999                    .map(|item| i32::from_le_bytes(item.try_into().expect("four bytes")))
10000                    .collect::<Vec<_>>()
10001                    .into(),
10002            )
10003        }
10004        LogicalType::BigInt
10005        | LogicalType::Timestamp
10006        | LogicalType::Time
10007        | LogicalType::TimeTz
10008        | LogicalType::TimestampTz
10009        | LogicalType::TimestampS
10010        | LogicalType::TimestampMs
10011        | LogicalType::TimestampNs => {
10012            let values =
10013                cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
10014            Data::Int64(
10015                values
10016                    .chunks_exact(8)
10017                    .map(|item| i64::from_le_bytes(item.try_into().expect("eight bytes")))
10018                    .collect::<Vec<_>>()
10019                    .into(),
10020            )
10021        }
10022        LogicalType::HugeInt | LogicalType::Uuid => {
10023            let values =
10024                cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
10025            Data::Int128(
10026                values
10027                    .chunks_exact(16)
10028                    .map(|item| i128::from_le_bytes(item.try_into().expect("sixteen bytes")))
10029                    .collect::<Vec<_>>()
10030                    .into(),
10031            )
10032        }
10033        LogicalType::UHugeInt => {
10034            let values =
10035                cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
10036            Data::UInt128(
10037                values
10038                    .chunks_exact(16)
10039                    .map(|item| u128::from_le_bytes(item.try_into().expect("sixteen bytes")))
10040                    .collect::<Vec<_>>()
10041                    .into(),
10042            )
10043        }
10044        LogicalType::Float => {
10045            let values =
10046                cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
10047            Data::Float32(
10048                values
10049                    .chunks_exact(4)
10050                    .map(|item| f32::from_le_bytes(item.try_into().expect("four bytes")))
10051                    .collect::<Vec<_>>()
10052                    .into(),
10053            )
10054        }
10055        LogicalType::Double => {
10056            let values =
10057                cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
10058            Data::Float64(
10059                values
10060                    .chunks_exact(8)
10061                    .map(|item| f64::from_le_bytes(item.try_into().expect("eight bytes")))
10062                    .collect::<Vec<_>>()
10063                    .into(),
10064            )
10065        }
10066        LogicalType::Interval => {
10067            let values =
10068                cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
10069            Data::Interval(
10070                values
10071                    .chunks_exact(16)
10072                    .map(|item| {
10073                        (
10074                            i32::from_le_bytes(item[..4].try_into().expect("four bytes")),
10075                            i32::from_le_bytes(item[4..8].try_into().expect("four bytes")),
10076                            i64::from_le_bytes(item[8..].try_into().expect("eight bytes")),
10077                        )
10078                    })
10079                    .collect::<Vec<_>>()
10080                    .into(),
10081            )
10082        }
10083        LogicalType::Boolean => {
10084            let values = cur.take(rows)?;
10085            if values.iter().any(|value| *value > 1) {
10086                return Err(invalid("boolean page has another value"));
10087            }
10088            Data::Bool(values.iter().map(|value| *value == 1).collect::<Vec<_>>().into())
10089        }
10090        // Whichever integer the declared width says, which is the mapping the rest of the engine
10091        // already uses for a decimal in memory.
10092        LogicalType::Decimal { .. } => match ty.physical() {
10093            PhysicalType::Int16 => {
10094                let values =
10095                    cur.take(rows.checked_mul(2).ok_or_else(|| invalid("page size overflow"))?)?;
10096                Data::Int16(
10097                    values
10098                        .chunks_exact(2)
10099                        .map(|item| i16::from_le_bytes(item.try_into().expect("two bytes")))
10100                        .collect::<Vec<_>>()
10101                        .into(),
10102                )
10103            }
10104            PhysicalType::Int32 => {
10105                let values =
10106                    cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
10107                Data::Int32(
10108                    values
10109                        .chunks_exact(4)
10110                        .map(|item| i32::from_le_bytes(item.try_into().expect("four bytes")))
10111                        .collect::<Vec<_>>()
10112                        .into(),
10113                )
10114            }
10115            PhysicalType::Int64 => {
10116                let values =
10117                    cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
10118                Data::Int64(
10119                    values
10120                        .chunks_exact(8)
10121                        .map(|item| i64::from_le_bytes(item.try_into().expect("eight bytes")))
10122                        .collect::<Vec<_>>()
10123                        .into(),
10124                )
10125            }
10126            _ => {
10127                let values =
10128                    cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
10129                Data::Int128(
10130                    values
10131                        .chunks_exact(16)
10132                        .map(|item| i128::from_le_bytes(item.try_into().expect("sixteen bytes")))
10133                        .collect::<Vec<_>>()
10134                        .into(),
10135                )
10136            }
10137        },
10138        LogicalType::Varchar | LogicalType::Blob | LogicalType::Bit => {
10139            let offset_bytes = cur
10140                .take((rows + 1).checked_mul(4).ok_or_else(|| invalid("offset count overflow"))?)?;
10141            let offsets = offset_bytes
10142                .chunks_exact(4)
10143                .map(|part| u32::from_le_bytes(part.try_into().expect("four bytes")))
10144                .collect::<Vec<_>>();
10145            let payload = cur.take(bytes.len() - cur.at)?.to_vec();
10146            if offsets.first() != Some(&0)
10147                || offsets.last().copied().map(|last| last as usize) != Some(payload.len())
10148                || offsets.windows(2).any(|pair| pair[0] > pair[1])
10149            {
10150                return Err(invalid("string offsets do not bound the payload"));
10151            }
10152            // A page for the reason the dictionary payload above is one: the page is read once and
10153            // handed out a chunk at a time, and a cut of a paged payload moves views rather than
10154            // bytes.
10155            //
10156            // A varchar is checked for text on the way in and a blob and a bit string are not,
10157            // because the second pair never claimed to hold any. Reading them through the checking
10158            // seam would refuse a column for holding exactly what it was told to hold.
10159            let mut values = StringColumn::over(Buffer::from_vec(payload).into_page());
10160            let text = ty == &LogicalType::Varchar;
10161            for pair in offsets.windows(2) {
10162                let (at, len) = (pair[0] as usize, (pair[1] - pair[0]) as usize);
10163                if text {
10164                    values.push_in_place(at, len)?;
10165                } else {
10166                    values.push_bytes_in_place(at, len)?;
10167                }
10168            }
10169            Data::Varlen(values)
10170        }
10171        _ => return Err(Error::not_implemented(format!("native page for {ty}"))),
10172    };
10173    if cur.at != bytes.len() {
10174        return Err(invalid("page has trailing bytes"));
10175    }
10176    Ok(Vector::flat(ty.clone(), data)?.with_validity(validity))
10177}
10178
10179#[cfg(test)]
10180mod tests {
10181    use std::fs;
10182    use std::io::{Seek, SeekFrom, Write};
10183    use std::path::PathBuf;
10184    use std::time::{SystemTime, UNIX_EPOCH};
10185
10186    use rudb_common::Stat;
10187    use rudb_common::Value;
10188    use rudb_common::bounds::{Frequencies, Op, Remainder, Zones};
10189    use rudb_common::stat::Provenance;
10190
10191    use super::*;
10192
10193    #[test]
10194    fn a_name_taken_in_pieces_is_the_name_of_the_pieces_joined() {
10195        let bytes: Vec<u8> =
10196            (0..300_u32).map(|at| (at.wrapping_mul(2_654_435_761) >> 13) as u8).collect();
10197        for length in [0, 1, 7, 31, 32, 33, 63, 64, 65, 100, 300] {
10198            let whole = content_name(&bytes[..length]);
10199            for step in [1, 3, 8, 31, 32, 33, 64, 301] {
10200                let mut namer = ContentNamer::default();
10201                bytes[..length].chunks(step).for_each(|piece| namer.update(piece));
10202                assert_eq!(namer.finish(), whole, "{length} bytes in pieces of {step}");
10203            }
10204        }
10205    }
10206
10207    /// The chooser as it was before it could rule kinds out up front: the same narrowing, with every
10208    /// kind tested for. What it writes is what the file used to hold.
10209    #[derive(Debug)]
10210    struct TestsEverything<'a>(&'a dyn chooser::Chooser);
10211
10212    impl chooser::Chooser for TestsEverything<'_> {
10213        fn name(&self) -> &'static str {
10214            "tests everything"
10215        }
10216
10217        fn narrow_strings(
10218            &self,
10219            values: &[&[u8]],
10220            offered: &[string::Kind],
10221            depth: u8,
10222        ) -> Vec<string::Kind> {
10223            self.0.narrow_strings(values, offered, depth)
10224        }
10225
10226        fn narrow_integers(
10227            &self,
10228            values: &[i64],
10229            offered: &[integer::Kind],
10230            depth: u8,
10231        ) -> Vec<integer::Kind> {
10232            self.0.narrow_integers(values, offered, depth)
10233        }
10234    }
10235
10236    #[test]
10237    fn ruling_kinds_out_before_testing_for_them_writes_the_same_bytes() {
10238        let columns: Vec<Vec<i64>> = vec![
10239            vec![],
10240            vec![5; 1000],
10241            (0..1000).collect(),
10242            (0..1000).map(|row| 1_600_000_000_000_000 + row * 1_000_000).collect(),
10243            (0..1000).map(|row| row / 50).collect(),
10244            (0..1000).map(|row| if row % 97 == 0 { row } else { 0 }).collect(),
10245            (0..1000).map(|row| (row * 7919) % 13).collect(),
10246            (0..1000).map(|row| (row * 2_654_435_761) % 1_000_003).collect(),
10247            (0..1000).map(|row| [3, 3, 3, 9, 9, 1][row as usize % 6]).collect(),
10248            (0..1000).map(|row| i64::MIN + row % 3).collect(),
10249        ];
10250        let choosers: [&dyn chooser::Chooser; 2] = [&Fixed, &Codes];
10251        for column in &columns {
10252            for chooser in choosers {
10253                let quick = integer::encode_with(column, chooser).unwrap();
10254                let full = integer::encode_with(column, &TestsEverything(chooser)).unwrap();
10255                assert_eq!(
10256                    quick,
10257                    full,
10258                    "{} on {:?}",
10259                    chooser.name(),
10260                    &column[..column.len().min(8)]
10261                );
10262            }
10263        }
10264    }
10265
10266    /// Parts of a column that all look alike come out of a settled shape byte for byte as they
10267    /// come out of a search, because the search would have kept the same tree on every one.
10268    #[test]
10269    fn parts_that_look_alike_replay_to_the_bytes_a_search_writes() {
10270        let mut settling = Settling::default();
10271        for part in 0..STRIPE_PARTS as i64 {
10272            let values: Vec<i64> = (0..2048)
10273                .map(|row| 1_600_000_000_000_000 + (part * 2048 + row) * 1_000_000 + row % 7)
10274                .collect();
10275            let searched = integer::encode_with(&values, &Fixed).unwrap();
10276            assert_eq!(settling.encode(&values).unwrap(), searched, "part {part}");
10277        }
10278    }
10279
10280    /// A column that changes shape partway through a stripe still reads back, and no part comes
10281    /// out much bigger than a search would have made it, because a replay that stops fitting or
10282    /// grows past a quarter a row is searched.
10283    #[test]
10284    fn a_column_that_changes_under_the_shape_is_searched_again() {
10285        let mut state = 0x9e37_79b9_7f4a_7c15_u64;
10286        let mut noise = move || {
10287            state ^= state << 13;
10288            state ^= state >> 7;
10289            state ^= state << 17;
10290            (state % 1_000_000) as i64
10291        };
10292        let mut settling = Settling::default();
10293        for part in 0..STRIPE_PARTS as i64 {
10294            let values: Vec<i64> = match part / 16 {
10295                0 => (0..2048).map(|row| (part * 2048 + row) / 300).collect(),
10296                1 => (0..2048).map(|_| noise()).collect(),
10297                2 => (0..2048).map(|row| if row % 97 == 0 { row } else { 42 }).collect(),
10298                _ => (0..2048).map(|row| 5 + (part * 2048 + row) * 1_000_000).collect(),
10299            };
10300            let settled = settling.encode(&values).unwrap();
10301            assert_eq!(integer::decode(&settled).unwrap(), values, "part {part}");
10302            let searched = integer::encode_with(&values, &Fixed).unwrap();
10303            assert!(
10304                settled.len() * 4 <= searched.len() * 5,
10305                "part {part}: {} settled against {} searched, {} against {}",
10306                settled.len(),
10307                searched.len(),
10308                integer::describe(&settled).unwrap(),
10309                integer::describe(&searched).unwrap(),
10310            );
10311        }
10312    }
10313
10314    #[test]
10315    fn checksum_matches_fixed_vectors() {
10316        assert_eq!(checksum(b""), 0xef46_db37_51d8_e999);
10317        assert_eq!(checksum(b"a"), 0xd24e_c4f1_a98c_6e5b);
10318        assert_eq!(checksum(b"abc"), 0x44bc_2cf5_ad77_0999);
10319    }
10320
10321    #[test]
10322    fn sorting_across_threads_matches_sorting_on_one() {
10323        let mut state = 0x9e37_79b9_7f4a_7c15_u64;
10324        let mut next = move || {
10325            state ^= state << 13;
10326            state ^= state >> 7;
10327            state ^= state << 17;
10328            state
10329        };
10330        let mut values = Vec::new();
10331        for at in 0..150_000_u64 {
10332            let value = match next() % 6 {
10333                0 => Vec::new(),
10334                1 => format!("https://example.com/{}", next() % 5_000).into_bytes(),
10335                2 => format!("https://example.com/path/{at}").into_bytes(),
10336                3 => b"same".to_vec(),
10337                4 => vec![0xff; (next() % 12) as usize],
10338                _ => (0..next() % 20).map(|_| (next() % 3) as u8).collect(),
10339            };
10340            values.push(value);
10341        }
10342        let value = |code: u32| values[code as usize].as_slice();
10343        for workers in [1, 2, 3, 8, 32] {
10344            let mut one = (0..values.len() as u32).rev().collect::<Vec<_>>();
10345            let mut across = one.clone();
10346            sort_by_value(&mut one, value);
10347            sort_by_value_across(&mut across, value, workers);
10348            assert_eq!(one, across, "{workers} workers");
10349        }
10350        let mut sorted = (0..values.len() as u32).collect::<Vec<_>>();
10351        sort_by_value_across(&mut sorted, value, 8);
10352        assert!(sorted.windows(2).all(|pair| value(pair[0]) <= value(pair[1])));
10353    }
10354
10355    fn path(label: &str) -> PathBuf {
10356        let stamp = SystemTime::now().duration_since(UNIX_EPOCH).expect("time advances").as_nanos();
10357        std::env::temp_dir().join(format!("rudb-native-{label}-{}-{stamp}.rdb", std::process::id()))
10358    }
10359
10360    /// Every value of a dictionary in code order, which the tests have no other way to ask for now
10361    /// that a dictionary does not keep the bytes of the values it has seen.
10362    ///
10363    /// Only valid once `finish_blocks` has run, because until then the last part block is still raw.
10364    fn dictionary_values(dictionary: &GlobalDictionary) -> Vec<Vec<u8>> {
10365        let (flat, bases) = dictionary.decoded(None).expect("the blocks decode");
10366        (0..dictionary.values())
10367            .map(|code| {
10368                let (from, to) = GlobalDictionary::value_span(&dictionary.ends, &bases, code);
10369                flat[from..to].to_vec()
10370            })
10371            .collect()
10372    }
10373
10374    /// The sections a test put in the table, which is every one the writer did not.
10375    ///
10376    /// A table now carries a summary and a sketch per column out of the write itself, and a test
10377    /// about the section table is not about those. Filtering by kind rather than by count, so a
10378    /// table that turns out to have no room for its summaries does not quietly change what these
10379    /// tests are asserting over.
10380    fn attached(table: &Table) -> Vec<&Section> {
10381        table.sections().iter().filter(|held| !held.among(section::STATISTICS_KINDS)).collect()
10382    }
10383
10384    /// A read names the offset it wants, so a cursor somebody else moved cannot reach it.
10385    #[test]
10386    fn a_read_at_an_offset_ignores_where_another_thread_left_the_cursor() {
10387        const SPANS: usize = 64;
10388        const SPAN: usize = 512;
10389        let path = path("positional");
10390        let content: Vec<u8> =
10391            (0..SPANS).flat_map(|span| std::iter::repeat_n(span as u8, SPAN)).collect();
10392        fs::write(&path, &content).expect("the file is written");
10393        let file = Arc::new(File::open(&path).expect("the file opens"));
10394        std::thread::scope(|scope| {
10395            for _ in 0..8 {
10396                let file = Arc::clone(&file);
10397                scope.spawn(move || {
10398                    for _ in 0..64 {
10399                        for span in 0..SPANS {
10400                            let mut bytes = [0_u8; SPAN];
10401                            read_at(&file, (span * SPAN) as u64, &mut bytes)
10402                                .expect("the span reads");
10403                            assert!(
10404                                bytes.iter().all(|byte| *byte == span as u8),
10405                                "span {span} came back as {}",
10406                                bytes[0],
10407                            );
10408                        }
10409                    }
10410                });
10411            }
10412        });
10413        let mut past = [0_u8; SPAN];
10414        let end = (SPANS * SPAN) as u64;
10415        let error = read_at(&file, end, &mut past).expect_err("a read past the end is refused");
10416        assert!(error.message().contains("ends before its declared length"), "{error}");
10417        drop(file);
10418        let _ = fs::remove_file(&path);
10419    }
10420
10421    /// The writer records where it put a page and puts it there, whatever the cursor is doing.
10422    ///
10423    /// The cursor is moved between the steps that record an offset, which is what reading the pages
10424    /// back to build the frequencies does on a platform with no `pread`. Without the fix the
10425    /// directory lands on top of a page and the file fails to reopen.
10426    #[test]
10427    fn a_writer_puts_a_page_where_it_said_it_did_wherever_the_cursor_has_got_to() {
10428        let path = path("cursor");
10429        let mut writer = Writer::create(
10430            &path,
10431            "items",
10432            vec![
10433                Field::required("id", LogicalType::Integer),
10434                Field::new("text", LogicalType::Varchar),
10435            ],
10436        )
10437        .expect("new file");
10438        writer.append(&sample()).expect("first part");
10439        writer.file.seek(SeekFrom::Start(0)).expect("the cursor goes back to the header");
10440        writer.append(&sample()).expect("second part");
10441        writer.file.seek(SeekFrom::Start(1)).expect("and somewhere useless again");
10442        writer.finish().expect("commit");
10443        let reader = Reader::open(&path).expect("reopen from disk");
10444        assert_eq!(reader.table().rows(), 6);
10445        let ids = reader.read(0, &[0]).expect("the integer page reads back");
10446        assert_eq!(ids.value_at(0, 0), Value::Integer(4));
10447        assert_eq!(ids.value_at(2, 0), Value::Integer(-2));
10448        let text = reader.read(1, &[1]).expect("the text page reads back");
10449        assert_eq!(text.value_at(1, 0), Value::Null);
10450        assert_eq!(text.value_at(2, 0), Value::Varchar("long text after a slash".into()));
10451        // Nothing the directory points at may run past the end of the file, which is the shape the
10452        // failure took: a page recorded at an offset the directory had already been written over.
10453        let end = reader.table().stripes().iter().flat_map(|stripe| {
10454            stripe
10455                .pages
10456                .iter()
10457                .map(|page| page.offset + u64::from(page.length))
10458                .chain(std::iter::once(stripe.index.offset + u64::from(stripe.index.length)))
10459        });
10460        let last = end.fold(HEADER, u64::max);
10461        let directory = fs::metadata(&path).expect("the file is there").len();
10462        assert!(last <= directory, "a page runs to {last} in a file of {directory} bytes");
10463        fs::remove_file(path).expect("remove scratch file");
10464    }
10465
10466    /// How long a global dictionary index is, read out of the page's own header.
10467    ///
10468    /// The tests below damage a byte of the order or of the payload, so they need to know where each
10469    /// one starts, and working it out here rather than writing a number down means adding something
10470    /// to the index does not quietly turn one of them into a test that damages the index instead.
10471    fn dictionary_index_len(header: &[u8; DICTIONARY_HEADER]) -> u64 {
10472        let count = u64::from(u32::from_le_bytes(header[0..4].try_into().expect("four bytes")));
10473        let blocks = u64::from(u32::from_le_bytes(header[8..12].try_into().expect("four bytes")));
10474        let width = u32::from_le_bytes(header[12..16].try_into().expect("four bytes"));
10475        let bits = (width & !(DICTIONARY_SCATTERED | DICTIONARY_GRAMS)) as usize;
10476        let payload_words = if width & DICTIONARY_SCATTERED == 0 { 2 } else { 3 };
10477        let rank_blocks = count.div_ceil(TEXT_RANK_BLOCK as u64);
10478        DICTIONARY_HEADER as u64
10479            + offset_bytes(count as usize, bits) as u64
10480            + blocks * payload_words * 8
10481            + rank_blocks * 16
10482            + if width & DICTIONARY_GRAMS == 0 { 0 } else { 8 }
10483    }
10484
10485    fn sample() -> Chunk {
10486        Chunk::new(vec![
10487            Vector::from_values(
10488                LogicalType::Integer,
10489                &[Value::Integer(4), Value::Integer(9), Value::Integer(-2)],
10490            )
10491            .expect("integers"),
10492            Vector::from_values(
10493                LogicalType::Varchar,
10494                &[
10495                    Value::Varchar("alpha".into()),
10496                    Value::Null,
10497                    Value::Varchar("long text after a slash".into()),
10498                ],
10499            )
10500            .expect("strings"),
10501        ])
10502        .expect("matching rows")
10503    }
10504
10505    fn sample_ids() -> Chunk {
10506        Chunk::new(vec![
10507            Vector::flat(LogicalType::Integer, Data::Int32(vec![7, 8, 9].into()))
10508                .expect("integers"),
10509        ])
10510        .expect("one column")
10511    }
10512
10513    #[test]
10514    fn the_planner_gets_the_null_count_off_the_same_directory_the_bounds_are_in() {
10515        // Six rows, two of them null. `IS NULL` used to get the same fifth any unreadable
10516        // condition gets, and the number was in the stripe entry next to the bounds all along.
10517        let path = path("nulls_for_the_planner");
10518        let mut writer =
10519            Writer::create(&path, "items", vec![Field::new("a", LogicalType::Integer)])
10520                .expect("new file");
10521        let rows = Chunk::new(vec![
10522            Vector::from_values(
10523                LogicalType::Integer,
10524                &[
10525                    Value::Integer(4),
10526                    Value::Null,
10527                    Value::Integer(9),
10528                    Value::Null,
10529                    Value::Integer(1),
10530                    Value::Integer(2),
10531                ],
10532            )
10533            .expect("integers"),
10534        ])
10535        .expect("one column");
10536        writer.append(&rows).expect("the only part");
10537        writer.finish().expect("commit");
10538        let reader = Reader::open(&path).expect("reopen from disk");
10539        let stripes = Stripes::new(reader);
10540        let column = stripes.column("a").expect("the file has that column");
10541        assert_eq!(stripes.nulls(column), Stat::exact(2, Provenance::NullCount));
10542        // A column the file does not have. Zero here would be a fact about a column that is not
10543        // there, which the planner would then divide by.
10544        assert_eq!(stripes.nulls(column + 1), Stat::Unknown);
10545        fs::remove_file(&path).expect("clean up");
10546    }
10547
10548    #[test]
10549    fn the_planner_gets_a_row_count_per_value_off_a_complete_synopsis() {
10550        // The whole of the frequency half of #1106, end to end over a real file. Six rows, three
10551        // of one value and two of another, and a complete synopsis because six rows is well inside
10552        // what the writer can account for. The estimate for `id = 4` is three rows rather than a
10553        // sixth of the table, and for a value the file does not hold it is none.
10554        let path = path("frequencies_for_the_planner");
10555        let mut writer =
10556            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
10557                .expect("new file");
10558        let rows = Chunk::new(vec![
10559            Vector::from_values(
10560                LogicalType::Integer,
10561                &[
10562                    Value::Integer(4),
10563                    Value::Integer(4),
10564                    Value::Integer(4),
10565                    Value::Integer(9),
10566                    Value::Integer(9),
10567                    Value::Integer(1),
10568                ],
10569            )
10570            .expect("integers"),
10571        ])
10572        .expect("one column");
10573        writer.append(&rows).expect("the only part");
10574        writer.finish().expect("commit");
10575        let reader = Reader::open(&path).expect("reopen from disk");
10576        let common = Common::new(reader);
10577        assert_eq!(common.rows(), 6);
10578        let column = common.column("id").expect("the file has that column");
10579        assert_eq!(common.column("nothing"), None);
10580        assert_eq!(
10581            common.rows_with(column, &Bound::Int(4)),
10582            Stat::exact(3, Provenance::FrequencySynopsis)
10583        );
10584        // Not in the file, and a synopsis that accounts for all six rows proves it.
10585        assert_eq!(
10586            common.rows_with(column, &Bound::Int(7)),
10587            Stat::exact(0, Provenance::FrequencySynopsis)
10588        );
10589        // A constant of another domain against an integer column. Nothing in the list compares
10590        // with it, so the zero above would be an artefact of the mismatch rather than a fact.
10591        assert_eq!(common.rows_with(column, &Bound::Bytes(b"four".to_vec())), Stat::Unknown);
10592        // A complete list has no remainder. Answering one of no rows over no values would hand the
10593        // caller a division to special case, and the counts above already answer this column.
10594        assert_eq!(common.remainder(column), None);
10595        fs::remove_file(&path).expect("clean up");
10596    }
10597
10598    #[test]
10599    fn string_frequency_estimates_do_not_open_the_global_dictionary() {
10600        let path = path("string_frequencies_for_the_planner");
10601        let mut writer =
10602            Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
10603                .expect("new file");
10604        let rows = Chunk::new(vec![
10605            Vector::from_values(
10606                LogicalType::Varchar,
10607                &[
10608                    Value::Varchar(String::new()),
10609                    Value::Varchar("alpha".into()),
10610                    Value::Varchar(String::new()),
10611                    Value::Varchar("beta".into()),
10612                    Value::Varchar(String::new()),
10613                ],
10614            )
10615            .expect("strings"),
10616        ])
10617        .expect("one column");
10618        writer.append(&rows).expect("the only part");
10619        writer.finish().expect("commit");
10620
10621        let reader = Reader::open(&path).expect("reopen from disk");
10622        assert_eq!(reader.reads().dictionaries, 0, "open reads only the directory");
10623        let common = Common::new(reader.clone());
10624        let column = common.column("text").expect("the file has that column");
10625        assert_eq!(
10626            common.rows_with(column, &Bound::Bytes(Vec::new())),
10627            Stat::exact(3, Provenance::FrequencySynopsis)
10628        );
10629        assert_eq!(
10630            common.rows_with(column, &Bound::Bytes(b"missing".to_vec())),
10631            Stat::exact(0, Provenance::FrequencySynopsis)
10632        );
10633        assert_eq!(
10634            reader.reads().dictionaries,
10635            0,
10636            "the bounded spellings answer without opening the dictionary index"
10637        );
10638        fs::remove_file(&path).expect("clean up");
10639    }
10640
10641    #[test]
10642    fn host_groups_certify_omitted_hosts_and_keep_exact_aggregates() {
10643        let path = path("certified_host_groups");
10644        let mut writer =
10645            Writer::create(&path, "hits", vec![Field::required("Referer", LogicalType::Varchar)])
10646                .expect("new file");
10647        let mut values = vec![Value::Varchar("http://www.example.com/a".into()); 150];
10648        values.extend(vec![Value::Varchar("https://example.com/b".into()); 70]);
10649        values.extend((0..550).map(|at| Value::Varchar(format!("https://site{at}.test/x"))));
10650        values.push(Value::Varchar(String::new()));
10651        for part in values.chunks(512) {
10652            writer
10653                .append(
10654                    &Chunk::new(vec![
10655                        Vector::from_values(LogicalType::Varchar, part).expect("strings"),
10656                    ])
10657                    .expect("one column"),
10658                )
10659                .expect("part written");
10660        }
10661        writer.finish().expect("commit");
10662        let reader = Reader::open(&path).expect("reopen");
10663        let summary = reader.table.host_groups.as_ref().expect("bounded host metadata");
10664        assert!(summary.omitted_max < 220);
10665        assert!(reader.host_groups(0, summary.omitted_max).expect("valid column").is_none());
10666        let groups = reader.host_groups(0, 220).expect("valid column").expect("certified");
10667        let example = groups.iter().find(|entry| entry.host == "example.com").expect("leader");
10668        assert_eq!(example.count, 220);
10669        assert_eq!(example.bytes_sum, 150 * 24 + 70 * 21);
10670        assert_eq!(example.minimum, "http://www.example.com/a");
10671        assert_eq!(reader.reads().dictionaries, 0, "the directory settles the question");
10672        fs::remove_file(&path).expect("clean up");
10673    }
10674
10675    /// A table directory with nothing in it but a name and one column, for the section tests.
10676    ///
10677    /// The section table is orthogonal to everything else in a directory, so the tests that pin it
10678    /// say so by starting from the emptiest table that encodes.
10679    fn bare_table(sections: Vec<Section>) -> Table {
10680        Table {
10681            name: "linked".to_owned(),
10682            fields: vec![Field::required("id", LogicalType::Integer)],
10683            stripes: Vec::new(),
10684            rows: 0,
10685            dictionaries: vec![None],
10686            dictionary_payloads: Vec::new(),
10687            distincts: vec![None],
10688            frequencies: vec![None],
10689            pair_frequencies: Vec::new(),
10690            frequency_texts: Vec::new(),
10691            host_groups: None,
10692            clustering: None,
10693            generation: 1,
10694            sections,
10695        }
10696    }
10697
10698    fn a_key_map_section() -> Section {
10699        Section {
10700            kind: *section::KEY_MAP,
10701            id: 1,
10702            generation: 3,
10703            extents: 1,
10704            extent_page: HEADER,
10705            extent_bytes: section::EXTENT_BYTES as u32,
10706            hash: 0x1234_5678_9abc_def0,
10707            flags: 0,
10708            header_bytes: 24,
10709        }
10710    }
10711
10712    #[test]
10713    fn a_section_table_round_trips_through_a_directory() {
10714        let mut later = a_key_map_section();
10715        later.kind = *b"RUDBZZ9\0";
10716        later.id = 2;
10717        let table = bare_table(vec![a_key_map_section(), later]);
10718        let directory = encode_directory(&table).expect("directory");
10719        let decoded = decode_directory(&directory, 1 << 20).expect("reopen");
10720        assert_eq!(decoded.sections(), &[a_key_map_section(), later]);
10721        // The second is a kind this build has no name for, and it survived the round trip anyway.
10722        // That is what keeps an old build from silently discarding a newer build's work when it
10723        // rewrites a directory.
10724        assert!(decoded.sections()[0].known());
10725        assert!(!decoded.sections()[1].known());
10726    }
10727
10728    #[test]
10729    fn a_directory_written_before_the_section_table_reads_as_a_table_with_none() {
10730        // The G1 exit criterion, at the directory level. A format 22 directory is exactly this
10731        // build's directory with the trailing section block cut off, so cutting it off is the
10732        // honest way to make one: no fixture to go stale, and no separate encoder to drift.
10733        let directory = encode_directory(&bare_table(Vec::new())).expect("directory");
10734        let block = SECTIONS.len() + size_of::<u64>() + size_of::<u16>();
10735        let older = &directory[..directory.len() - block];
10736        let decoded = decode_directory(older, 1 << 20).expect("a directory from before sections");
10737        assert!(decoded.sections().is_empty());
10738        assert_eq!(decoded.generation(), 0, "a format 22 table recorded no generation");
10739        assert_eq!(decoded.name(), "linked");
10740        assert_eq!(decoded.fields().len(), 1, "everything before the block still decodes");
10741    }
10742
10743    #[test]
10744    fn a_file_stamped_with_the_previous_format_still_opens_and_reads() {
10745        // The same criterion end to end, which is the one the milestone actually asks for: a build
10746        // that knows about sections opens a file written by a build that did not, with no rewrite
10747        // and no repair, and answers from it. The version field is patched rather than a file
10748        // committed by an old binary because the bytes either side of it are identical: format 22
10749        // and format 23 differ only in a trailing directory block, and a reader that stops before
10750        // that block gets a table with no sections.
10751        let path = path("format_twenty_two");
10752        let mut writer =
10753            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
10754                .expect("new file");
10755        let rows = Chunk::new(vec![
10756            Vector::from_values(
10757                LogicalType::Integer,
10758                &[Value::Integer(1), Value::Integer(2), Value::Integer(3)],
10759            )
10760            .expect("integers"),
10761        ])
10762        .expect("one column");
10763        writer.append(&rows).expect("the only part");
10764        writer.finish().expect("commit");
10765
10766        let file = OpenOptions::new().write(true).open(&path).expect("reopen to patch");
10767        write_at(&file, 8, &22_u32.to_le_bytes()).expect("stamp the older format");
10768        drop(file);
10769
10770        let reader = Reader::open(&path).expect("a format 22 file opens unchanged");
10771        assert_eq!(reader.table().rows(), 3);
10772        // The rows and not the section table, because the section block is found by the magic at
10773        // the end of the directory rather than by the number in the header, so stamping the header
10774        // back does not take away the summaries this writer put there. What the test is about is
10775        // that the version check accepts 22, and the rows coming back is what says it did.
10776        assert_eq!(reader.read(0, &[0]).expect("the part still reads").len(), 3);
10777
10778        // And a format this build has never written is still refused, so the accept set is a list
10779        // and not an absence of a check.
10780        let file = OpenOptions::new().write(true).open(&path).expect("reopen to patch");
10781        write_at(&file, 8, &21_u32.to_le_bytes()).expect("stamp an unreadable format");
10782        drop(file);
10783        let error = Reader::open(&path).expect_err("format 21 is not readable");
10784        assert!(error.to_string().contains("format 21"), "{error}");
10785
10786        fs::remove_file(&path).expect("clean up");
10787    }
10788
10789    #[test]
10790    fn a_section_whose_extent_table_is_outside_the_file_is_refused() {
10791        // The bound the format has to check and `section` cannot, because only the reader knows how
10792        // big the file is. Reading the payload a section like this names would be reading whatever
10793        // else happens to be at that offset, which is the one way a graph section could turn into a
10794        // wrong answer rather than a slow one.
10795        let mut past = a_key_map_section();
10796        past.extent_page = 1 << 30;
10797        let directory = encode_directory(&bare_table(vec![past])).expect("directory");
10798        let error = decode_directory(&directory, 1 << 20).expect_err("refused");
10799        assert!(error.to_string().contains("outside the file"), "{error}");
10800
10801        let mut inside_the_header = a_key_map_section();
10802        inside_the_header.extent_page = 8;
10803        let directory = encode_directory(&bare_table(vec![inside_the_header])).expect("directory");
10804        assert!(
10805            decode_directory(&directory, 1 << 20).is_err(),
10806            "a section may not overlap a header"
10807        );
10808    }
10809
10810    #[test]
10811    fn a_section_recorded_as_not_built_is_legal_and_names_no_bytes() {
10812        // Section 3.7: a relationship that does not fit the budget is recorded with its size so
10813        // that `rudb_links()` can report what a larger budget would buy. That record is a section
10814        // entry with no extents, so it has to survive a round trip while naming nothing.
10815        let not_built = Section {
10816            kind: *section::FORWARD_LINK,
10817            id: 9,
10818            generation: 3,
10819            extents: 0,
10820            extent_page: 0,
10821            extent_bytes: 0,
10822            hash: 0,
10823            flags: 0,
10824            header_bytes: 0,
10825        };
10826        let directory = encode_directory(&bare_table(vec![not_built])).expect("directory");
10827        let decoded = decode_directory(&directory, 1 << 20).expect("reopen");
10828        assert_eq!(decoded.sections(), &[not_built]);
10829
10830        // But a section with no extents that still names an extent table is incoherent, and an
10831        // incoherent entry is a torn directory rather than a relationship that was skipped.
10832        let mut incoherent = not_built;
10833        incoherent.extent_bytes = 28;
10834        incoherent.extent_page = HEADER;
10835        let directory = encode_directory(&bare_table(vec![incoherent])).expect("directory");
10836        assert!(decode_directory(&directory, 1 << 20).is_err());
10837    }
10838
10839    #[test]
10840    fn a_directory_naming_more_sections_than_the_bound_is_refused() {
10841        let directory = encode_directory(&bare_table(Vec::new())).expect("directory");
10842        let mut torn = directory.clone();
10843        let count_at = torn.len() - size_of::<u16>();
10844        torn[count_at..].copy_from_slice(&u16::MAX.to_le_bytes());
10845        // Not an allocation of sixty five thousand entries off a torn count: either the bound
10846        // refuses it or the bytes run out, and both are errors rather than a read past the end.
10847        assert!(decode_directory(&torn, 1 << 20).is_err());
10848    }
10849
10850    /// A committed one column file of `rows` integers, for the attach tests.
10851    fn linked_file(label: &str, rows: i32) -> PathBuf {
10852        let path = path(label);
10853        let mut writer =
10854            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
10855                .expect("new file");
10856        let values = (0..rows).map(Value::Integer).collect::<Vec<_>>();
10857        let chunk =
10858            Chunk::new(vec![Vector::from_values(LogicalType::Integer, &values).expect("integers")])
10859                .expect("one column");
10860        writer.append(&chunk).expect("the only part");
10861        writer.finish().expect("commit");
10862        path
10863    }
10864
10865    fn a_key_map_payload() -> Vec<u8> {
10866        // Shaped like one without being one: this crate never reads a payload, so what matters here
10867        // is that every byte comes back and that the header the entry measures is at the front.
10868        (0..512_u32).flat_map(u32::to_le_bytes).collect()
10869    }
10870
10871    #[test]
10872    fn a_section_attached_to_a_committed_file_reads_back_byte_for_byte() {
10873        let path = linked_file("attach", 64);
10874        let payload = a_key_map_payload();
10875        let table = attach(
10876            &path,
10877            "items",
10878            &[section::Attachment {
10879                kind: *section::KEY_MAP,
10880                id: 0,
10881                flags: 2,
10882                header_bytes: 40,
10883                bytes: &payload,
10884            }],
10885        )
10886        .expect("attach a key map");
10887        assert_eq!(attached(&table).len(), 1);
10888
10889        let reader = Reader::open(&path).expect("reopen after the attach");
10890        let held = attached(reader.table());
10891        assert_eq!(held.len(), 1);
10892        assert_eq!(held[0].kind, *section::KEY_MAP);
10893        assert_eq!(held[0].flags, 2, "the form a reader must not have to guess");
10894        assert_eq!(held[0].header_bytes, 40);
10895        // The generation is the one the pages were written at, not the one the attach committed at.
10896        // Attaching a section moved no row, so a section written by it is current, and a second
10897        // table added to this file later would not make it stale.
10898        assert_eq!(held[0].generation, 1);
10899        assert!(held[0].usable(reader.table().generation()));
10900        assert_eq!(reader.payload(held[0]).expect("read the payload"), payload);
10901        assert_eq!(reader.extents(held[0]).expect("extent table").len(), 1);
10902
10903        fs::remove_file(&path).expect("clean up");
10904    }
10905
10906    #[test]
10907    fn attaching_a_section_answers_every_row_exactly_as_before() {
10908        // Section 3.1 end to end, and the reason the whole layer is safe to build incrementally. A
10909        // file with a section in it and the same file without one have to agree row for row, so the
10910        // comparison is made against the answers taken before the attach rather than against a
10911        // constant somebody typed.
10912        let path = linked_file("attach_changes_nothing", 300);
10913        let before = Reader::open(&path).expect("open before");
10914        let rows = before.table().rows();
10915        let first = before.read(0, &[0]).expect("read before");
10916        let values = (0..rows).map(|at| first.value_at(at, 0)).collect::<Vec<_>>();
10917        let layout = before.layout().columns_total();
10918        drop(before);
10919
10920        let payload = a_key_map_payload();
10921        attach(
10922            &path,
10923            "items",
10924            &[section::Attachment {
10925                kind: *section::KEY_MAP,
10926                id: 0,
10927                flags: 0,
10928                header_bytes: 0,
10929                bytes: &payload,
10930            }],
10931        )
10932        .expect("attach");
10933
10934        let after = Reader::open(&path).expect("open after");
10935        assert_eq!(after.table().rows(), rows);
10936        let read = after.read(0, &[0]).expect("read after");
10937        for (at, value) in values.iter().enumerate() {
10938            assert_eq!(&read.value_at(at, 0), value, "row {at} moved");
10939        }
10940        assert_eq!(
10941            after.layout().columns_total(),
10942            layout,
10943            "an attach appends and does not rewrite a column page"
10944        );
10945
10946        fs::remove_file(&path).expect("clean up");
10947    }
10948
10949    #[test]
10950    fn a_rebuilt_section_replaces_the_one_it_supersedes() {
10951        // Rebuilding a key map has to be a write and not a question. If an attach added rather than
10952        // replaced, a table rebuilt a few times would name several maps for one column and a reader
10953        // would have to pick, which is a decision with no right answer in it.
10954        let path = linked_file("attach_twice", 32);
10955        let one = a_key_map_payload();
10956        let two = vec![7_u8; 1024];
10957        let entry = |bytes| section::Attachment {
10958            kind: *section::KEY_MAP,
10959            id: 4,
10960            flags: 1,
10961            header_bytes: 0,
10962            bytes,
10963        };
10964        attach(&path, "items", &[entry(&one)]).expect("first build");
10965        attach(&path, "items", &[entry(&two)]).expect("rebuild");
10966
10967        let reader = Reader::open(&path).expect("reopen");
10968        let held = attached(reader.table());
10969        assert_eq!(held.len(), 1, "one map per column and not one per build");
10970        assert_eq!(reader.payload(held[0]).expect("payload"), two);
10971
10972        fs::remove_file(&path).expect("clean up");
10973    }
10974
10975    #[test]
10976    fn an_attach_carries_through_a_kind_it_does_not_know() {
10977        // The first of section 3.2's three rules, at the point where it is easiest to break: a build
10978        // that rewrites a directory has to carry an entry it has no name for, or opening a file with
10979        // an older binary and attaching one section quietly deletes the work of a newer one.
10980        let path = linked_file("attach_unknown", 16);
10981        let payload = vec![3_u8; 96];
10982        attach(
10983            &path,
10984            "items",
10985            &[section::Attachment {
10986                kind: *b"RUDBZZ9\0",
10987                id: 1,
10988                flags: 0,
10989                header_bytes: 0,
10990                bytes: &payload,
10991            }],
10992        )
10993        .expect("a kind this build does not know still writes");
10994        let key_map = a_key_map_payload();
10995        attach(
10996            &path,
10997            "items",
10998            &[section::Attachment {
10999                kind: *section::KEY_MAP,
11000                id: 0,
11001                flags: 0,
11002                header_bytes: 0,
11003                bytes: &key_map,
11004            }],
11005        )
11006        .expect("attach beside it");
11007
11008        let reader = Reader::open(&path).expect("reopen");
11009        let held = attached(reader.table());
11010        assert_eq!(held.len(), 2, "the unfamiliar entry survived a directory rewrite");
11011        let unknown = held.iter().find(|one| !one.known()).expect("the unfamiliar one");
11012        assert_eq!(reader.payload(unknown).expect("its bytes are still there"), payload);
11013
11014        fs::remove_file(&path).expect("clean up");
11015    }
11016
11017    #[test]
11018    fn a_payload_of_nothing_is_a_relationship_recorded_as_not_built() {
11019        let path = linked_file("attach_not_built", 8);
11020        attach(
11021            &path,
11022            "items",
11023            &[section::Attachment {
11024                kind: *section::FORWARD_LINK,
11025                id: 2,
11026                flags: 0,
11027                header_bytes: 0,
11028                bytes: &[],
11029            }],
11030        )
11031        .expect("record a link that did not fit the budget");
11032
11033        let reader = Reader::open(&path).expect("reopen");
11034        let held = attached(reader.table());
11035        assert_eq!(held.len(), 1);
11036        assert_eq!(held[0].extents, 0);
11037        assert_eq!(held[0].extent_page, 0, "an entry that names no bytes points at none");
11038        assert!(reader.extents(held[0]).expect("no extent table").is_empty());
11039        assert!(reader.payload(held[0]).expect("no payload").is_empty());
11040
11041        fs::remove_file(&path).expect("clean up");
11042    }
11043
11044    #[test]
11045    fn a_payload_past_one_extent_is_split_and_joined_back() {
11046        // Issue #745's rule, exercised rather than argued. One byte past the bound is the smallest
11047        // payload that has to be two extents, and it is the case a split written for the common
11048        // size gets wrong.
11049        let path = linked_file("attach_two_extents", 8);
11050        let payload = vec![0x5a_u8; section::MAX_EXTENT as usize + 1];
11051        attach(
11052            &path,
11053            "items",
11054            &[section::Attachment {
11055                kind: *section::KEY_MAP,
11056                id: 0,
11057                flags: 0,
11058                header_bytes: 0,
11059                bytes: &payload,
11060            }],
11061        )
11062        .expect("attach a payload past the bound");
11063
11064        let reader = Reader::open(&path).expect("reopen");
11065        let held = attached(reader.table());
11066        let extents = reader.extents(held[0]).expect("extent table");
11067        assert_eq!(extents.len(), 2, "one byte past the bound is two extents");
11068        assert_eq!(extents[0].length, section::MAX_EXTENT);
11069        assert_eq!(extents[1].length, 1);
11070        assert_eq!(extents[1].first, u64::from(section::MAX_EXTENT));
11071        // And the extent the caller wants is readable on its own, which is the point of the split.
11072        assert_eq!(reader.extent(&extents[1]).expect("the last extent"), vec![0x5a]);
11073        assert_eq!(reader.payload(held[0]).expect("the whole payload").len(), payload.len());
11074
11075        fs::remove_file(&path).expect("clean up");
11076    }
11077
11078    #[test]
11079    fn a_torn_extent_is_refused_rather_than_decoded() {
11080        let path = linked_file("attach_torn", 8);
11081        let payload = a_key_map_payload();
11082        attach(
11083            &path,
11084            "items",
11085            &[section::Attachment {
11086                kind: *section::KEY_MAP,
11087                id: 0,
11088                flags: 0,
11089                header_bytes: 0,
11090                bytes: &payload,
11091            }],
11092        )
11093        .expect("attach");
11094
11095        let reader = Reader::open(&path).expect("reopen");
11096        let extent = reader.extents(&reader.table().sections()[0]).expect("extent table")[0];
11097        let file = OpenOptions::new().write(true).open(&path).expect("reopen to corrupt");
11098        write_at(&file, extent.offset + 7, &[0xff]).expect("flip a byte of the payload");
11099        drop(file);
11100
11101        let reader = Reader::open(&path).expect("the table still opens");
11102        let error = reader
11103            .payload(&reader.table().sections()[0])
11104            .expect_err("a corrupt payload is not handed out");
11105        assert!(error.to_string().contains("checksum"), "{error}");
11106        // And the table is still readable, which is section 3.1: a section that cannot be trusted
11107        // costs the query its shortcut and nothing else.
11108        assert_eq!(reader.read(0, &[0]).expect("the column is untouched").width(), 1);
11109
11110        fs::remove_file(&path).expect("clean up");
11111    }
11112
11113    #[test]
11114    fn attaching_to_a_file_of_the_previous_format_is_refused_rather_than_done() {
11115        // Readable is not writable. A format 22 directory has no section block, and adding one
11116        // without moving the number in the header would leave a file claiming a format it is not.
11117        let path = linked_file("attach_old_format", 8);
11118        let file = OpenOptions::new().write(true).open(&path).expect("reopen to patch");
11119        write_at(&file, 8, &22_u32.to_le_bytes()).expect("stamp the older format");
11120        drop(file);
11121
11122        let payload = a_key_map_payload();
11123        let error = attach(
11124            &path,
11125            "items",
11126            &[section::Attachment {
11127                kind: *section::KEY_MAP,
11128                id: 0,
11129                flags: 0,
11130                header_bytes: 0,
11131                bytes: &payload,
11132            }],
11133        )
11134        .expect_err("format 22 cannot gain a section");
11135        assert!(error.to_string().contains("format 22"), "{error}");
11136        assert!(Reader::open(&path).expect("and the file is untouched").table().rows() == 8);
11137
11138        fs::remove_file(&path).expect("clean up");
11139    }
11140
11141    #[test]
11142    fn a_section_header_longer_than_its_payload_is_refused_at_the_write() {
11143        let path = linked_file("attach_bad_header", 8);
11144        let error = attach(
11145            &path,
11146            "items",
11147            &[section::Attachment {
11148                kind: *section::KEY_MAP,
11149                id: 0,
11150                flags: 0,
11151                header_bytes: 40,
11152                bytes: &[1, 2, 3],
11153            }],
11154        )
11155        .expect_err("a writer's bug stops at the write");
11156        assert!(error.to_string().contains("header is longer"), "{error}");
11157
11158        fs::remove_file(&path).expect("clean up");
11159    }
11160
11161    #[test]
11162    fn attaching_to_a_name_the_file_does_not_hold_says_so() {
11163        let path = linked_file("attach_wrong_name", 8);
11164        let error = attach(&path, "orders", &[]).expect_err("no such table");
11165        assert!(error.to_string().contains("orders"), "{error}");
11166        fs::remove_file(&path).expect("clean up");
11167    }
11168
11169    #[test]
11170    fn the_planner_gets_an_exact_count_for_a_leading_value_of_an_incomplete_synopsis() {
11171        // The case a complete synopsis does not cover, and the one worth the most. 16,000 rows over
11172        // 601 distinct values, 10,000 of them holding a single value and the rest spread ten apiece
11173        // over six hundred more. The writer holds 512 values, so the list is a prefix and most of
11174        // the tail is outside it. The counts inside it are still exact, because the pass recounts
11175        // the candidates that survived it, so `id = 1` is ten thousand rows rather than the
11176        // twenty six a distinct count of 601 would divide its way to.
11177        let path = path("frequency_prefix_for_the_planner");
11178        let mut writer =
11179            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
11180                .expect("new file");
11181        let mut values = vec![Value::Integer(1); 10_000];
11182        for _ in 0..10 {
11183            values.extend((0..600).map(|tail| Value::Integer(1_000 + tail)));
11184        }
11185        // A vector holds 8,192 rows, so this goes in as several parts. The pass that takes the
11186        // synopsis walks the whole column rather than a part, so the counts are the same either way.
11187        for part in values.chunks(8_000) {
11188            let rows = Chunk::new(vec![
11189                Vector::from_values(LogicalType::Integer, part).expect("integers"),
11190            ])
11191            .expect("one column");
11192            writer.append(&rows).expect("a part");
11193        }
11194        writer.finish().expect("commit");
11195        let reader = Reader::open(&path).expect("reopen from disk");
11196        let prefix =
11197            reader.frequency_prefix(0).expect("a readable synopsis").expect("the column has one");
11198        // A prefix and not the whole column, and the writer said how many rows anything left out of
11199        // it can hold.
11200        assert_eq!(prefix.entries.len(), 512);
11201        assert_eq!(prefix.omitted_max, 10);
11202        let common = Common::new(reader);
11203        assert_eq!(common.rows(), 16_000);
11204        let column = common.column("id").expect("the file has that column");
11205        assert_eq!(
11206            common.rows_with(column, &Bound::Int(1)),
11207            Stat::exact(10_000, Provenance::FrequencySynopsis)
11208        );
11209        // In the prefix, because ties go to the smaller value and the prefix reaches 1,510.
11210        assert_eq!(
11211            common.rows_with(column, &Bound::Int(1_100)),
11212            Stat::exact(10, Provenance::FrequencySynopsis)
11213        );
11214        // Outside it, and a prefix says nothing about a value it does not list. Not zero, which is
11215        // what a complete list would say, and the file holds ten rows of this one.
11216        assert_eq!(common.rows_with(column, &Bound::Int(1_550)), Stat::Unknown);
11217        // Not in the file at all, and still nothing rather than a zero. A prefix cannot tell the
11218        // two apart, which is the whole of what it gives up.
11219        assert_eq!(common.rows_with(column, &Bound::Int(9_999)), Stat::Unknown);
11220        // What the prefix left out, which is what turns the unknown above into a number. The 512
11221        // entries account for 15,110 rows, so 890 are left for the 89 values the writer dropped,
11222        // and 890 over 89 is the ten rows each of them really holds.
11223        let remainder = common.remainder(column).expect("the list is a prefix");
11224        assert_eq!(remainder, Remainder { rows: 890, listed: 512, most: 10 });
11225        assert_eq!(remainder.rows / (601 - remainder.listed), 10);
11226        fs::remove_file(&path).expect("clean up");
11227    }
11228
11229    /// A file with no table in it is a file, and opening it says so rather than failing.
11230    #[test]
11231    fn a_file_holding_no_table_commits_and_opens_and_a_table_can_be_added_to_it() {
11232        let path = path("empty");
11233        Writer::empty(&path, &[]).expect("a file with nothing in it");
11234        let catalog = Catalog::open(&path).expect("the empty file opens");
11235        assert_eq!(catalog.len(), 0);
11236        assert!(catalog.is_empty());
11237        assert_eq!(catalog.names().count(), 0);
11238        // The next generation goes over the top of it the way it goes over any other, which is what
11239        // says this is a committed file and not a special case somebody has to know about.
11240        let mut writer =
11241            Writer::open(&path, "items", vec![Field::required("id", LogicalType::Integer)])
11242                .expect("a table goes into the empty file");
11243        writer.append(&sample_ids()).expect("rows");
11244        writer.finish().expect("commit");
11245        let catalog = Catalog::open(&path).expect("the file opens again");
11246        assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
11247        fs::remove_file(&path).expect("clean up");
11248    }
11249
11250    /// A committed table with no rows is a name the next generation takes over, and one with rows
11251    /// is a name it refuses.
11252    ///
11253    /// The refusal is what it always was and it is load bearing: carrying a table that holds rows
11254    /// forward means reading and rewriting its pages, and a writer that quietly wrote a second
11255    /// entry under the same name would leave a file with two tables a reader cannot tell apart. An
11256    /// empty one has no pages and no reader, so there is nothing to carry and nothing to lose, and
11257    /// taking its place is what lets a schema committed by an earlier session be loaded by a stream
11258    /// instead of through memory.
11259    #[test]
11260    fn a_committed_empty_table_gives_up_its_name_and_one_with_rows_does_not() {
11261        let path = path("empty-name");
11262        let field = || vec![Field::required("id", LogicalType::Integer)];
11263        Writer::create(&path, "items", field()).expect("new file").finish().expect("commit");
11264        let catalog = Catalog::open(&path).expect("the file opens");
11265        assert_eq!(catalog.rows().collect::<Vec<_>>(), vec![("items", 0)]);
11266
11267        let mut writer = Writer::open(&path, "items", field()).expect("the empty name is free");
11268        writer.append(&sample_ids()).expect("rows");
11269        writer.finish().expect("commit");
11270        let catalog = Catalog::open(&path).expect("the file opens again");
11271        // One entry and not two. The generation replaced the empty table rather than joining it.
11272        assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
11273        let held = catalog.rows().collect::<Vec<_>>();
11274        assert_eq!(held.len(), 1);
11275        assert!(held[0].1 > 0, "the rows that were appended are the ones the catalog counts");
11276
11277        // The same call against the same name now that it holds rows, which is still refused.
11278        let error = Writer::open(&path, "items", field()).expect_err("a name with rows is taken");
11279        assert!(error.to_string().contains("same name"), "{error}");
11280        fs::remove_file(&path).expect("clean up");
11281    }
11282
11283    /// A view, with everything about it that a reopened catalog has to be able to answer from.
11284    fn sample_view(name: &str) -> ViewEntry {
11285        ViewEntry {
11286            name: name.to_string(),
11287            sql: "SELECT id FROM items WHERE id > 0".to_string(),
11288            statement: format!("CREATE VIEW {name} AS SELECT id FROM items WHERE (id > 0);"),
11289            aliases: vec!["n".to_string()],
11290            columns: vec![Field::new("n", LogicalType::Integer)],
11291        }
11292    }
11293
11294    #[test]
11295    fn a_view_written_into_the_catalog_comes_back_whole() {
11296        let path = path("views");
11297        let mut writer =
11298            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
11299                .expect("new file");
11300        writer.append(&sample_ids()).expect("rows");
11301        writer.with_views(vec![sample_view("v")]).finish().expect("commit");
11302        let catalog = Catalog::open(&path).expect("reopen");
11303        assert_eq!(catalog.views().cloned().collect::<Vec<_>>(), vec![sample_view("v")]);
11304        // The tables are still there and are still read the same way, so the section on the end did
11305        // not move anything in front of it.
11306        assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
11307        fs::remove_file(&path).expect("clean up");
11308    }
11309
11310    /// A writer opened to append a table says nothing about views and must not lose them.
11311    #[test]
11312    fn appending_a_table_carries_the_views_forward() {
11313        let path = path("viewscarry");
11314        let mut writer =
11315            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
11316                .expect("new file");
11317        writer.append(&sample_ids()).expect("rows");
11318        writer.with_views(vec![sample_view("v")]).finish().expect("commit");
11319        let mut writer =
11320            Writer::open(&path, "other", vec![Field::required("id", LogicalType::Integer)])
11321                .expect("a second table");
11322        writer.append(&sample_ids()).expect("rows");
11323        writer.finish().expect("commit");
11324        let catalog = Catalog::open(&path).expect("reopen");
11325        assert_eq!(catalog.views().count(), 1);
11326        assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items", "other"]);
11327        fs::remove_file(&path).expect("clean up");
11328    }
11329
11330    /// The whole point of [`Writer::restate`]: the views change and the pages do not move.
11331    #[test]
11332    fn restating_the_views_leaves_every_table_where_it_was() {
11333        let path = path("restate");
11334        let mut writer =
11335            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
11336                .expect("new file");
11337        writer.append(&sample_ids()).expect("rows");
11338        writer.finish().expect("commit");
11339        let before = fs::metadata(&path).expect("the file is there").len();
11340        Writer::restate(&path, &[sample_view("v"), sample_view("w")]).expect("two views");
11341        let catalog = Catalog::open(&path).expect("reopen");
11342        assert_eq!(catalog.views().count(), 2);
11343        assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
11344        // A catalog on the end and nothing else, so what it grew by is the size of a catalog rather
11345        // than the size of the table.
11346        let after = fs::metadata(&path).expect("the file is there").len();
11347        assert!(after > before, "a generation was written");
11348        assert!(after - before < before, "the table was not written again");
11349        // The rows are still readable through the new generation, which is the part that would go
11350        // wrong if the catalog carried the wrong directory pointers forward.
11351        let reader = Catalog::open(&path).expect("reopen").table("items").expect("the table");
11352        assert_eq!(reader.table().rows, 3);
11353        // And a restate over a restate keeps working, because each one reads the slot that
11354        // checksummed rather than the highest number in the header.
11355        Writer::restate(&path, &[]).expect("no views at all");
11356        assert_eq!(Catalog::open(&path).expect("reopen").views().count(), 0);
11357        fs::remove_file(&path).expect("clean up");
11358    }
11359
11360    /// Two entries under one name is a catalog no lookup can answer, whichever two they are.
11361    #[test]
11362    fn a_view_named_after_a_table_is_refused_when_the_catalog_is_read() {
11363        let bytes = encode_catalog(
11364            &[Entry {
11365                name: "items".to_string(),
11366                fields: vec![Field::required("id", LogicalType::Integer)],
11367                rows: 1,
11368                directory: Page { offset: HEADER, length: 8, hash: 0 },
11369                nonzero: vec![None],
11370                aggregates: vec![None],
11371            }],
11372            &[sample_view("items")],
11373        )
11374        .expect("it encodes, because encoding does not look");
11375        let error = decode_catalog(&bytes, HEADER + 8).expect_err("and decoding does");
11376        assert!(error.to_string().contains("same name"), "{error}");
11377    }
11378
11379    #[test]
11380    fn committed_file_reopens_and_reads_only_requested_columns() {
11381        let path = path("reopen");
11382        let mut writer = Writer::create(
11383            &path,
11384            "items",
11385            vec![
11386                Field::required("id", LogicalType::Integer),
11387                Field::new("text", LogicalType::Varchar),
11388            ],
11389        )
11390        .expect("new file");
11391        writer.append(&sample()).expect("first part");
11392        writer.append(&sample()).expect("second part");
11393        writer.finish().expect("commit");
11394        let reader = Reader::open(&path).expect("reopen from disk");
11395        assert_eq!(reader.table().rows(), 6);
11396        // Two appends below the stripe bound are two parts of one stripe, which is the whole point
11397        // of the split: the directory describes the stripe and the scan still reads a part.
11398        assert_eq!(reader.table().stripes().len(), 1);
11399        assert_eq!(reader.parts(), 2);
11400        assert_eq!(reader.part_rows(0), 3);
11401        assert_eq!(reader.part_rows(1), 3);
11402        let text = reader.read(1, &[1]).expect("only text page");
11403        assert_eq!(text.width(), 1);
11404        assert_eq!(text.value_at(1, 0), Value::Null);
11405        assert_eq!(text.value_at(2, 0), Value::Varchar("long text after a slash".into()));
11406        let sparse = reader.read_sparse(1, &[1]).expect("one part without its whole page");
11407        assert_eq!(sparse.width(), 1);
11408        assert_eq!(sparse.value_at(1, 0), Value::Null);
11409        assert_eq!(sparse.value_at(2, 0), Value::Varchar("long text after a slash".into()));
11410        assert!(!reader.skips_codes(0, 1, &[0]).expect("alpha is in the stripe"));
11411        assert!(!reader.skips_codes(0, 1, &[2]).expect("long text is in the stripe"));
11412        assert!(reader.skips_codes(0, 1, &[3]).expect("unknown code is absent"));
11413        let count = reader.read(0, &[]).expect("no page is needed for count");
11414        assert_eq!(count.len(), 3);
11415        assert!(reader.skips(0, &[Probe { column: 0, op: Op::Greater, value: Bound::Int(100) }]));
11416        assert!(!reader.skips(0, &[Probe { column: 0, op: Op::Greater, value: Bound::Int(0) }]));
11417        let integers = reader.top_frequencies(0, 1).expect("valid integer synopsis").expect("kept");
11418        assert_eq!(
11419            integers,
11420            vec![(Value::Integer(-2), 2), (Value::Integer(4), 2), (Value::Integer(9), 2),]
11421        );
11422        let strings = reader.top_frequencies(1, 1).expect("valid string synopsis").expect("kept");
11423        assert_eq!(strings.len(), 3);
11424        assert!(strings.contains(&(Value::Null, 2)));
11425        assert!(strings.contains(&(Value::Varchar("alpha".into()), 2)));
11426        assert!(strings.contains(&(Value::Varchar("long text after a slash".into()), 2)));
11427        fs::remove_file(path).expect("remove scratch file");
11428    }
11429
11430    /// Two pipeline instances handing over whole runs, which is what makes the native sink safe to
11431    /// instance.
11432    ///
11433    /// The runs arrive in the order the instances finished reading them rather than in source
11434    /// order, and the second one to finish is the one that read the earlier rows. Each run is still
11435    /// a stripe of its own and the table still reads back in source order, which is the whole of
11436    /// what the writer promises about ordering.
11437    #[test]
11438    fn runs_handed_over_out_of_order_still_read_back_in_source_order() {
11439        let path = path("interleaved-runs");
11440        let mut writer =
11441            Writer::create(&path, "interleaved", vec![Field::new("v", LogicalType::BigInt)])
11442                .expect("new file");
11443        for morsel in [2_u64, 0, 3, 1] {
11444            let parts = (0..4_u64)
11445                .map(|chunk| {
11446                    let first = i64::try_from(morsel * 32 + chunk * 8).expect("small");
11447                    let values =
11448                        (0..8_i64).map(|row| Value::BigInt(first + row)).collect::<Vec<_>>();
11449                    let column =
11450                        Vector::from_values(LogicalType::BigInt, &values).expect("a column");
11451                    ((morsel, chunk), Chunk::new(vec![column]).expect("one column"))
11452                })
11453                .collect::<Vec<_>>();
11454            writer.append_stripe(parts).expect("a stripe");
11455        }
11456        writer.finish().expect("commit");
11457
11458        let reader = Reader::open(&path).expect("valid directory");
11459        assert_eq!(reader.table().stripes().len(), 4, "a run is a stripe of its own");
11460        assert_eq!(reader.table().rows(), 128);
11461        for part in 0..16_usize {
11462            let read = reader.read(part, &[0]).expect("a part back");
11463            for row in 0..8_usize {
11464                let want = i64::try_from(part * 8 + row).expect("small");
11465                assert_eq!(read.value_at(row, 0), Value::BigInt(want), "part {part} row {row}");
11466            }
11467        }
11468        fs::remove_file(path).expect("remove scratch file");
11469    }
11470
11471    /// Runs from different callers may interleave and may not overlap, and the commit is what
11472    /// catches an overlap.
11473    #[test]
11474    fn runs_that_overlap_each_other_are_refused_at_commit() {
11475        let path = path("overlapping-runs");
11476        let mut writer =
11477            Writer::create(&path, "overlapping", vec![Field::new("v", LogicalType::BigInt)])
11478                .expect("new file");
11479        let one = |order: (u64, u64)| {
11480            let column =
11481                Vector::from_values(LogicalType::BigInt, &[Value::BigInt(1)]).expect("a column");
11482            (order, Chunk::new(vec![column]).expect("one column"))
11483        };
11484        // The second run sits inside the first rather than after it, which is a thing no instance
11485        // holding its own contiguous run can produce and a thing the file cannot represent.
11486        writer.append_stripe(vec![one((0, 0)), one((0, 2))]).expect("a stripe");
11487        writer.append_stripe(vec![one((0, 1))]).expect("a stripe");
11488        let error = writer.finish().expect_err("the runs overlap");
11489        assert!(error.message().contains("source order"), "{error}");
11490        fs::remove_file(path).expect("remove scratch file");
11491    }
11492
11493    /// A stripe holds [`STRIPE_PARTS`] parts, so a run longer than that is a caller bug rather than
11494    /// something to split, and the writer says so at the door instead of quietly cutting it in two.
11495    #[test]
11496    fn a_run_longer_than_a_stripe_is_refused() {
11497        let path = path("overlong-run");
11498        let mut writer =
11499            Writer::create(&path, "overlong", vec![Field::new("v", LogicalType::BigInt)])
11500                .expect("new file");
11501        let parts = (0..=STRIPE_PARTS)
11502            .map(|at| {
11503                let column = Vector::from_values(LogicalType::BigInt, &[Value::BigInt(1)])
11504                    .expect("a column");
11505                let chunk = Chunk::new(vec![column]).expect("one column");
11506                ((0, u64::try_from(at).expect("small")), chunk)
11507            })
11508            .collect::<Vec<_>>();
11509        let error = writer.append_stripe(parts).expect_err("one part too many");
11510        assert!(error.message().contains("more parts than it holds"), "{error}");
11511        fs::remove_file(path).expect("remove scratch file");
11512    }
11513
11514    /// Parts past the stripe bound start a new stripe, and every part stays addressable on its own.
11515    ///
11516    /// This is the shape the format exists for, so both ends of the split are checked here. The
11517    /// directory holds three stripes rather than a hundred and thirty one, and a read of any one
11518    /// part still answers with that part's rows rather than with its whole stripe's.
11519    #[test]
11520    fn parts_past_the_stripe_bound_start_a_new_stripe() {
11521        let path = path("stripe-bound");
11522        let mut writer = Writer::create(
11523            &path,
11524            "items",
11525            vec![
11526                Field::required("id", LogicalType::Integer),
11527                Field::new("text", LogicalType::Varchar),
11528            ],
11529        )
11530        .expect("new file");
11531        let parts = STRIPE_PARTS * 2 + 3;
11532        for part in 0..parts {
11533            let id = part as i32;
11534            let chunk = Chunk::new(vec![
11535                Vector::from_values(
11536                    LogicalType::Integer,
11537                    &[Value::Integer(id), Value::Integer(-id)],
11538                )
11539                .expect("integers"),
11540                Vector::from_values(
11541                    LogicalType::Varchar,
11542                    &[Value::Varchar(format!("value {part}")), Value::Null],
11543                )
11544                .expect("strings"),
11545            ])
11546            .expect("matching rows");
11547            writer.append(&chunk).expect("one part");
11548        }
11549        writer.finish().expect("commit");
11550
11551        let reader = Reader::open(&path).expect("reopen from disk");
11552        assert_eq!(reader.parts(), parts);
11553        assert_eq!(reader.table().rows(), parts * 2);
11554        assert_eq!(reader.table().stripes().len(), parts.div_ceil(STRIPE_PARTS));
11555        assert_eq!(reader.table().stripes()[0].parts(), STRIPE_PARTS);
11556        assert_eq!(reader.table().stripes()[0].rows(), STRIPE_PARTS * 2);
11557        assert_eq!(reader.table().stripes()[2].parts(), 3);
11558        // Backwards on purpose. The reader keeps four stripes a column, so a scan that walks the
11559        // table the other way is what catches a cache that only ever holds what it just read.
11560        for part in (0..parts).rev() {
11561            let dense = reader.read(part, &[0, 1]).expect("a whole page read");
11562            let sparse = reader.read_sparse(part, &[0, 1]).expect("one part read");
11563            for chunk in [&dense, &sparse] {
11564                assert_eq!(chunk.len(), 2, "part {part} has its own row count");
11565                assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
11566                assert_eq!(chunk.value_at(1, 0), Value::Integer(-(part as i32)));
11567                assert_eq!(chunk.value_at(0, 1), Value::Varchar(format!("value {part}")));
11568                assert_eq!(chunk.value_at(1, 1), Value::Null);
11569            }
11570        }
11571        // The bounds are merged over the stripe, so they answer for the range the whole stripe
11572        // covers and not for the part that was asked about.
11573        let above = [Probe { column: 0, op: Op::Greater, value: Bound::Int(100) }];
11574        assert!(reader.skips(0, &above), "the first stripe stops at 63");
11575        assert!(!reader.skips(STRIPE_PARTS * 2, &above), "the third stripe reaches 130");
11576        fs::remove_file(path).expect("remove scratch file");
11577    }
11578
11579    /// A scattered value in the column that decides `WHERE UserID = ?`.
11580    fn scattered(n: i64) -> i64 {
11581        n.wrapping_mul(-7_046_029_254_386_353_131)
11582    }
11583
11584    /// A part whose sieve does not hold the constant is skipped, and a range would skip none of them.
11585    ///
11586    /// This is ClickBench query 19 in miniature. The values are spread over the whole of `BIGINT`, so
11587    /// every stripe's bounds cover nearly all of it and rule out nothing, and the part that really
11588    /// holds the value is the only one a scan has to read.
11589    #[test]
11590    fn a_part_is_skipped_when_its_sieve_does_not_hold_the_constant() {
11591        let path = path("sieve-skip");
11592        let mut writer =
11593            Writer::create(&path, "hits", vec![Field::required("id", LogicalType::BigInt)])
11594                .expect("new file");
11595        let parts = STRIPE_PARTS + 3;
11596        // Big enough that the filter is worth its bytes. A part of eight numbers packs to under a
11597        // hundred bytes and the smallest filter there is is sixty nine, so a filter over a part
11598        // that small costs about as much to read as the rows do and is no longer written.
11599        let per_part = 128;
11600        for part in 0..parts {
11601            let held: Vec<Value> = (0..per_part)
11602                .map(|row| Value::BigInt(scattered((part * per_part + row) as i64)))
11603                .collect();
11604            let chunk =
11605                Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
11606                    .expect("one column");
11607            writer.append(&chunk).expect("one part");
11608        }
11609        writer.finish().expect("commit");
11610
11611        let reader = Reader::open(&path).expect("reopen from disk");
11612        let probe = |value: i64| Probe {
11613            column: 0,
11614            op: Op::Equal,
11615            value: Bound::Int(i128::from(scattered(value))),
11616        };
11617        for wanted in [0_i64, (per_part + 1) as i64, (parts * per_part - 1) as i64] {
11618            let tests = [probe(wanted)];
11619            let kept: Vec<usize> = (0..parts).filter(|&part| !reader.skips(part, &tests)).collect();
11620            let home = wanted as usize / per_part;
11621            assert!(kept.contains(&home), "the part holding {wanted} is read");
11622            // A filter answers maybe, so a part it keeps need not hold the value. Sixty seven parts
11623            // of a hundred and twenty eight numbers each, at a dozen bits a value, is about one
11624            // stray part across the whole file and that is what this leaves room for.
11625            assert!(kept.len() <= 2, "{wanted} keeps {kept:?}, which is more than one stray part");
11626        }
11627        let absent = [probe((parts * per_part) as i64 + 1)];
11628        let kept = (0..parts).filter(|&part| !reader.skips(part, &absent)).count();
11629        assert!(kept <= 1, "{kept} parts of {parts} kept a value no part holds");
11630        // The same probes against the bounds alone, which is what this replaces. A column of
11631        // scattered numbers has a range per stripe that covers nearly the whole type.
11632        let tests = [probe(0)];
11633        assert!(
11634            reader.table().stripes().iter().all(|stripe| !stripe.zone.skips(&tests)),
11635            "the bounds rule out no stripe at all"
11636        );
11637        fs::remove_file(path).expect("remove scratch file");
11638    }
11639
11640    /// A part whose own bounds rule out an ordered comparison is skipped where the stripe's keep it.
11641    ///
11642    /// This is the shape of ClickBench 24. Each part covers a narrow stretch of the column and the
11643    /// stripe covers all sixty four of them at once, so a comparison that lands inside the stripe
11644    /// rules out none of it and rules out all but a few parts.
11645    #[test]
11646    fn a_part_is_skipped_when_its_own_bounds_rule_out_a_comparison_the_stripe_keeps() {
11647        let path = path("part-range-skip");
11648        let mut writer =
11649            Writer::create(&path, "hits", vec![Field::required("at", LogicalType::BigInt)])
11650                .expect("new file");
11651        let parts = STRIPE_PARTS + 3;
11652        let per_part = 128;
11653        for part in 0..parts {
11654            // Scattered inside the part's own band rather than a run, because a run of
11655            // consecutive numbers encodes to a stride of a few bytes and then the page of ranges
11656            // costs more than reading the column it indexes, which is the case the writer declines.
11657            let held: Vec<Value> = (0..per_part)
11658                .map(|row| {
11659                    Value::BigInt((part * 1_000) as i64 + (scattered(row as i64).rem_euclid(900)))
11660                })
11661                .collect();
11662            let chunk =
11663                Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
11664                    .expect("one column");
11665            writer.append(&chunk).expect("one part");
11666        }
11667        writer.finish().expect("commit");
11668
11669        let reader = Reader::open(&path).expect("reopen from disk");
11670        let under = [Probe { column: 0, op: Op::Less, value: Bound::Int(3_000) }];
11671        let kept: Vec<usize> = (0..parts).filter(|&part| !reader.skips(part, &under)).collect();
11672        assert_eq!(kept, vec![0, 1, 2], "only the three parts that start under three thousand");
11673        // The same question asked of the stripe alone, which is what this replaces.
11674        assert!(!reader.stripe_skips(0, &under), "the stripe reaches from zero and keeps itself");
11675        fs::remove_file(path).expect("remove scratch file");
11676    }
11677
11678    /// The other half of the same page. A part whose own bounds put every row of it inside the
11679    /// filter is waved through, so the comparison never runs on it, where the stripe's bounds reach
11680    /// across every part and can prove nothing.
11681    #[test]
11682    fn a_part_is_waved_through_when_its_own_bounds_pass_a_comparison_the_stripe_cannot() {
11683        let path = path("part-range-certain");
11684        let mut writer =
11685            Writer::create(&path, "hits", vec![Field::required("at", LogicalType::BigInt)])
11686                .expect("new file");
11687        let parts = STRIPE_PARTS + 3;
11688        let per_part = 128;
11689        for part in 0..parts {
11690            let held: Vec<Value> = (0..per_part)
11691                .map(|row| {
11692                    Value::BigInt((part * 1_000) as i64 + (scattered(row as i64).rem_euclid(900)))
11693                })
11694                .collect();
11695            let chunk =
11696                Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
11697                    .expect("one column");
11698            writer.append(&chunk).expect("one part");
11699        }
11700        writer.finish().expect("commit");
11701
11702        let reader = Reader::open(&path).expect("reopen from disk");
11703        let under = [Probe { column: 0, op: Op::Less, value: Bound::Int(3_000) }];
11704        let waved: Vec<usize> = (0..parts).filter(|&part| reader.certain(part, &under)).collect();
11705        assert_eq!(waved, vec![0, 1, 2], "the three parts that end under three thousand");
11706        // The first stripe reaches from zero to past sixty thousand, so it straddles three thousand
11707        // and settles nothing either way. The three yeses above are the parts' own ends talking.
11708        assert!(!reader.stripe_skips(0, &under), "the stripe straddles the comparison");
11709        fs::remove_file(path).expect("remove scratch file");
11710    }
11711
11712    /// The page is worth its bytes on a column with parts to tell apart and is not written on one
11713    /// that has a single part, where the stripe bounds already are the part's.
11714    #[test]
11715    fn a_stripe_of_one_part_writes_no_range_page_and_a_stripe_of_many_does() {
11716        for (parts, wanted) in [(1_usize, false), (STRIPE_PARTS, true)] {
11717            let path = path("part-range-page");
11718            let mut writer =
11719                Writer::create(&path, "hits", vec![Field::required("at", LogicalType::BigInt)])
11720                    .expect("new file");
11721            for part in 0..parts {
11722                let held: Vec<Value> = (0..128)
11723                    .map(|row| {
11724                        Value::BigInt((part * 1_000) as i64 + scattered(row as i64).rem_euclid(900))
11725                    })
11726                    .collect();
11727                let chunk = Chunk::new(vec![
11728                    Vector::from_values(LogicalType::BigInt, &held).expect("numbers"),
11729                ])
11730                .expect("one column");
11731                writer.append(&chunk).expect("one part");
11732            }
11733            writer.finish().expect("commit");
11734            let reader = Reader::open(&path).expect("reopen from disk");
11735            let bytes = reader.layout().columns[0].part_ranges;
11736            assert_eq!(bytes > 0, wanted, "{parts} parts wrote {bytes} bytes of ranges");
11737            fs::remove_file(path).expect("remove scratch file");
11738        }
11739    }
11740
11741    /// A cut down string end is still an end on the side it was, which is the only thing that keeps
11742    /// a shortened bound from turning a skip into a wrong answer.
11743    #[test]
11744    fn a_string_end_that_is_cut_down_still_covers_the_value_it_came_from() {
11745        let long = vec![b'a'; PART_BOUND_BYTES * 2];
11746        let low = shortened(Some(Bound::Bytes(long.clone())), false).expect("a low end");
11747        let high = shortened(Some(Bound::Bytes(long.clone())), true).expect("a high end");
11748        let Bound::Bytes(low) = low else { panic!("a string stays a string") };
11749        let Bound::Bytes(high) = high else { panic!("a string stays a string") };
11750        assert!(low.len() <= PART_BOUND_BYTES && high.len() <= PART_BOUND_BYTES);
11751        assert!(low.as_slice() <= long.as_slice(), "the low end is at or under the value");
11752        assert!(high.as_slice() >= long.as_slice(), "the high end is at or over the value");
11753    }
11754
11755    /// A string of nothing but the largest byte has no prefix that can be stepped up, so the high
11756    /// end is given up rather than claimed too small. No end keeps the part, which is always safe.
11757    #[test]
11758    fn a_string_end_with_no_room_to_step_up_gives_up_the_bound() {
11759        let long = vec![u8::MAX; PART_BOUND_BYTES * 2];
11760        assert_eq!(shortened(Some(Bound::Bytes(long.clone())), true), None);
11761        let low = shortened(Some(Bound::Bytes(long)), false).expect("a low end is still a prefix");
11762        assert_eq!(low, Bound::Bytes(vec![u8::MAX; PART_BOUND_BYTES]));
11763    }
11764
11765    /// What a column is stored as, asked of two files holding the same rows in a different order.
11766    ///
11767    /// This is the question the report exists to answer and it is the one the directory cannot. The
11768    /// two files have the same rows, the same schema and the same number of parts, and the column
11769    /// comes out four times smaller in one of them, because ascending keys delta encode to a few
11770    /// bits a row and shuffled ones do not. Nothing about the file's shape says so. The page header
11771    /// says so, and reading it is what this does.
11772    ///
11773    /// It is q18 on TPC-H in miniature: clustering lineitem by ship date leaves `l_orderkey`
11774    /// ascending inside a partition but sparse, its deltas go from six bits to twelve, and the scan
11775    /// pays for the wider ones.
11776    #[test]
11777    fn what_a_column_is_stored_as_follows_the_order_the_rows_were_written_in() {
11778        let parts = 4;
11779        let per_part = 1024;
11780        let rows = parts * per_part;
11781        let written = |name: &str, keys: &[i64]| {
11782            let path = path(name);
11783            let fields = vec![Field::required("key", LogicalType::BigInt)];
11784            let mut writer = Writer::create(&path, "keys", fields).expect("new file");
11785            for part in 0..parts {
11786                let values: Vec<Value> = keys[part * per_part..(part + 1) * per_part]
11787                    .iter()
11788                    .map(|key| Value::BigInt(*key))
11789                    .collect();
11790                let chunk = Chunk::new(vec![
11791                    Vector::from_values(LogicalType::BigInt, &values).expect("numbers"),
11792                ])
11793                .expect("one column");
11794                writer.append(&chunk).expect("one part");
11795            }
11796            writer.finish().expect("commit");
11797            path
11798        };
11799        // Ascending with a small irregular step, which is what a key column in arrival order looks
11800        // like: an order has one to seven line items, so the key repeats and then moves on by one.
11801        let climbing = |step: &dyn Fn(usize) -> i64| {
11802            let mut key = 0;
11803            (0..rows)
11804                .map(|row| {
11805                    key += step(row);
11806                    key
11807                })
11808                .collect::<Vec<i64>>()
11809        };
11810        let ascending = climbing(&|row| (row % 3) as i64);
11811        // The same rows in the same direction over a range a thousand times wider, which is what a
11812        // partition of a clustered table holds: still ascending, and far enough apart that the
11813        // deltas no longer fit in a handful of bits.
11814        let sparse = climbing(&|row| ((row * 2_654_435_761) % 4096) as i64);
11815        let near_path = written("stored-near", &ascending);
11816        let far_path = written("stored-far", &sparse);
11817
11818        let one = Reader::open(&near_path).expect("reopen from disk");
11819        let other = Reader::open(&far_path).expect("reopen from disk");
11820        let near = one.stored(0).expect("the column is stored");
11821        let far = other.stored(0).expect("the column is stored");
11822        assert_eq!(near.len(), parts, "one row per part");
11823        assert_eq!(far.len(), parts);
11824        // The bytes are the same bytes the directory totals, which is the check that this is
11825        // reading the pages the file really holds rather than some other pages.
11826        let total = |stored: &[StoredPart]| stored.iter().map(|part| part.bytes).sum::<u64>();
11827        assert_eq!(total(&near), one.layout().columns[0].pages);
11828        assert_eq!(total(&far), other.layout().columns[0].pages);
11829        assert!(
11830            total(&near) * 2 < total(&far),
11831            "the sparse keys cost more, {} against {}",
11832            total(&far),
11833            total(&near)
11834        );
11835        // Every part accounted for, in order, with the row it starts at following the one before.
11836        for (at, part) in near.iter().enumerate() {
11837            assert_eq!(part.part, at);
11838            assert_eq!(part.row, at * per_part);
11839            assert_eq!(part.rows, per_part);
11840            let held = &ascending[at * per_part..(at + 1) * per_part];
11841            assert_eq!(part.low, Some(Value::BigInt(held[0])));
11842            assert_eq!(part.high, Some(Value::BigInt(held[per_part - 1])));
11843            assert_eq!(part.nulls, Some(0));
11844        }
11845        // And the encoding is a line of text that names what the encoder chose, which is the whole
11846        // point. Both are a cascade over deltas and the widths inside them are what differ.
11847        assert!(near[0].encoding.contains("DELTA"), "{}", near[0].encoding);
11848        assert!(far[0].encoding.contains("DELTA"), "{}", far[0].encoding);
11849        assert_ne!(near[0].encoding, far[0].encoding);
11850        fs::remove_file(near_path).expect("remove scratch file");
11851        fs::remove_file(far_path).expect("remove scratch file");
11852    }
11853
11854    /// A sieve bigger than the part it indexes is not written, and one smaller than it still is.
11855    ///
11856    /// Both columns hold values spread over the whole of `BIGINT`, so neither gets a bitmap and both
11857    /// reach the filter. They differ in what the part costs to read. `spread` is a thousand distinct
11858    /// numbers and packs to eight kilobytes, so a filter of about thirteen hundred bytes is a good
11859    /// trade. `repeated` is the same thousand rows over four numbers in runs and encodes to
11860    /// almost nothing, but the filter is sized for the rows rather than the values it turns out to
11861    /// hold, so it comes out larger than the data. Reading it to decide whether to read the part spends more than
11862    /// the part, every time, and that is the case this drops.
11863    #[test]
11864    fn a_sieve_larger_than_the_part_it_indexes_is_not_written() {
11865        let path = path("sieve-pays");
11866        let fields = vec![
11867            Field::required("spread", LogicalType::BigInt),
11868            Field::required("repeated", LogicalType::BigInt),
11869        ];
11870        let mut writer = Writer::create(&path, "hits", fields).expect("new file");
11871        let parts = 3;
11872        let per_part = 1024;
11873        for part in 0..parts {
11874            let base = (part * per_part) as i64;
11875            let spread: Vec<Value> =
11876                (0..per_part).map(|row| Value::BigInt(scattered(base + row as i64))).collect();
11877            let repeated: Vec<Value> =
11878                (0..per_part).map(|row| Value::BigInt(scattered((row / 256) as i64))).collect();
11879            let chunk = Chunk::new(vec![
11880                Vector::from_values(LogicalType::BigInt, &spread).expect("numbers"),
11881                Vector::from_values(LogicalType::BigInt, &repeated).expect("numbers"),
11882            ])
11883            .expect("two columns");
11884            writer.append(&chunk).expect("one part");
11885        }
11886        writer.finish().expect("commit");
11887
11888        let reader = Reader::open(&path).expect("reopen from disk");
11889        let layout = reader.layout();
11890        let spread = &layout.columns[0];
11891        let repeated = &layout.columns[1];
11892        assert!(spread.sieves > 0, "a column whose parts are worth a filter keeps one");
11893        assert_eq!(
11894            repeated.sieves, 0,
11895            "a column whose filter costs more than its parts keeps none"
11896        );
11897        // Per part this is the rule itself, so it holds over the column as well: a part without a
11898        // sieve adds to one side of this and to nothing on the other.
11899        for column in &layout.columns {
11900            assert!(
11901                column.sieves < column.pages,
11902                "{} spends {} on sieves over {} of data",
11903                column.name,
11904                column.sieves,
11905                column.pages
11906            );
11907        }
11908        // The filter that was kept still does what it is for.
11909        let absent = [Probe {
11910            column: 0,
11911            op: Op::Equal,
11912            value: Bound::Int(i128::from(scattered((parts * per_part) as i64 + 1))),
11913        }];
11914        assert!((0..parts).all(|part| reader.skips(part, &absent)), "no part holds it");
11915        fs::remove_file(path).expect("remove scratch file");
11916    }
11917
11918    /// A damaged sieve page is a part that gets read, not a query that fails.
11919    ///
11920    /// A sieve is an index over rows that are still there and still correct, so losing one costs
11921    /// time and costs no answers. That is the opposite of the membership index beside it, which is
11922    /// the only thing standing between a string page and a wrong answer.
11923    #[test]
11924    fn a_damaged_sieve_page_is_read_through_rather_than_refused() {
11925        let path = path("sieve-damaged");
11926        let mut writer =
11927            Writer::create(&path, "hits", vec![Field::required("id", LogicalType::BigInt)])
11928                .expect("new file");
11929        let rows = 128;
11930        let held: Vec<Value> = (0..rows).map(|row| Value::BigInt(scattered(row))).collect();
11931        let chunk =
11932            Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
11933                .expect("one column");
11934        writer.append(&chunk).expect("one part");
11935        writer.finish().expect("commit");
11936
11937        let page = Reader::open(&path).expect("reopen").table.stripes[0]
11938            .sieves
11939            .get(0)
11940            .expect("a sieve page");
11941        let mut file = OpenOptions::new().write(true).open(&path).expect("open the sieve page");
11942        file.seek(SeekFrom::Start(page.offset + u64::from(page.length) - 1)).expect("seek");
11943        file.write_all(&[0xff]).expect("damage one byte");
11944        drop(file);
11945
11946        let reader = Reader::open(&path).expect("reopen the damaged file");
11947        let absent =
11948            [Probe { column: 0, op: Op::Equal, value: Bound::Int(i128::from(scattered(99))) }];
11949        assert!(!reader.skips(0, &absent), "a sieve that cannot be read skips nothing");
11950        assert_eq!(
11951            reader.read(0, &[0]).expect("the rows are untouched").len(),
11952            usize::try_from(rows).expect("a small count")
11953        );
11954        fs::remove_file(path).expect("remove scratch file");
11955    }
11956
11957    /// Eight workers over one stripe read it once between them.
11958    ///
11959    /// This is the shape a scan actually has. Parts are handed out in order, so every worker on a
11960    /// column crosses into a stripe within a few parts of the others, and before [`Reader::held`]
11961    /// started sharing the read every one of them read the whole page. On the full ClickBench file
11962    /// that was a `MIN(EventDate), MAX(EventDate)` moving 3.2 GB off the disk to look at 400 MB of
11963    /// column, which is most of what a first touch costs.
11964    ///
11965    /// The workers that lose the race still answer, out of the part reads they do instead, which is
11966    /// what the values below are checking.
11967    #[test]
11968    fn workers_that_want_the_same_stripe_read_it_once() {
11969        let path = path("single-flight");
11970        let mut writer =
11971            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
11972                .expect("new file");
11973        for part in 0..STRIPE_PARTS {
11974            let id = part as i32;
11975            let chunk = Chunk::new(vec![
11976                Vector::from_values(
11977                    LogicalType::Integer,
11978                    &[Value::Integer(id), Value::Integer(-id)],
11979                )
11980                .expect("integers"),
11981            ])
11982            .expect("matching rows");
11983            writer.append(&chunk).expect("one part");
11984        }
11985        writer.finish().expect("commit");
11986
11987        let reader = Reader::open(&path).expect("reopen from disk");
11988        assert_eq!(reader.table().stripes().len(), 1, "one stripe is the point of the test");
11989        let barrier = std::sync::Barrier::new(8);
11990        std::thread::scope(|scope| {
11991            for worker in 0..8 {
11992                let reader = &reader;
11993                let barrier = &barrier;
11994                scope.spawn(move || {
11995                    barrier.wait();
11996                    for part in (worker..STRIPE_PARTS).step_by(8) {
11997                        let chunk = reader.read(part, &[0]).expect("a whole page read");
11998                        assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
11999                        assert_eq!(chunk.value_at(1, 0), Value::Integer(-(part as i32)));
12000                    }
12001                });
12002            }
12003        });
12004        assert_eq!(reader.pages.load(Atomic::Relaxed), 1, "one stripe, one page read, whoever won");
12005        fs::remove_file(path).expect("remove scratch file");
12006    }
12007
12008    /// Opening a file reads the header and the directory, and nothing that depends on the rows.
12009    ///
12010    /// `spec/stats/04-in-memory.md` section 4.2. There are no statistics in the file yet, so this
12011    /// holds today by not having anything to load, and that is exactly why it is worth pinning now.
12012    /// The change that breaks it is the reasonable looking one: summaries are a few hundred bytes,
12013    /// the next query will want them, so read them on the way past. A process that opened the
12014    /// database to run one trivial query pays for all of it and gets nothing.
12015    ///
12016    /// Two files of the same shape and a thousand times the rows in one of them, opened, and the
12017    /// two openings cost the same. The stripe count is held equal so that the directory is the same
12018    /// size in both, which leaves the rows as the only thing that changed. Anything read out of the
12019    /// data would show up here.
12020    #[test]
12021    fn opening_costs_the_same_over_a_thousand_times_the_rows() {
12022        let opened = |label: &str, rows_per_part: i32| {
12023            let path = path(label);
12024            let mut writer =
12025                Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
12026                    .expect("new file");
12027            for part in 0..STRIPE_PARTS * 3 {
12028                // Scrambled rather than sequential, so that the fat file is actually fatter. A run
12029                // of consecutive integers encodes to almost nothing and would leave the two files
12030                // the same size, which would make this test pass for the wrong reason.
12031                let values = (0..rows_per_part)
12032                    .map(|row| {
12033                        Value::Integer((part as i32 * rows_per_part + row).wrapping_mul(2_654_435))
12034                    })
12035                    .collect::<Vec<_>>();
12036                let chunk = Chunk::new(vec![
12037                    Vector::from_values(LogicalType::Integer, &values).expect("integers"),
12038                ])
12039                .expect("matching rows");
12040                writer.append(&chunk).expect("one part");
12041            }
12042            writer.finish().expect("commit");
12043            let reader = Reader::open(&path).expect("reopen from disk");
12044            let size = fs::metadata(&path).expect("the file is there").len();
12045            let out = (reader.reads(), reader.table().stripes().len(), size);
12046            fs::remove_file(path).expect("remove scratch file");
12047            out
12048        };
12049
12050        let (thin, thin_stripes, thin_size) = opened("open-thin", 1);
12051        let (fat, fat_stripes, fat_size) = opened("open-fat", 1000);
12052        assert_eq!(
12053            thin_stripes, fat_stripes,
12054            "the same stripe count is what makes this a fair ask"
12055        );
12056        assert!(
12057            fat_size > thin_size * 50,
12058            "the fat file has to actually be larger, and it is {fat_size} against {thin_size}"
12059        );
12060
12061        assert_eq!(thin.opening.reads, fat.opening.reads, "the same reads either way");
12062        assert_eq!(thin.pages, 0, "opening read a page");
12063        assert_eq!(fat.pages, 0, "opening read a page");
12064        assert_eq!(thin.indexes, 0, "opening read an index");
12065        assert_eq!(fat.indexes, 0, "opening read an index");
12066        // Not exactly equal, because a directory holds offsets and a larger file has larger ones,
12067        // and a handful of bytes of varint is not somebody loading statistics. A factor is.
12068        assert!(
12069            fat.opening.bytes < thin.opening.bytes * 2,
12070            "opening the thin file read {} bytes and the fat one read {}",
12071            thin.opening.bytes,
12072            fat.opening.bytes
12073        );
12074    }
12075
12076    /// The reads a file costs to open are fixed by its shape and not by what ran before.
12077    ///
12078    /// `spec/stats/04-in-memory.md` section 4.3, which is the rule that keeps a plan reproducible:
12079    /// the plan is a function of the data, the generation and the settings, and never of what
12080    /// happened to be in cache. Opening the same file twice in the same process has to cost the
12081    /// same, because a second open that read less would be an open that was about to plan
12082    /// differently.
12083    #[test]
12084    fn two_opens_of_one_file_cost_the_same_and_the_second_is_not_cheaper() {
12085        let path = path("open-twice");
12086        let mut writer =
12087            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
12088                .expect("new file");
12089        for part in 0..STRIPE_PARTS * 3 {
12090            let chunk = Chunk::new(vec![
12091                Vector::from_values(LogicalType::Integer, &[Value::Integer(part as i32)])
12092                    .expect("integers"),
12093            ])
12094            .expect("matching rows");
12095            writer.append(&chunk).expect("one part");
12096        }
12097        writer.finish().expect("commit");
12098
12099        let first = Reader::open(&path).expect("open");
12100        // A whole scan in between, so the operating system's page cache is as warm as it gets and
12101        // anything that consulted it would show up in the second open.
12102        for part in 0..first.parts() {
12103            first.read(part, &[0]).expect("a part");
12104        }
12105        assert!(first.reads().pages > 0, "the scan has to have read something");
12106        let second = Reader::open(&path).expect("open again");
12107
12108        assert_eq!(first.reads().opening, second.reads().opening);
12109        assert_eq!(
12110            second.reads().pages,
12111            0,
12112            "the second open read a page off the back of the first"
12113        );
12114        assert_eq!(second.reads().indexes, 0, "the second open read an index it inherited");
12115        fs::remove_file(path).expect("remove scratch file");
12116    }
12117
12118    /// A scan reads a stripe's index once for the whole scan, not once per part that misses.
12119    ///
12120    /// The page cache holds four stripes and an index used to ride inside it, so a table with more
12121    /// stripes than that read the index again every time a stripe came back around. The index is a
12122    /// few hundred bytes and the page is a quarter of a megabyte, which is why they are now under
12123    /// different budgets. This is the test that keeps them there, since the saving is small enough
12124    /// that nothing in a benchmark would notice it going away again.
12125    #[test]
12126    fn an_index_is_read_once_per_stripe_however_often_the_page_is_evicted() {
12127        let path = path("index-cache");
12128        let mut writer =
12129            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
12130                .expect("new file");
12131        let parts = STRIPE_PARTS * (CACHED_STRIPES_PER_COLUMN + 2);
12132        for part in 0..parts {
12133            let id = part as i32;
12134            let chunk = Chunk::new(vec![
12135                Vector::from_values(LogicalType::Integer, &[Value::Integer(id)]).expect("integers"),
12136            ])
12137            .expect("matching rows");
12138            writer.append(&chunk).expect("one part");
12139        }
12140        writer.finish().expect("commit");
12141
12142        let reader = Reader::open(&path).expect("reopen from disk");
12143        let stripes = reader.table().stripes().len();
12144        assert!(stripes > CACHED_STRIPES_PER_COLUMN, "the page cache has to be too small for this");
12145        // Twice over, so that the second pass finds every page evicted and every index kept.
12146        for _ in 0..2 {
12147            for part in 0..parts {
12148                let chunk = reader.read(part, &[0]).expect("a part");
12149                assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
12150            }
12151        }
12152        assert_eq!(reader.indexes.load(Atomic::Relaxed), stripes, "one index read per stripe");
12153        assert!(
12154            reader.pages.load(Atomic::Relaxed) > stripes,
12155            "the pages are the ones that get read again, which is what makes the index count mean \
12156             something"
12157        );
12158        fs::remove_file(path).expect("remove scratch file");
12159    }
12160
12161    /// A page stays in memory from one scan to the next while the pool has room for it, and a
12162    /// table that is being read takes room from one that is not, down to the floor and no further.
12163    ///
12164    /// This is what the pool is for. Each reader lives as long as its database, so a second query
12165    /// over the same table should find every page it read the first time, and before the pool it
12166    /// found four stripes a column and read the rest off the file again.
12167    #[test]
12168    fn a_pool_keeps_pages_between_scans_and_gives_them_to_the_table_being_read() {
12169        let path = path("page-pool");
12170        let parts = STRIPE_PARTS * (CACHED_STRIPES_PER_COLUMN + 2);
12171        let fields = || vec![Field::required("id", LogicalType::Integer)];
12172        let mut writer = Writer::create(&path, "a", fields()).expect("new file");
12173        for table in ["a", "b"] {
12174            if table == "b" {
12175                writer = writer.next("b".to_string(), fields()).expect("a second table");
12176            }
12177            for part in 0..parts {
12178                let chunk = Chunk::new(vec![
12179                    Vector::from_values(LogicalType::Integer, &[Value::Integer(part as i32)])
12180                        .expect("integers"),
12181                ])
12182                .expect("matching rows");
12183                writer.append(&chunk).expect("one part");
12184            }
12185        }
12186        writer.finish().expect("commit");
12187
12188        let pool = PagePool::new(usize::MAX);
12189        let catalog = Catalog::open_in(&path, &pool).expect("the file opens");
12190        let (a, b) = (catalog.table("a").expect("a"), catalog.table("b").expect("b"));
12191        let stripes = a.table().stripes().len();
12192        assert!(stripes > CACHED_STRIPES_PER_COLUMN, "the floor has to be smaller than a table");
12193        let scan = |reader: &Reader| {
12194            for part in 0..parts {
12195                let chunk = reader.read(part, &[0]).expect("a part");
12196                assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
12197            }
12198        };
12199        scan(&a);
12200        scan(&a);
12201        assert_eq!(a.pages.load(Atomic::Relaxed), stripes, "the second scan reads nothing");
12202        let one = pool.bytes();
12203        assert!(one > 0, "the pool counts what the reader holds");
12204
12205        // Room for one table. Reading the other takes the first one's pages down to its floor.
12206        pool.budget.store(one, Atomic::Relaxed);
12207        scan(&b);
12208        assert_eq!(b.pages.load(Atomic::Relaxed), stripes, "a page is never let go while in use");
12209        assert_eq!(a.cache.held[0].load(Atomic::Relaxed), CACHED_STRIPES_PER_COLUMN);
12210        let held = a.cache.columns[0].lock().expect("the column").pages.iter().flatten().count();
12211        assert_eq!(held, CACHED_STRIPES_PER_COLUMN, "the count and the slots agree");
12212
12213        // A reader that goes takes its pages out of the count with it.
12214        drop((a, b, catalog));
12215        let c = Catalog::open_in(&path, &pool).expect("again").table("a").expect("a");
12216        scan(&c);
12217        assert!(pool.bytes() <= one, "only what the live reader holds is counted");
12218        fs::remove_file(path).expect("remove scratch file");
12219    }
12220
12221    /// A worker per stripe reads its stripe once, once the cache has been told how many there are.
12222    ///
12223    /// This is the shape a scan has when it hands out a whole stripe per morsel rather than a part.
12224    /// Nobody races for a page any more, but every worker holds a different one for the length of a
12225    /// stripe, so a cache that keeps four pages while eight workers are in eight stripes evicts
12226    /// every one of them before its owner has finished with it, and the owner reads a quarter of a
12227    /// megabyte again for the next part. The barrier is what makes that certain rather than likely:
12228    /// without it a worker can run a whole stripe before the next one starts and never collide.
12229    #[test]
12230    fn a_worker_per_stripe_reads_its_page_once_when_the_cache_was_told_to_expect_it() {
12231        let workers = CACHED_STRIPES_PER_COLUMN + 4;
12232        let path = path("stripe-per-worker");
12233        let mut writer =
12234            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
12235                .expect("new file");
12236        for part in 0..STRIPE_PARTS * workers {
12237            let chunk = Chunk::new(vec![
12238                Vector::from_values(LogicalType::Integer, &[Value::Integer(part as i32)])
12239                    .expect("integers"),
12240            ])
12241            .expect("matching rows");
12242            writer.append(&chunk).expect("one part");
12243        }
12244        writer.finish().expect("commit");
12245
12246        let read = |told: bool| {
12247            let reader = Reader::open(&path).expect("reopen from disk");
12248            assert_eq!(reader.table().stripes().len(), workers, "a stripe per worker");
12249            if told {
12250                reader.keep_stripes(workers);
12251            }
12252            let barrier = std::sync::Barrier::new(workers);
12253            std::thread::scope(|scope| {
12254                for (worker, run) in reader.stripe_parts().into_iter().enumerate() {
12255                    let reader = &reader;
12256                    let barrier = &barrier;
12257                    scope.spawn(move || {
12258                        for part in run {
12259                            barrier.wait();
12260                            let chunk = reader.read(part, &[0]).expect("a part of my own stripe");
12261                            assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
12262                        }
12263                        assert!(worker < workers);
12264                    });
12265                }
12266            });
12267            reader.pages.load(Atomic::Relaxed)
12268        };
12269
12270        assert_eq!(read(true), workers, "one page read per stripe and no more");
12271        assert!(read(false) > workers, "a cache that small is read again on every part");
12272        fs::remove_file(path).expect("remove scratch file");
12273    }
12274
12275    /// A damaged index page is caught before anything decodes a part out of it.
12276    ///
12277    /// The index is the one structure a reader trusts to find bytes with, so it carries a checksum
12278    /// per column section rather than one for the page, and this is what says that check runs.
12279    #[test]
12280    fn a_damaged_index_page_is_an_error() {
12281        let path = path("damaged-index");
12282        let mut writer =
12283            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
12284                .expect("new file");
12285        writer.append(&sample_ids()).expect("first part");
12286        writer.append(&sample_ids()).expect("second part");
12287        writer.finish().expect("commit");
12288
12289        let reader = Reader::open(&path).expect("valid directory");
12290        let index = reader.table.stripes[0].index;
12291        let mut byte = [0; 1];
12292        read_at(&reader.file, index.offset, &mut byte).expect("the first part length");
12293        let mut file = OpenOptions::new().write(true).open(&path).expect("open index page");
12294        file.seek(SeekFrom::Start(index.offset)).expect("index start");
12295        file.write_all(&[!byte[0]]).expect("damage the first part length");
12296        let error = reader.read(1, &[0]).expect_err("a damaged index must not be used");
12297        assert!(error.message().contains("index page section checksum differs"), "{error}");
12298        fs::remove_file(path).expect("remove scratch file");
12299    }
12300
12301    /// Every integer width the format knows about, written and read back.
12302    ///
12303    /// The unsigned ones are the reason ClickBench can be stored at all: `hits` types `EventDate`
12304    /// as `USMALLINT`, and one unsupported column meant the whole table was refused. The extremes
12305    /// are in here on purpose, because a width that round trips through the wrong signedness only
12306    /// goes wrong at the end of its range.
12307    #[test]
12308    fn every_integer_width_round_trips_through_a_page() {
12309        let path = path("integer-widths");
12310        let columns = [
12311            (LogicalType::TinyInt, vec![Value::TinyInt(i8::MIN), Value::TinyInt(i8::MAX)]),
12312            (LogicalType::UTinyInt, vec![Value::UTinyInt(0), Value::UTinyInt(u8::MAX)]),
12313            (LogicalType::SmallInt, vec![Value::SmallInt(i16::MIN), Value::SmallInt(i16::MAX)]),
12314            (LogicalType::USmallInt, vec![Value::USmallInt(0), Value::USmallInt(u16::MAX)]),
12315            (LogicalType::Integer, vec![Value::Integer(i32::MIN), Value::Integer(i32::MAX)]),
12316            (LogicalType::UInteger, vec![Value::UInteger(0), Value::UInteger(u32::MAX)]),
12317            (LogicalType::BigInt, vec![Value::BigInt(i64::MIN), Value::BigInt(i64::MAX)]),
12318            (LogicalType::UBigInt, vec![Value::UBigInt(0), Value::UBigInt(u64::MAX)]),
12319        ];
12320        let fields = columns
12321            .iter()
12322            .enumerate()
12323            .map(|(at, (ty, _))| Field::required(format!("c{at}"), ty.clone()))
12324            .collect::<Vec<_>>();
12325        let vectors = columns
12326            .iter()
12327            .map(|(ty, values)| Vector::from_values(ty.clone(), values).expect("a vector"))
12328            .collect::<Vec<_>>();
12329        let mut writer = Writer::create(&path, "widths", fields).expect("new file");
12330        writer.append(&Chunk::new(vectors).expect("matching rows")).expect("one stripe");
12331        writer.finish().expect("commit");
12332
12333        let reader = Reader::open(&path).expect("reopen from disk");
12334        let wanted = (0..columns.len()).collect::<Vec<_>>();
12335        let read = reader.read(0, &wanted).expect("every column");
12336        assert_eq!(read.len(), 2);
12337        // row at a time: each column has its own type and its own pair of extremes.
12338        for (at, (ty, values)) in columns.iter().enumerate() {
12339            assert_eq!(read.value_at(0, at), values[0], "the low end of {ty}");
12340            assert_eq!(read.value_at(1, at), values[1], "the high end of {ty}");
12341        }
12342        fs::remove_file(path).expect("remove scratch file");
12343    }
12344
12345    /// The rest of the fixed width types, and the byte strings, written and read back.
12346    ///
12347    /// The extremes again, and for a float that means more than the ends of the range. Negative
12348    /// zero and a NaN are the two values that go through an encoder unnoticed and come back
12349    /// different, so they are here on purpose, and the NaN is compared by its bits rather than by
12350    /// `==`, which a NaN fails against itself.
12351    ///
12352    /// A blob is here beside them because it is the same round trip asked of bytes that are not
12353    /// text. The value in it is not UTF-8, so a path that reads a payload as a string on the way
12354    /// past turns this test red rather than turning a user's column into nulls.
12355    #[test]
12356    fn every_other_type_the_format_knows_round_trips_through_a_page() {
12357        let path = path("other-types");
12358        let columns = [
12359            (LogicalType::Float, vec![Value::Float(f32::MIN), Value::Float(-0.0)]),
12360            (LogicalType::Double, vec![Value::Double(f64::MIN), Value::Double(f64::MAX)]),
12361            (LogicalType::HugeInt, vec![Value::HugeInt(i128::MIN), Value::HugeInt(i128::MAX)]),
12362            (LogicalType::UHugeInt, vec![Value::UHugeInt(0), Value::UHugeInt(u128::MAX)]),
12363            (LogicalType::Time, vec![Value::Time(0), Value::Time(86_399_999_999)]),
12364            (LogicalType::TimeTz, vec![Value::TimeTz(-50_400_000_000), Value::TimeTz(0)]),
12365            (
12366                LogicalType::TimestampTz,
12367                vec![Value::TimestampTz(i64::MIN + 1), Value::TimestampTz(i64::MAX)],
12368            ),
12369            (
12370                LogicalType::Interval,
12371                vec![
12372                    Value::Interval { months: i32::MIN, days: i32::MAX, micros: i64::MIN },
12373                    Value::Interval { months: 13, days: -1, micros: 1 },
12374                ],
12375            ),
12376            (
12377                LogicalType::Blob,
12378                vec![Value::Blob(vec![0, 0xff, 0x80, 0xfe]), Value::Blob(Vec::new())],
12379            ),
12380        ];
12381        let fields = columns
12382            .iter()
12383            .enumerate()
12384            .map(|(at, (ty, _))| Field::required(format!("c{at}"), ty.clone()))
12385            .collect::<Vec<_>>();
12386        let vectors = columns
12387            .iter()
12388            .map(|(ty, values)| Vector::from_values(ty.clone(), values).expect("a vector"))
12389            .collect::<Vec<_>>();
12390        let mut writer = Writer::create(&path, "others", fields).expect("new file");
12391        writer.append(&Chunk::new(vectors).expect("matching rows")).expect("one stripe");
12392        writer.finish().expect("commit");
12393
12394        let reader = Reader::open(&path).expect("reopen from disk");
12395        let wanted = (0..columns.len()).collect::<Vec<_>>();
12396        let read = reader.read(0, &wanted).expect("every column");
12397        assert_eq!(read.len(), 2);
12398        for (at, (ty, values)) in columns.iter().enumerate() {
12399            assert_eq!(read.value_at(0, at), values[0], "the low end of {ty}");
12400            assert_eq!(read.value_at(1, at), values[1], "the high end of {ty}");
12401        }
12402        // A float keeps its sign through a zero, which `==` says nothing about because negative
12403        // zero and zero compare equal.
12404        let Value::Float(zero) = read.value_at(1, 0) else { panic!("a float stays a float") };
12405        assert!(zero.is_sign_negative(), "a negative zero came back as {zero}");
12406
12407        fs::remove_file(path).expect("remove scratch file");
12408    }
12409
12410    /// A NaN is still a NaN after a trip through a page.
12411    ///
12412    /// Apart from the other floats because it cannot be asserted the same way. A NaN is not equal
12413    /// to itself, so a comparison against the value that was written passes for every NaN and for
12414    /// nothing else, which is the one assertion that would not catch a page that lost it.
12415    #[test]
12416    fn a_nan_survives_being_written_down() {
12417        let path = path("nan");
12418        let nan = Vector::from_values(LogicalType::Double, &[Value::Double(f64::NAN)])
12419            .expect("a NaN vector");
12420        let mut writer =
12421            Writer::create(&path, "nan", vec![Field::required("d", LogicalType::Double)])
12422                .expect("new file");
12423        writer.append(&Chunk::new(vec![nan]).expect("one column")).expect("one stripe");
12424        writer.finish().expect("commit");
12425        let read = Reader::open(&path).expect("reopen").read(0, &[0]).expect("the column");
12426        let Value::Double(back) = read.value_at(0, 0) else { panic!("a double stays a double") };
12427        assert!(back.is_nan(), "a NaN came back as {back}");
12428        fs::remove_file(path).expect("remove scratch file");
12429    }
12430
12431    /// A uuid and a bit string, which have no `Value` arm of their own and are checked as bits.
12432    ///
12433    /// A uuid is the 128 bit lane and a bit string is bytes, and neither of them reads back as
12434    /// anything in `Value` today, so asking for a value here would compare two nulls and pass
12435    /// whatever the file held. The data underneath is what the storage promise is about, so that is
12436    /// what this reads.
12437    #[test]
12438    fn a_uuid_and_a_bit_string_come_back_as_the_bits_that_went_in() {
12439        let path = path("uuid-and-bit");
12440        let uuids = vec![0_i128, i128::MIN, -1];
12441        let mut bits = StringColumn::new();
12442        for value in [&b"\x02\xff"[..], &b""[..], &b"\x00\x01\x02\x03\x04\x05"[..]] {
12443            bits.push_bytes(value);
12444        }
12445        let expected = bits.clone();
12446        let fields =
12447            vec![Field::required("u", LogicalType::Uuid), Field::required("b", LogicalType::Bit)];
12448        let vectors = vec![
12449            Vector::flat(LogicalType::Uuid, Data::Int128(uuids.clone().into())).expect("uuids"),
12450            Vector::flat(LogicalType::Bit, Data::Varlen(bits)).expect("bit strings"),
12451        ];
12452        let mut writer = Writer::create(&path, "ids", fields).expect("new file");
12453        writer.append(&Chunk::new(vectors).expect("matching rows")).expect("one stripe");
12454        writer.finish().expect("commit");
12455
12456        let reader = Reader::open(&path).expect("reopen from disk");
12457        let read = reader.read(0, &[0, 1]).expect("both columns").flatten().expect("flat");
12458        let Some(Data::Int128(back)) = read.column(0).expect("the uuids").data() else {
12459            panic!("a uuid column is the 128 bit lane")
12460        };
12461        assert_eq!(back.as_slice(), uuids.as_slice());
12462        let Some(Data::Varlen(back)) = read.column(1).expect("the bits").data() else {
12463            panic!("a bit column is bytes")
12464        };
12465        for row in 0..expected.len() {
12466            assert_eq!(back.bytes(row), expected.bytes(row), "row {row} of the bit column");
12467        }
12468        fs::remove_file(path).expect("remove scratch file");
12469    }
12470
12471    #[test]
12472    fn numeric_frequency_candidates_keep_bounded_row_ordinals() {
12473        let path = path("frequency-ordinals");
12474        let mut writer =
12475            Writer::create(&path, "items", vec![Field::required("id", LogicalType::BigInt)])
12476                .expect("new file");
12477        let mut values = Vec::new();
12478        for leader in 0..10_i64 {
12479            values.extend(std::iter::repeat_n(leader, 100));
12480        }
12481        values.extend(1_000_i64..41_000);
12482        for part in values.chunks(1_024) {
12483            let vector = Vector::flat(LogicalType::BigInt, Data::Int64(part.to_vec().into()))
12484                .expect("big integers");
12485            writer.append(&Chunk::new(vec![vector]).expect("one column")).expect("one stripe");
12486        }
12487        writer.finish().expect("commit");
12488
12489        let reader = Reader::open(&path).expect("reopen from disk");
12490        let occurrences =
12491            reader.frequency_occurrences(0).expect("valid metadata").expect("bounded ordinals");
12492        assert!(occurrences.omitted_max < 100);
12493        assert!(occurrences.ordinals.len() <= FREQUENCY_ORDINALS);
12494        assert_eq!(occurrences.anchor_indices.len(), occurrences.ordinals.len());
12495        assert!(occurrences.ordinals.windows(2).all(|pair| pair[0] < pair[1]));
12496        assert_eq!(&occurrences.ordinals[..1_000], &(0_u64..1_000).collect::<Vec<_>>());
12497        assert_eq!(
12498            &occurrences.anchor_indices[..1_000]
12499                .iter()
12500                .map(|&entry| occurrences.anchors[entry as usize].clone())
12501                .collect::<Vec<_>>(),
12502            &(0_i64..10)
12503                .flat_map(|leader| std::iter::repeat_n(Value::BigInt(leader), 100))
12504                .collect::<Vec<_>>()
12505        );
12506        fs::remove_file(path).expect("remove scratch file");
12507    }
12508
12509    #[test]
12510    fn numeric_frequencies_count_nulls_and_values_past_the_top_of_bigint() {
12511        // Ten leaders, then more unique values than the candidate table holds, so the first pass
12512        // has to decrement and the counts come from the recount. The unsigned leaders sit above
12513        // `i64::MAX`, where reading the bits as signed would give a different value, and the signed
12514        // ones are negative, where reading them as unsigned would.
12515        let path = path("frequency-bits");
12516        let mut writer = Writer::create(
12517            &path,
12518            "items",
12519            vec![Field::new("u", LogicalType::UBigInt), Field::new("s", LogicalType::BigInt)],
12520        )
12521        .expect("new file");
12522        let mut rows = Vec::new();
12523        let mut leaders = Vec::new();
12524        for leader in 0..10_u64 {
12525            let count = 300 - leader * 10;
12526            let (unsigned, signed) = if leader == 0 {
12527                (Value::Null, Value::Null)
12528            } else {
12529                (Value::UBigInt(u64::MAX - leader), Value::BigInt(-(leader as i64)))
12530            };
12531            rows.extend(std::iter::repeat_n((unsigned.clone(), signed.clone()), count as usize));
12532            leaders.push(((unsigned, count), (signed, count)));
12533        }
12534        rows.extend((1_000..41_000_u64).map(|id| (Value::UBigInt(id), Value::BigInt(id as i64))));
12535        for part in rows.chunks(1_024) {
12536            let unsigned = part.iter().map(|(value, _)| value.clone()).collect::<Vec<_>>();
12537            let signed = part.iter().map(|(_, value)| value.clone()).collect::<Vec<_>>();
12538            let chunk = Chunk::new(vec![
12539                Vector::from_values(LogicalType::UBigInt, &unsigned).expect("unsigned"),
12540                Vector::from_values(LogicalType::BigInt, &signed).expect("signed"),
12541            ])
12542            .expect("matching columns");
12543            writer.append(&chunk).expect("rows");
12544        }
12545        writer.finish().expect("commit");
12546
12547        let reader = Reader::open(&path).expect("reopen from disk");
12548        for column in 0..2 {
12549            let prefix =
12550                reader.frequency_prefix(column).expect("valid metadata").expect("a synopsis");
12551            let wanted = leaders
12552                .iter()
12553                .map(|(unsigned, signed)| if column == 0 { unsigned } else { signed })
12554                .cloned()
12555                .collect::<Vec<_>>();
12556            assert_eq!(&prefix.entries[..10], &wanted[..], "column {column}");
12557            assert!(prefix.omitted_max < 210, "column {column}");
12558            assert_eq!(
12559                reader.distinct_values(column).expect("valid metadata"),
12560                Some(9 + 40_000),
12561                "column {column}"
12562            );
12563        }
12564        fs::remove_file(path).expect("remove scratch file");
12565    }
12566
12567    #[test]
12568    fn narrow_nonzero_count_matches_the_full_reader_across_stripes() {
12569        let path = path("quick-nonzero");
12570        let mut writer = Writer::create(
12571            &path,
12572            "items",
12573            vec![Field::new("label", LogicalType::Varchar), Field::new("id", LogicalType::Integer)],
12574        )
12575        .expect("create");
12576        for ids in [
12577            &[Value::Integer(0), Value::Null, Value::Integer(3)][..],
12578            &[Value::Integer(0), Value::Integer(7), Value::Null][..],
12579        ] {
12580            let labels = vec![Value::Varchar("same".into()); ids.len()];
12581            writer
12582                .append(
12583                    &Chunk::new(vec![
12584                        Vector::from_values(LogicalType::Varchar, &labels).expect("labels"),
12585                        Vector::from_values(LogicalType::Integer, ids).expect("ids"),
12586                    ])
12587                    .expect("chunk"),
12588                )
12589                .expect("append");
12590        }
12591        writer.finish().expect("finish");
12592        let catalog = Catalog::open(&path).expect("catalog");
12593        assert_eq!(catalog.entries[0].nonzero, vec![None, Some(2)]);
12594        assert_eq!(catalog.entries[0].aggregates, vec![None, Some((10, 4))]);
12595        assert_eq!(
12596            catalog.aggregate_sums("items", &[1]).expect("catalog sums"),
12597            Some(CertifiedSums { columns: vec![(10, 4)], rows: 6 })
12598        );
12599        assert_eq!(catalog.nonzero_count("items", 1).expect("quick count"), Some(2));
12600        assert_eq!(
12601            reader_nonzero_counts(&catalog.table("items").expect("reader")).expect("counts"),
12602            vec![None, Some(2)]
12603        );
12604        Writer::certify_counts(&path).expect("recertify");
12605        assert_eq!(
12606            Catalog::open(&path).expect("reopen").nonzero_count("items", 1).expect("count"),
12607            Some(2)
12608        );
12609        assert_eq!(
12610            Catalog::open(&path).expect("reopen").aggregate_sums("items", &[1]).expect("sums"),
12611            Some(CertifiedSums { columns: vec![(10, 4)], rows: 6 })
12612        );
12613        assert_eq!(catalog.table("items").expect("reader").null_count(1).expect("nulls"), 2);
12614        fs::remove_file(path).expect("remove scratch file");
12615    }
12616
12617    #[test]
12618    fn numeric_string_pair_leaders_are_certified_in_the_directory() {
12619        let path = path("pair-frequencies");
12620        let mut pairs = Vec::new();
12621        pairs.extend(std::iter::repeat_n((1_i64, "alpha".to_string()), 100));
12622        pairs.extend(std::iter::repeat_n((1_i64, "beta".to_string()), 50));
12623        pairs.extend(std::iter::repeat_n((2_i64, "gamma".to_string()), 40));
12624        pairs.extend((1_000_i64..1_600).map(|id| (id, format!("tail {id}"))));
12625        let mut writer = Writer::create(
12626            &path,
12627            "items",
12628            vec![
12629                Field::required("id", LogicalType::BigInt),
12630                Field::required("phrase", LogicalType::Varchar),
12631            ],
12632        )
12633        .expect("new file");
12634        for part in pairs.chunks(1_024) {
12635            let ids = part.iter().map(|(id, _)| Value::BigInt(*id)).collect::<Vec<_>>();
12636            let phrases =
12637                part.iter().map(|(_, phrase)| Value::Varchar(phrase.clone())).collect::<Vec<_>>();
12638            writer
12639                .append(
12640                    &Chunk::new(vec![
12641                        Vector::from_values(LogicalType::BigInt, &ids).expect("ids"),
12642                        Vector::from_values(LogicalType::Varchar, &phrases).expect("phrases"),
12643                    ])
12644                    .expect("matching columns"),
12645                )
12646                .expect("rows");
12647        }
12648        writer.finish().expect("commit");
12649
12650        let reader = Reader::open(&path).expect("reopen from disk");
12651        let leaders = reader
12652            .top_pair_frequencies(0, 1, 2)
12653            .expect("valid pair metadata")
12654            .expect("the top two beat the omitted tail");
12655        assert!(
12656            leaders.contains(&(vec![Value::BigInt(1), Value::Varchar("alpha".to_string())], 100,))
12657        );
12658        assert!(
12659            leaders.contains(&(vec![Value::BigInt(1), Value::Varchar("beta".to_string())], 50,))
12660        );
12661        fs::remove_file(path).expect("remove scratch file");
12662    }
12663
12664    /// The bug this is here for cost a 43 GB ClickBench table and an hour of reloading it. The
12665    /// format went from 11 to 12, every binary built after that said "magic or major version is
12666    /// unsupported" about the file, and there was no way to tell from the message whether the path
12667    /// was wrong, the file was truncated, or it was ours and simply older. The number this build
12668    /// wants is the whole answer and it was the one thing the message did not carry.
12669    #[test]
12670    fn a_file_from_another_format_says_which_format_it_is() {
12671        let older = path("older-format");
12672        let mut writer =
12673            Writer::create(&older, "items", vec![Field::new("id", LogicalType::Integer)])
12674                .expect("new file");
12675        let chunk = Chunk::new(vec![
12676            Vector::flat(LogicalType::Integer, Data::Int32(vec![1, 2, 3].into()))
12677                .expect("integers"),
12678        ])
12679        .expect("chunk");
12680        writer.append(&chunk).expect("page written");
12681        writer.finish().expect("commit");
12682
12683        // A format below the whole readable set, rather than `FORMAT - 1`, because the set has
12684        // more than one member now: format 22 is deliberately still readable, so the version that
12685        // has to be refused is the one under the oldest one accepted.
12686        let unreadable =
12687            READABLE.iter().copied().min().expect("at least one format is readable") - 1;
12688        let mut file = OpenOptions::new().write(true).open(&older).expect("open for the header");
12689        file.seek(SeekFrom::Start(8)).expect("the version follows the magic");
12690        file.write_all(&unreadable.to_le_bytes()).expect("write an older version");
12691        drop(file);
12692        let complaint = Reader::open(&older).expect_err("an older format is refused").to_string();
12693        assert!(complaint.contains(&format!("format {unreadable}")), "{complaint}");
12694        assert!(complaint.contains(&format!("format {FORMAT}")), "{complaint}");
12695
12696        let mut file = OpenOptions::new().write(true).open(&older).expect("open for the header");
12697        file.seek(SeekFrom::Start(0)).expect("the magic is first");
12698        file.write_all(b"NOTRUDB!").expect("write another engine's magic");
12699        drop(file);
12700        let complaint = Reader::open(&older).expect_err("a foreign file is refused").to_string();
12701        assert!(complaint.contains("magic"), "{complaint}");
12702        assert!(!complaint.contains("format"), "a version has nothing to do with it: {complaint}");
12703        fs::remove_file(older).expect("remove scratch file");
12704    }
12705
12706    #[test]
12707    fn an_unfinished_or_damaged_file_does_not_answer_with_partial_rows() {
12708        let unfinished = path("unfinished");
12709        let mut writer =
12710            Writer::create(&unfinished, "items", vec![Field::new("id", LogicalType::Integer)])
12711                .expect("new file");
12712        let chunk = Chunk::new(vec![
12713            Vector::flat(LogicalType::Integer, Data::Int32(vec![1, 2, 3].into()))
12714                .expect("integers"),
12715        ])
12716        .expect("chunk");
12717        writer.append(&chunk).expect("page written");
12718        drop(writer);
12719        assert!(Reader::open(&unfinished).is_err(), "no directory was committed");
12720        fs::remove_file(unfinished).expect("remove scratch file");
12721
12722        let damaged = path("damaged");
12723        let mut writer =
12724            Writer::create(&damaged, "items", vec![Field::new("id", LogicalType::Integer)])
12725                .expect("new file");
12726        writer.append(&chunk).expect("page written");
12727        writer.finish().expect("commit");
12728        let reader = Reader::open(&damaged).expect("valid directory");
12729        let mut file =
12730            OpenOptions::new().write(true).open(&damaged).expect("open for a damaged page");
12731        file.seek(SeekFrom::Start(HEADER + 1)).expect("inside first page");
12732        file.write_all(&[255]).expect("damage one byte");
12733        assert!(reader.read(0, &[0]).is_err(), "page checksum rejects corruption");
12734        fs::remove_file(damaged).expect("remove scratch file");
12735    }
12736
12737    #[test]
12738    fn damaged_lazy_dictionary_payload_is_an_error() {
12739        let path = path("damaged-dictionary");
12740        let mut writer = Writer::create(
12741            &path,
12742            "items",
12743            vec![
12744                Field::required("id", LogicalType::Integer),
12745                Field::new("text", LogicalType::Varchar),
12746            ],
12747        )
12748        .expect("new file");
12749        writer.append(&sample()).expect("stripe written");
12750        writer.finish().expect("commit");
12751
12752        let reader = Reader::open(&path).expect("valid directory");
12753        let dictionary = reader.table.dictionaries[1].expect("string dictionary page");
12754        // Read the count out of the page rather than writing it here, so that adding something
12755        // else to the index does not silently turn this into a test that damages the index.
12756        let mut header = [0; DICTIONARY_HEADER];
12757        read_at(&reader.file, dictionary.offset, &mut header).expect("dictionary header");
12758        // The first block's start is the first word after the offsets, since the blocks are written
12759        // during the load and are wherever the writer was when each was encoded.
12760        let count = u32::from_le_bytes(header[0..4].try_into().expect("four bytes")) as usize;
12761        let width = u32::from_le_bytes(header[12..16].try_into().expect("four bytes"));
12762        assert_ne!(width & DICTIONARY_SCATTERED, 0, "the blocks say where they are");
12763        let bits = (width & !(DICTIONARY_SCATTERED | DICTIONARY_GRAMS)) as usize;
12764        let mut start = [0; 8];
12765        let at = dictionary.offset + (DICTIONARY_HEADER + offset_bytes(count, bits)) as u64;
12766        read_at(&reader.file, at, &mut start).expect("the first block's start");
12767        let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
12768        file.seek(SeekFrom::Start(u64::from_le_bytes(start))).expect("inside dictionary payload");
12769        file.write_all(&[255]).expect("damage dictionary payload");
12770
12771        let chunk = reader.read(0, &[1]).expect("code page and dictionary index remain valid");
12772        let error =
12773            chunk.validate_external().expect_err("payload corruption must reach the caller");
12774        assert!(error.message().contains("payload checksum differs"), "{error}");
12775        fs::remove_file(path).expect("remove scratch file");
12776    }
12777
12778    /// A column whose values are all different is written without a dictionary, and one whose
12779    /// values repeat keeps it.
12780    ///
12781    /// The two columns go in the same table and hold the same number of rows, so the only thing
12782    /// separating them is how much of the first stripe was a value it had not seen before. Both have
12783    /// to read back the values that were written, because the decision is about cost and nothing
12784    /// else. The file size is the other half of it: a column written without a dictionary goes
12785    /// through the string cascade instead, so dropping the dictionary must not turn into storing the
12786    /// column raw.
12787    #[test]
12788    fn a_column_of_all_different_values_is_written_without_a_dictionary() {
12789        let path = path("dictionary-decide");
12790        let rows = 20_000;
12791        // Long enough that storing it raw would show, and different in every row.
12792        let unique =
12793            |row: usize| format!("{row:09} a value that appears exactly once in the table");
12794        // The same values in the same shape, each one used forty times over.
12795        let repeated = |row: usize| unique(row / 40);
12796        let mut writer = Writer::create(
12797            &path,
12798            "items",
12799            vec![
12800                Field::required("unique", LogicalType::Varchar),
12801                Field::required("repeated", LogicalType::Varchar),
12802            ],
12803        )
12804        .expect("new file");
12805        for part in (0..rows).step_by(1_000) {
12806            let span = part..(part + 1_000).min(rows);
12807            let left = span.clone().map(|row| Value::Varchar(unique(row))).collect::<Vec<_>>();
12808            let right = span.map(|row| Value::Varchar(repeated(row))).collect::<Vec<_>>();
12809            writer
12810                .append(
12811                    &Chunk::new(vec![
12812                        Vector::from_values(LogicalType::Varchar, &left).expect("strings"),
12813                        Vector::from_values(LogicalType::Varchar, &right).expect("strings"),
12814                    ])
12815                    .expect("two columns"),
12816                )
12817                .expect("a part");
12818        }
12819        writer.finish().expect("commit");
12820
12821        let reader = Reader::open(&path).expect("reopen from disk");
12822        assert!(
12823            reader.table.dictionaries[0].is_none(),
12824            "a column with no repeats has nothing to say twice"
12825        );
12826        assert!(
12827            reader.table.dictionaries[1].is_some(),
12828            "a column whose values come round again keeps its dictionary"
12829        );
12830        let mut first = 0;
12831        for part in 0..reader.parts() {
12832            let chunk = reader.read(part, &[0, 1]).expect("a part");
12833            for row in 0..chunk.len() {
12834                assert_eq!(chunk.value_at(row, 0), Value::Varchar(unique(first + row)));
12835                assert_eq!(chunk.value_at(row, 1), Value::Varchar(repeated(first + row)));
12836            }
12837            first += chunk.len();
12838        }
12839        assert_eq!(first, rows, "every row was read back");
12840        let raw = (0..rows).map(|row| unique(row).len()).sum::<usize>();
12841        let size = fs::metadata(&path).expect("the file is there").len() as usize;
12842        assert!(size < raw, "a column without a dictionary is still encoded: {size} against {raw}");
12843        fs::remove_file(path).expect("remove scratch file");
12844    }
12845
12846    /// A payload of many blocks reads and checks every block of it.
12847    ///
12848    /// The test above has a dictionary of three values, which is one block, so it says nothing
12849    /// about a reader finding the right block among many. This one has thirty two thousand values,
12850    /// which is thirty two blocks, and it reads a value out of the first block and a value out of
12851    /// the last and then damages the last and asks for it again.
12852    ///
12853    /// Forty thousand rows over those thirty two thousand values, because a column the writer finds
12854    /// to be all distinct does not get a dictionary at all and there would be nothing here to test.
12855    /// Four rows in five holding a value the stripe has not seen before is a column that keeps one.
12856    /// The repeats are put at the front so that the values still arrive in order after them, which
12857    /// is what keeps the last part of the table on the last block of the payload.
12858    #[test]
12859    fn a_dictionary_over_many_blocks_checks_every_block_of_it() {
12860        let path = path("dictionary-blocks");
12861        let value = |row: usize| {
12862            let row = row.saturating_sub(8_000);
12863            format!("{row:07} a value long enough to be worth a payload block")
12864        };
12865        let parts = 40;
12866        let per_part = 1000;
12867        let mut writer =
12868            Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
12869                .expect("new file");
12870        for part in 0..parts {
12871            let values = (0..per_part)
12872                .map(|row| Value::Varchar(value(part * per_part + row)))
12873                .collect::<Vec<_>>();
12874            let chunk = Chunk::new(vec![
12875                Vector::from_values(LogicalType::Varchar, &values).expect("strings"),
12876            ])
12877            .expect("matching rows");
12878            writer.append(&chunk).expect("a part");
12879        }
12880        writer.finish().expect("commit");
12881
12882        let reader = Reader::open(&path).expect("reopen from disk");
12883        let dictionary = reader.table.dictionaries[0].expect("string dictionary page");
12884        assert!(
12885            parts * per_part > TEXT_PAYLOAD_VALUES * 4,
12886            "the dictionary has to be several blocks for this to be testing anything"
12887        );
12888        for part in [0, parts - 1] {
12889            let chunk = reader.read(part, &[0]).expect("a part");
12890            chunk.validate_external().expect("every payload block checks out");
12891            assert_eq!(chunk.value_at(0, 0), Value::Varchar(value(part * per_part)));
12892        }
12893
12894        // The last block is wherever the writer was when it was encoded, which the index says.
12895        let mut header = [0; DICTIONARY_HEADER];
12896        read_at(&reader.file, dictionary.offset, &mut header).expect("dictionary header");
12897        let count = u32::from_le_bytes(header[0..4].try_into().expect("four bytes")) as usize;
12898        let blocks = u32::from_le_bytes(header[8..12].try_into().expect("four bytes")) as usize;
12899        let width = u32::from_le_bytes(header[12..16].try_into().expect("four bytes"));
12900        let bits = (width & !(DICTIONARY_SCATTERED | DICTIONARY_GRAMS)) as usize;
12901        let mut place = [0; 16];
12902        let at = DICTIONARY_HEADER + offset_bytes(count, bits) + (blocks - 1) * 16;
12903        read_at(&reader.file, dictionary.offset + at as u64, &mut place).expect("its place");
12904        let start = u64::from_le_bytes(place[..8].try_into().expect("eight bytes"));
12905        let length = u64::from_le_bytes(place[8..].try_into().expect("eight bytes"));
12906        let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
12907        file.seek(SeekFrom::Start(start + length - 4)).expect("the last bytes of the last block");
12908        file.write_all(&[255]).expect("damage the last payload block");
12909        let reader = Reader::open(&path).expect("the directory and the index are untouched");
12910        let chunk = reader.read(parts - 1, &[0]).expect("the code page remains valid");
12911        let error = chunk.validate_external().expect_err("the damage must reach the caller");
12912        assert!(error.message().contains("payload checksum differs"), "{error}");
12913        fs::remove_file(path).expect("remove scratch file");
12914    }
12915
12916    /// Values of different lengths read back where the offsets say they do.
12917    ///
12918    /// The offsets are packed at one width for the column, they are relative to the payload block a
12919    /// value lands in, and they go in runs of half a block, so there are two boundaries where the
12920    /// arithmetic could be off by one and neither shows up on values that are all the same length.
12921    /// This writes 5,000 values whose lengths cycle through a wide range and reads every one back,
12922    /// so the first value of a block, the last value of a run and the last value of a block are all
12923    /// covered several times over. An empty value is in the cycle because a zero length span is the
12924    /// case the reader short circuits.
12925    ///
12926    /// Six thousand rows over those 5,000 values, because a column the writer finds to be all
12927    /// distinct is written without a dictionary and then there are no packed offsets to be off by
12928    /// one in.
12929    #[test]
12930    fn values_of_different_lengths_read_back_out_of_packed_offsets() {
12931        let path = path("dictionary-offsets");
12932        let value = |row: usize| {
12933            let row = row % 5_000;
12934            if row % 511 == 3 { String::new() } else { "x".repeat(row % 97) + &format!("{row:05}") }
12935        };
12936        let rows = 6_000;
12937        let mut writer =
12938            Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
12939                .expect("new file");
12940        let values = (0..rows).map(|row| Value::Varchar(value(row))).collect::<Vec<_>>();
12941        for part in values.chunks(1_000) {
12942            let chunk =
12943                Chunk::new(vec![Vector::from_values(LogicalType::Varchar, part).expect("strings")])
12944                    .expect("matching rows");
12945            writer.append(&chunk).expect("a part");
12946        }
12947        writer.finish().expect("commit");
12948
12949        let reader = Reader::open(&path).expect("reopen from disk");
12950        assert!(
12951            rows > TEXT_PAYLOAD_VALUES * 4,
12952            "the dictionary has to be several blocks for this to be testing anything"
12953        );
12954        for part in 0..rows / 1_000 {
12955            let chunk = reader.read(part, &[0]).expect("a part");
12956            for row in 0..1_000 {
12957                let row = part * 1_000 + row;
12958                assert_eq!(
12959                    chunk.value_at(row % 1_000, 0),
12960                    Value::Varchar(value(row)),
12961                    "value {row}"
12962                );
12963            }
12964        }
12965        // The lengths a vector at a time, twice over, because the first pass is what makes the
12966        // table of ends worth building and the second is read out of the lengths worked out of it.
12967        for _ in 0..2 {
12968            for part in 0..rows / 1_000 {
12969                let chunk = reader.read(part, &[0]).expect("a part");
12970                let mut lens = vec![0_i64; 1_000];
12971                let column = chunk.column(0).expect("one column");
12972                assert!(column.try_bytes_lens(&mut lens).expect("lengths"), "a stored column");
12973                for (row, &len) in lens.iter().enumerate() {
12974                    let row = part * 1_000 + row;
12975                    assert_eq!(len as usize, value(row).len(), "the length of value {row}");
12976                }
12977            }
12978        }
12979        fs::remove_file(path).expect("remove scratch file");
12980    }
12981
12982    /// Lengths start again at every block, and ends that go backwards inside one give no table.
12983    #[test]
12984    fn lengths_restart_at_each_block_and_refuse_ends_that_go_backwards() {
12985        let mut ends: Vec<u32> = (1..=TEXT_PAYLOAD_VALUES as u32).map(|at| at * 2).collect();
12986        ends.extend([3, 3, 10]);
12987        let lens = lengths_of(&ends).expect("ordered ends");
12988        assert!(lens[..TEXT_PAYLOAD_VALUES].iter().all(|&len| len == 2));
12989        assert_eq!(&lens[TEXT_PAYLOAD_VALUES..], &[3, 0, 7]);
12990        ends.push(9);
12991        assert_eq!(lengths_of(&ends), None);
12992    }
12993
12994    /// Every worker of a scan wants the dictionary at the same moment and one of them fetches it.
12995    ///
12996    /// Asking a `OnceLock` whether it holds something answers the question a worker that already has
12997    /// the dictionary is asking and not the one a worker without it is asking, which is whether
12998    /// somebody is already on their way with it. Sixteen workers that all miss will all read the
12999    /// page, all verify it and all decode it, and fifteen will drop the result. Nothing about that
13000    /// is incorrect, which is why it went unnoticed, and it showed up as ClickBench 38 getting
13001    /// slower when the scan in front of it got faster and stopped staggering the arrivals.
13002    ///
13003    /// The barrier is what makes the test about that rather than about luck. Without it the first
13004    /// thread is usually finished before the last one starts and the count is one either way.
13005    #[test]
13006    fn a_global_dictionary_is_opened_once_however_many_workers_ask_at_once() {
13007        let path = path("dictionary-once");
13008        let parts = 8;
13009        let per_part = 500;
13010        let value =
13011            |row: usize| format!("{row:07} a value long enough to be worth a payload block");
13012        let mut writer =
13013            Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
13014                .expect("new file");
13015        for part in 0..parts {
13016            let values = (0..per_part)
13017                .map(|row| Value::Varchar(value(part * per_part + row)))
13018                .collect::<Vec<_>>();
13019            let chunk = Chunk::new(vec![
13020                Vector::from_values(LogicalType::Varchar, &values).expect("strings"),
13021            ])
13022            .expect("matching rows");
13023            writer.append(&chunk).expect("a part");
13024        }
13025        writer.finish().expect("commit");
13026
13027        let reader = Reader::open(&path).expect("reopen from disk");
13028        assert!(reader.table.dictionaries[0].is_some(), "the column has to have one to share");
13029        assert_eq!(reader.reads().dictionaries, 0, "opening the file does not open a dictionary");
13030
13031        let workers = 16;
13032        let gate = std::sync::Barrier::new(workers);
13033        std::thread::scope(|scope| {
13034            for worker in 0..workers {
13035                let reader = reader.clone();
13036                let gate = &gate;
13037                scope.spawn(move || {
13038                    gate.wait();
13039                    let chunk = reader.read(worker % parts, &[0]).expect("a part");
13040                    assert_eq!(
13041                        chunk.value_at(0, 0),
13042                        Value::Varchar(value((worker % parts) * per_part))
13043                    );
13044                });
13045            }
13046        });
13047
13048        assert_eq!(reader.reads().dictionaries, 1, "sixteen workers, one dictionary, one open");
13049        fs::remove_file(path).expect("remove scratch file");
13050    }
13051
13052    /// The sorted order sits outside the index the page checksum covers, because a query that
13053    /// never searches a dictionary should not read it, so it carries its own checksums and this is
13054    /// what says they are checked. A search that trusted a damaged order would give a wrong answer
13055    /// rather than a slow one.
13056    #[test]
13057    fn a_damaged_sorted_order_is_an_error() {
13058        let path = path("damaged-order");
13059        let mut writer = Writer::create(
13060            &path,
13061            "items",
13062            vec![
13063                Field::required("id", LogicalType::Integer),
13064                Field::new("text", LogicalType::Varchar),
13065            ],
13066        )
13067        .expect("new file");
13068        writer.append(&sample()).expect("stripe written");
13069        writer.finish().expect("commit");
13070
13071        let reader = Reader::open(&path).expect("valid directory");
13072        let page = reader.table.dictionaries[1].expect("string dictionary page");
13073        let mut header = [0; DICTIONARY_HEADER];
13074        read_at(&reader.file, page.offset, &mut header).expect("dictionary header");
13075        let index_len = dictionary_index_len(&header);
13076        let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
13077        file.seek(SeekFrom::Start(page.offset + index_len)).expect("the first head");
13078        file.write_all(&[255]).expect("damage the order");
13079
13080        let dictionary = reader.dictionary(1).expect("read").expect("a string column has one");
13081        let error = dictionary.compare_rank(0, b"anything").expect_err("a damaged order is caught");
13082        assert!(error.message().contains("rank checksum differs"), "{error}");
13083        fs::remove_file(path).expect("remove scratch file");
13084    }
13085
13086    /// Codes stay in first appearance order and the sorted order is written beside them, so a
13087    /// reader can put the values back in order without the writer having had to know them all
13088    /// before it handed out the first code.
13089    #[test]
13090    fn a_global_dictionary_carries_the_sorted_order_of_its_values() {
13091        // Chosen so the sort cannot be decided on the first eight bytes alone. Three values share
13092        // a nine byte prefix, one is a prefix of another, and one is empty.
13093        let spellings = ["overlong1z", "b", "", "overlong1a", "overlong", "ab", "a", "overlong1"];
13094        let path = path("dictionary-order");
13095        let mut writer =
13096            Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
13097                .expect("new file");
13098        writer
13099            .append(
13100                &Chunk::new(vec![
13101                    Vector::from_values(
13102                        LogicalType::Varchar,
13103                        &spellings.map(|text| Value::Varchar(text.into())),
13104                    )
13105                    .expect("strings"),
13106                ])
13107                .expect("one column"),
13108            )
13109            .expect("stripe written");
13110        writer.finish().expect("commit");
13111
13112        let reader = Reader::open(&path).expect("valid directory");
13113        let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
13114        let count = dictionary.ranks().expect("a v10 file stores one");
13115        assert_eq!(count, spellings.len(), "every distinct value has a rank");
13116        let order = (0..count)
13117            .map(|rank| dictionary.code_at_rank(rank).expect("a code"))
13118            .collect::<Vec<_>>();
13119        let mut seen = order.clone();
13120        seen.sort_unstable();
13121        assert_eq!(seen, (0..spellings.len() as u32).collect::<Vec<_>>(), "a permutation of codes");
13122
13123        let ranked = order
13124            .iter()
13125            .map(|&code| {
13126                dictionary.try_bytes_at(code as usize).expect("read").expect("a value").to_vec()
13127            })
13128            .collect::<Vec<_>>();
13129        let mut expected = spellings.map(|text| text.as_bytes().to_vec()).to_vec();
13130        expected.sort();
13131        assert_eq!(ranked, expected, "rank order is value order");
13132
13133        // What a search asks, on the values themselves rather than through a kernel, so that a
13134        // file whose heads disagree with its bytes is caught here rather than as a wrong answer.
13135        for (rank, value) in expected.iter().enumerate() {
13136            assert_eq!(
13137                dictionary.compare_rank(rank, value).expect("compare"),
13138                Ordering::Equal,
13139                "rank {rank} is its own value"
13140            );
13141            if rank > 0 {
13142                assert_eq!(
13143                    dictionary.compare_rank(rank - 1, value).expect("compare"),
13144                    Ordering::Less,
13145                    "rank {rank} follows the one before it"
13146                );
13147            }
13148        }
13149        fs::remove_file(path).expect("remove scratch file");
13150    }
13151
13152    /// A dictionary large enough to be decoded and sorted on several threads ranks the way one small
13153    /// enough for one thread does.
13154    ///
13155    /// Seventy thousand values over sixty nine blocks, in no order and each four times over so the
13156    /// column is worth a dictionary, written and ranked in the close.
13157    /// Some share a long prefix and some differ only in the last byte, so the buckets of the sort cut
13158    /// through runs of values that agree for a long way.
13159    #[test]
13160    fn a_large_dictionary_ranks_in_value_order() {
13161        let path = path("dictionary-large-rank");
13162        let value = |row: u64| {
13163            let mixed = row.wrapping_mul(0x9e37_79b9_7f4a_7c15) >> 40;
13164            match row % 3 {
13165                0 => format!("https://example.com/a/long/shared/path/{mixed:08}"),
13166                1 => format!("{mixed}"),
13167                _ => format!("x{}", row % 1000).repeat(1 + (row % 4) as usize) + &row.to_string(),
13168            }
13169        };
13170        let distinct = 70_000;
13171        let parts = 4 * distinct / 1000;
13172        let per_part = 1000;
13173        let mut writer =
13174            Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
13175                .expect("new file");
13176        for part in 0..parts {
13177            let values = (0..per_part)
13178                .map(|row| Value::Varchar(value((part * per_part + row) / 4)))
13179                .collect::<Vec<_>>();
13180            let chunk = Chunk::new(vec![
13181                Vector::from_values(LogicalType::Varchar, &values).expect("strings"),
13182            ])
13183            .expect("matching rows");
13184            writer.append(&chunk).expect("a part");
13185        }
13186        writer.finish().expect("commit");
13187
13188        let reader = Reader::open(&path).expect("reopen from disk");
13189        let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
13190        let count = dictionary.ranks().expect("a ranked dictionary");
13191        assert_eq!(count, distinct as usize, "every distinct value has a rank");
13192        assert!(count >= PARALLEL_SORT_MIN, "too few values to be sorted on more than one thread");
13193        let ranked = (0..count)
13194            .map(|rank| {
13195                let code = dictionary.code_at_rank(rank).expect("a code");
13196                dictionary.try_bytes_at(code as usize).expect("read").expect("a value").to_vec()
13197            })
13198            .collect::<Vec<_>>();
13199        let mut expected = (0..distinct).map(|row| value(row).into_bytes()).collect::<Vec<_>>();
13200        expected.sort();
13201        assert_eq!(ranked, expected, "rank order is value order");
13202        fs::remove_file(path).expect("remove scratch file");
13203    }
13204
13205    /// A string column's synopsis is turned into values without keeping the blocks it went through.
13206    ///
13207    /// Three thousand values, every fifth of them four times over, so the synopsis is a prefix of
13208    /// five hundred and twelve codes spread over all three payload blocks. Reading it used to leave
13209    /// all three decoded for as long as the reader lived. It leaves none of them now, and the second
13210    /// read answers out of what the first remembered.
13211    /// A directory read out of the file a window at a time is the directory read whole.
13212    ///
13213    /// The windows here are far smaller than any field is long, so every kind of field is split
13214    /// across a refill somewhere, and a bound is offered to its codec short more than once. The
13215    /// synopses are left in the file, and each one read back from where it was left is the one the
13216    /// whole read decoded.
13217    #[test]
13218    fn a_directory_read_a_window_at_a_time_is_the_directory_read_whole() {
13219        let path = path("windowed-directory");
13220        let fields = vec![
13221            Field::required("id", LogicalType::BigInt),
13222            Field::required("word", LogicalType::Varchar),
13223            Field::new("score", LogicalType::Double),
13224        ];
13225        let mut writer = Writer::create(&path, "items", fields).expect("new file");
13226        for part in 0..70_i64 {
13227            let ids = (0..100).map(|row| Value::BigInt(part * 100 + row % 7)).collect::<Vec<_>>();
13228            let words = (0..100)
13229                .map(|row| Value::Varchar(format!("word {}", row % 13)))
13230                .collect::<Vec<_>>();
13231            let scores = (0..100)
13232                .map(|row| if row % 4 == 0 { Value::Null } else { Value::Double(row as f64) })
13233                .collect::<Vec<_>>();
13234            let chunk = Chunk::new(vec![
13235                Vector::from_values(LogicalType::BigInt, &ids).expect("integers"),
13236                Vector::from_values(LogicalType::Varchar, &words).expect("strings"),
13237                Vector::from_values(LogicalType::Double, &scores).expect("doubles"),
13238            ])
13239            .expect("three columns");
13240            writer.append(&chunk).expect("a part");
13241        }
13242        writer.finish().expect("commit");
13243
13244        let catalog = Catalog::open(&path).expect("reopen");
13245        let entry = catalog.entries.first().expect("one table").directory;
13246        let (offset, length) = (entry.offset, entry.length as usize);
13247        let mut bytes = vec![0; length];
13248        read_at(&catalog.file, offset, &mut bytes).expect("the directory");
13249        assert_eq!(file_checksum(&catalog.file, offset, length).expect("checksum"), entry.hash);
13250        let whole = decode_directory(&bytes, catalog.size).expect("whole");
13251        assert!(whole.stripes.len() > 1, "the table should span stripes");
13252        for size in [1, 7, 33, 4_096] {
13253            let mut cursor = Cursor::over(&catalog.file, offset, length);
13254            cursor.window.as_mut().expect("a window").size = size;
13255            let windowed = read_directory(cursor, catalog.size, Some(offset)).expect("windowed");
13256            assert_eq!(format!("{:?}", windowed.stripes), format!("{:?}", whole.stripes));
13257            assert_eq!(format!("{:?}", windowed.fields), format!("{:?}", whole.fields));
13258            let mut stored = 0;
13259            for (column, (left, held)) in
13260                windowed.frequencies.iter().zip(&whole.frequencies).enumerate()
13261            {
13262                match (left, held) {
13263                    (None, None) => {}
13264                    (
13265                        Some(super::Frequencies::Stored { span, values }),
13266                        Some(super::Frequencies::Held(summary)),
13267                    ) => {
13268                        let mut one = vec![0; span.length as usize];
13269                        read_at(&catalog.file, span.offset, &mut one).expect("a synopsis");
13270                        let read = decode_summary(
13271                            &mut Cursor::new(&one),
13272                            &whole.fields[column],
13273                            whole.rows,
13274                            *values,
13275                        )
13276                        .expect("a valid synopsis")
13277                        .expect("one is there");
13278                        assert_eq!(format!("{read:?}"), format!("{summary:?}"));
13279                        stored += 1;
13280                    }
13281                    other => panic!("column {column} came back as {other:?}"),
13282                }
13283            }
13284            assert!(stored >= 2, "only {stored} synopses were left in the file");
13285        }
13286        let reader = catalog.table("items").expect("the table");
13287        assert!(reader.frequency_summaries[1].get().is_none());
13288        assert!(reader.top_frequencies(1, 1).expect("a readable synopsis").is_some());
13289        let first = reader.frequency_summaries[1].get().expect("decoded synopsis");
13290        let clone = reader.clone();
13291        assert!(clone.top_frequencies(1, 1).expect("cached synopsis").is_some());
13292        assert!(Arc::ptr_eq(first, clone.frequency_summaries[1].get().expect("same synopsis")));
13293        fs::remove_file(path).expect("remove scratch file");
13294    }
13295
13296    #[test]
13297    fn a_checksum_carried_across_reads_is_the_checksum_of_the_whole() {
13298        let path = path("file-checksum");
13299        let bytes = (0..200_000_u32)
13300            .map(|at| (at.wrapping_mul(2_654_435_761) >> 13) as u8)
13301            .collect::<Vec<_>>();
13302        fs::write(&path, &bytes).expect("scratch file");
13303        let file = File::open(&path).expect("open");
13304        for (offset, length) in [
13305            (0, 0),
13306            (3, 1),
13307            (5, 31),
13308            (0, 32),
13309            (9, 33),
13310            (1, 65_536),
13311            (7, 65_567),
13312            (0, 200_000),
13313            (11, 131_101),
13314        ] {
13315            let whole = checksum(&bytes[offset..offset + length]);
13316            assert_eq!(
13317                file_checksum(&file, offset as u64, length).expect("read"),
13318                whole,
13319                "{offset} {length}"
13320            );
13321        }
13322        fs::remove_file(path).expect("remove scratch file");
13323    }
13324
13325    #[test]
13326    fn a_string_synopsis_is_read_without_keeping_the_dictionary_blocks() {
13327        let path = path("synopsis-keeps-no-block");
13328        let spelled = |index: usize| Value::Varchar(format!("phrase {index:05}"));
13329        let mut values = (0..3_000).map(spelled).collect::<Vec<_>>();
13330        for _ in 0..3 {
13331            values.extend((0..3_000).step_by(5).map(spelled));
13332        }
13333        let mut writer =
13334            Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
13335                .expect("new file");
13336        for part in values.chunks(1_024) {
13337            writer
13338                .append(
13339                    &Chunk::new(vec![
13340                        Vector::from_values(LogicalType::Varchar, part).expect("strings"),
13341                    ])
13342                    .expect("one column"),
13343                )
13344                .expect("a part");
13345        }
13346        writer.finish().expect("commit");
13347
13348        let reader = Reader::open(&path).expect("reopen from disk");
13349        let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
13350        let resting = dictionary.footprint();
13351        let prefix = reader.frequency_prefix(0).expect("a readable synopsis").expect("one");
13352        assert_eq!(prefix.entries.len(), 512);
13353        for (value, count) in &prefix.entries {
13354            let Value::Varchar(text) = value else { panic!("a string column gave {value:?}") };
13355            let index = text["phrase ".len()..].parse::<usize>().expect("a spelled number");
13356            assert_eq!((index % 5, *count), (0, 4), "{text} came back with {count}");
13357        }
13358        assert_eq!(dictionary.footprint(), resting, "reading the synopsis kept a decoded block");
13359        let again = reader.frequency_prefix(0).expect("a readable synopsis").expect("one");
13360        assert_eq!(again.entries, prefix.entries);
13361        fs::remove_file(path).expect("remove scratch file");
13362    }
13363
13364    /// A sweep of the dictionary reads every value, and the second sweep keeps what it read, up to
13365    /// the budget.
13366    ///
13367    /// The point of the sweep is the resident size rather than the answer, so both are checked
13368    /// here. The first sweep keeps nothing, because a process that runs one statement never reads
13369    /// a block twice. A dictionary this small is well under [`TEXT_KEEP_BUDGET`], so the second
13370    /// sweep keeps everything and a third decodes nothing, which is what makes a session asking the
13371    /// same question again cost what it should. The ceiling is the other half of it and it has its own
13372    /// test below, because a ceiling that never binds is not a ceiling anybody checked.
13373    #[test]
13374    fn a_dictionary_sweep_reads_every_value_and_keeps_it_under_the_budget() {
13375        let path = path("dictionary-sweep");
13376        // Two thousand five hundred distinct values is two whole payload blocks and a part of a
13377        // third, so the sweep has to be called more than once and the last call has to stop short.
13378        let spellings = (0..2_500)
13379            .map(|index| Value::Varchar(format!("value {index:08} {}", "x".repeat(index % 40))))
13380            .collect::<Vec<_>>();
13381        let mut writer =
13382            Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
13383                .expect("new file");
13384        // A chunk is a part and a part is at most 1,024 rows, so the values go in three of them.
13385        // The dictionary is table wide and does not care where a value was written.
13386        for part in spellings.chunks(1_024) {
13387            writer
13388                .append(
13389                    &Chunk::new(vec![
13390                        Vector::from_values(LogicalType::Varchar, part).expect("strings"),
13391                    ])
13392                    .expect("one column"),
13393                )
13394                .expect("stripe written");
13395        }
13396        writer.finish().expect("commit");
13397
13398        let reader = Reader::open(&path).expect("valid directory");
13399        let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
13400        assert_eq!(dictionary.len(), spellings.len(), "every value is distinct");
13401        for first in [0, TEXT_PAYLOAD_VALUES, TEXT_PAYLOAD_VALUES * 2] {
13402            assert!(dictionary.text_block_might_contain(first, b"value").expect("signature"));
13403            assert!(!dictionary.text_block_might_contain(first, b"google").expect("signature"));
13404        }
13405
13406        let resting = dictionary.footprint();
13407        let sweep = || {
13408            let mut swept: Vec<Vec<u8>> = Vec::new();
13409            let mut at = 0;
13410            let mut calls = 0;
13411            while at < dictionary.len() {
13412                let stopped = dictionary
13413                    .sweep_text(at, dictionary.len(), &mut |index: usize, text: &[u8]| {
13414                        assert_eq!(index, swept.len(), "a sweep hands its values over in order");
13415                        swept.push(text.to_vec());
13416                        Ok(())
13417                    })
13418                    .expect("a sweep reads");
13419                assert!(stopped > at, "a sweep moves");
13420                at = stopped;
13421                calls += 1;
13422            }
13423            assert_eq!(calls, 3, "a sweep hands over one block at a time");
13424            swept
13425        };
13426        let swept = sweep();
13427        assert_eq!(dictionary.footprint(), resting, "a first sweep keeps nothing it decoded");
13428        assert_eq!(sweep(), swept, "a second sweep reads what the first did");
13429        let after = dictionary.footprint();
13430        assert!(after > resting, "a second sweep under the budget keeps what it decoded");
13431
13432        let read = (0..dictionary.len())
13433            .map(|code| dictionary.try_bytes_at(code).expect("read").expect("a value").to_vec())
13434            .collect::<Vec<_>>();
13435        assert_eq!(swept, read, "a sweep answers what a point read answers");
13436        // A read per value is about what makes the unpacked ends worth building, so whether they
13437        // are built here depends on how many reads the sweep made on the way. They are the one thing
13438        // allowed to grow, by four bytes a value, and nothing of the payload is.
13439        let grown = dictionary.footprint() - after;
13440        assert!(
13441            grown == 0 || grown == dictionary.len() * size_of::<u32>(),
13442            "a point read of a kept block decodes nothing, and {grown} bytes grew"
13443        );
13444        fs::remove_file(path).expect("remove scratch file");
13445    }
13446
13447    #[test]
13448    fn a_damaged_substring_signature_is_checked_only_when_used() {
13449        let path = path("damaged-substring-signature");
13450        let mut writer =
13451            Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
13452                .expect("new file");
13453        let rows = [Value::Varchar("google".into()), Value::Varchar("example".into())];
13454        writer
13455            .append(
13456                &Chunk::new(vec![
13457                    Vector::from_values(LogicalType::Varchar, &rows).expect("strings"),
13458                ])
13459                .expect("one column"),
13460            )
13461            .expect("stripe written");
13462        writer.finish().expect("commit");
13463
13464        let reader = Reader::open(&path).expect("valid directory");
13465        let page = reader.table.dictionaries[0].expect("string dictionary page");
13466        let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
13467        file.seek(SeekFrom::Start(page.offset + u64::from(page.length) - 1))
13468            .expect("last signature byte");
13469        file.write_all(&[255]).expect("damage signature");
13470        let reader = Reader::open(&path).expect("the directory is still valid");
13471        let dictionary = reader.dictionary(0).expect("index is still valid").expect("dictionary");
13472        let error = dictionary
13473            .text_block_might_contain(0, b"goog")
13474            .expect_err("a used signature checks its own checksum");
13475        assert!(error.message().contains("substring signatures checksum differs"), "{error}");
13476        fs::remove_file(path).expect("remove scratch file");
13477    }
13478
13479    /// A sweep over a block whose second run of offsets is short reads the same values as a point
13480    /// read does.
13481    ///
13482    /// The sweep decodes the offsets of a whole run at a time rather than a value at a time, and a
13483    /// run holds half a block, so the count it asks for is the run length everywhere but at the end
13484    /// of the dictionary. Two thousand five hundred values, which is what the test above writes,
13485    /// never puts a short run second in its block: the last block there begins on a run boundary and
13486    /// holds one run. Two thousand eight hundred does, so the last block is a whole run of five
13487    /// hundred and twelve followed by two hundred and forty, and an off by one in either the count
13488    /// asked for or the slice taken out of the answer shows up as a wrong value or a refusal.
13489    #[test]
13490    fn a_sweep_over_a_block_with_a_short_second_run_reads_what_a_point_read_reads() {
13491        let path = path("dictionary-sweep-short-run");
13492        let spellings = (0..2_800)
13493            .map(|index| Value::Varchar(format!("value {index:08} {}", "x".repeat(index % 40))))
13494            .collect::<Vec<_>>();
13495        let mut writer =
13496            Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
13497                .expect("new file");
13498        for part in spellings.chunks(1_024) {
13499            writer
13500                .append(
13501                    &Chunk::new(vec![
13502                        Vector::from_values(LogicalType::Varchar, part).expect("strings"),
13503                    ])
13504                    .expect("one column"),
13505                )
13506                .expect("stripe written");
13507        }
13508        writer.finish().expect("commit");
13509
13510        let reader = Reader::open(&path).expect("valid directory");
13511        let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
13512        assert_eq!(dictionary.len(), spellings.len(), "every value is distinct");
13513        let last = dictionary.len() % TEXT_PAYLOAD_VALUES;
13514        assert!(last > TEXT_OFFSET_RUN, "the last block has to reach into a second run of offsets");
13515        assert!(last < TEXT_PAYLOAD_VALUES, "and that second run has to be short of a whole one");
13516
13517        let mut swept: Vec<Vec<u8>> = Vec::new();
13518        let mut at = 0;
13519        while at < dictionary.len() {
13520            let stopped = dictionary
13521                .sweep_text(at, dictionary.len(), &mut |index: usize, text: &[u8]| {
13522                    assert_eq!(index, swept.len(), "a sweep hands its values over in order");
13523                    swept.push(text.to_vec());
13524                    Ok(())
13525                })
13526                .expect("a sweep reads");
13527            assert!(stopped > at, "a sweep moves");
13528            at = stopped;
13529        }
13530        let read = (0..dictionary.len())
13531            .map(|code| dictionary.try_bytes_at(code).expect("read").expect("a value").to_vec())
13532            .collect::<Vec<_>>();
13533        assert_eq!(swept, read, "a sweep answers what a point read answers");
13534        fs::remove_file(path).expect("remove scratch file");
13535    }
13536
13537    /// The unpacked ends answer what the packed ends answer, on both sides of the switch.
13538    ///
13539    /// A column asked for one offset at a time reads them out of the packed form until the reads
13540    /// are worth a table and out of the table after that, so every value here is read twice and the
13541    /// two passes are compared against the spellings and against each other. Two thousand eight
13542    /// hundred values is two payload blocks and a bit, which puts the switch in the middle of the
13543    /// first pass and means the pass straddles a block boundary, where the start of a value is zero
13544    /// rather than the end of the value before it.
13545    #[test]
13546    fn the_unpacked_ends_answer_what_the_packed_ends_answer() {
13547        let path = path("dictionary-unpacked-ends");
13548        let spellings = (0..2_800)
13549            .map(|index| Value::Varchar(format!("value {index:08} {}", "x".repeat(index % 40))))
13550            .collect::<Vec<_>>();
13551        let mut writer =
13552            Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
13553                .expect("new file");
13554        for part in spellings.chunks(1_024) {
13555            writer
13556                .append(
13557                    &Chunk::new(vec![
13558                        Vector::from_values(LogicalType::Varchar, part).expect("strings"),
13559                    ])
13560                    .expect("one column"),
13561                )
13562                .expect("stripe written");
13563        }
13564        writer.finish().expect("commit");
13565
13566        let reader = Reader::open(&path).expect("valid directory");
13567        let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
13568        assert_eq!(dictionary.len(), spellings.len(), "every value is distinct");
13569        let wanted = (0..spellings.len())
13570            .map(|index| format!("value {index:08} {}", "x".repeat(index % 40)).into_bytes())
13571            .collect::<Vec<_>>();
13572
13573        let pass = |what: &str| {
13574            for (index, value) in wanted.iter().enumerate() {
13575                let len = dictionary.try_bytes_len_at(index).expect("read").expect("a value");
13576                assert_eq!(len, value.len(), "{what} has the wrong length at {index}");
13577                let bytes = dictionary.try_bytes_at(index).expect("read").expect("a value");
13578                assert_eq!(bytes, value.as_slice(), "{what} has the wrong value at {index}");
13579            }
13580        };
13581        pass("the first pass");
13582        pass("the second pass");
13583
13584        // The whole vector in one call, over the text and through codes into it, which is how a
13585        // scan of a stored column hands it out. The codes run backwards and repeat so that they are
13586        // neither the positions nor in order.
13587        let lens = wanted.iter().map(|value| value.len() as i64).collect::<Vec<_>>();
13588        let mut whole = vec![0i64; wanted.len()];
13589        assert!(dictionary.try_bytes_lens(&mut whole).expect("read"), "the text answers whole");
13590        assert_eq!(whole, lens, "a vector of lengths answers what a length at a time answers");
13591        let codes = (0..4_000_u32).map(|row| (7 * (4_000 - row)) % 2_800).collect::<Vec<_>>();
13592        let coded = Vector::dictionary_over(codes.clone(), dictionary).expect("codes in range");
13593        let mut through = vec![0i64; codes.len()];
13594        assert!(coded.try_bytes_lens(&mut through).expect("read"), "the codes answer whole");
13595        for (row, &code) in codes.iter().enumerate() {
13596            assert_eq!(through[row], lens[code as usize], "row {row} reads code {code}");
13597            let one = coded.try_bytes_len_at(row).expect("read").expect("a value");
13598            assert_eq!(through[row], one as i64, "row {row} a row at a time");
13599        }
13600
13601        // A handful of codes over a column nobody has read yet is short of the table, so the same
13602        // call answers out of the packed ends instead, and has to answer the same.
13603        let fresh = Reader::open(&path).expect("valid directory");
13604        let untouched = fresh.dictionary(0).expect("read").expect("a string column has one");
13605        let few = vec![2_799_u32, 0, 1_024, 1_023, 511, 512];
13606        let coded = Vector::dictionary_over(few.clone(), untouched).expect("in range");
13607        let mut short = vec![0i64; few.len()];
13608        assert!(coded.try_bytes_lens(&mut short).expect("read"), "the codes answer whole");
13609        let expected = few.iter().map(|&code| lens[code as usize]).collect::<Vec<_>>();
13610        assert_eq!(short, expected, "the packed ends answer what the table answers");
13611        fs::remove_file(path).expect("remove scratch file");
13612    }
13613
13614    /// Narrowing a page takes what fits and refuses the page for anything that does not.
13615    ///
13616    /// The edges of the range on both sides and one step past each of them, for every type, because
13617    /// checking a page separately from converting it is only right if the check refuses exactly what
13618    /// `TryFrom` would have refused, and off by one there is a file that reads back a different
13619    /// number than it was given. The check is a bit pattern rather than a comparison, so it is not
13620    /// the shape a reader would guess from the bounds, which is why all six are here. The empty page
13621    /// is here because a check written the obvious way starts with the extremes the wrong way round
13622    /// and refuses it.
13623    #[test]
13624    fn narrowing_a_page_takes_what_fits_and_refuses_what_does_not() {
13625        assert_eq!(fit::<i8>(&[]).expect("an empty page fits anything"), Vec::<i8>::new());
13626        assert_eq!(fit::<i8>(&[-128, 0, 127]).expect("the edges fit"), vec![-128_i8, 0, 127]);
13627        fit::<i8>(&[128]).expect_err("one past the top does not fit");
13628        fit::<i8>(&[-129]).expect_err("one past the bottom does not fit");
13629        assert_eq!(fit::<u8>(&[0, 255]).expect("the edges fit"), vec![0_u8, 255]);
13630        fit::<u8>(&[256]).expect_err("one past the top does not fit");
13631        fit::<u8>(&[-1]).expect_err("a negative does not fit an unsigned page");
13632        assert_eq!(
13633            fit::<i16>(&[-32_768, 0, 32_767]).expect("the edges fit"),
13634            vec![-32_768_i16, 0, 32_767]
13635        );
13636        fit::<i16>(&[32_768]).expect_err("one past the top does not fit");
13637        fit::<i16>(&[-32_769]).expect_err("one past the bottom does not fit");
13638        assert_eq!(fit::<u16>(&[0, 65_535]).expect("the edges fit"), vec![0_u16, 65_535]);
13639        fit::<u16>(&[65_536]).expect_err("one past the top does not fit");
13640        fit::<u16>(&[-1]).expect_err("a negative does not fit an unsigned page");
13641        assert_eq!(
13642            fit::<i32>(&[i64::from(i32::MIN), 0, i64::from(i32::MAX)]).expect("the edges fit"),
13643            vec![i32::MIN, 0, i32::MAX]
13644        );
13645        fit::<i32>(&[i64::from(i32::MAX) + 1]).expect_err("one past the top does not fit");
13646        fit::<i32>(&[i64::from(i32::MIN) - 1]).expect_err("one past the bottom does not fit");
13647        assert_eq!(
13648            fit::<u32>(&[0, 4_294_967_295]).expect("the edges fit"),
13649            vec![0_u32, 4_294_967_295]
13650        );
13651        fit::<u32>(&[4_294_967_296]).expect_err("one past the top does not fit");
13652        fit::<u32>(&[-1]).expect_err("a negative does not fit an unsigned page");
13653
13654        // One value in a page that fits is still a page that does not, which is the thing an or
13655        // into an accumulator could get wrong in a way a page of one value would never show.
13656        fit::<i8>(&[0, 1, 2, 128, 3]).expect_err("one bad value spoils the page");
13657    }
13658
13659    /// The residue says yes to exactly what `TryFrom` says yes to.
13660    ///
13661    /// The edges above are the cases anyone would think to write down. This is the argument that
13662    /// there are no others, made by asking both questions about every value either narrow type could
13663    /// have an opinion about, and then about the values around the wide edges and the ends of an
13664    /// `i64`, which a range that size cannot reach.
13665    #[test]
13666    fn the_residue_agrees_with_a_checked_conversion_everywhere() {
13667        for value in -70_000_i64..70_000 {
13668            assert_eq!(fit::<i8>(&[value]).is_ok(), i8::try_from(value).is_ok(), "{value} as i8");
13669            assert_eq!(fit::<u8>(&[value]).is_ok(), u8::try_from(value).is_ok(), "{value} as u8");
13670            assert_eq!(fit::<i16>(&[value]).is_ok(), i16::try_from(value).is_ok(), "{value} i16");
13671            assert_eq!(fit::<u16>(&[value]).is_ok(), u16::try_from(value).is_ok(), "{value} u16");
13672        }
13673        let wide = [i64::MIN, i64::MIN + 1, i64::from(i32::MIN), 0, i64::from(u32::MAX), i64::MAX];
13674        for edge in wide {
13675            for step in -2_i64..=2 {
13676                let value = edge.saturating_add(step);
13677                assert_eq!(
13678                    fit::<i32>(&[value]).is_ok(),
13679                    i32::try_from(value).is_ok(),
13680                    "{value} as i32"
13681                );
13682                assert_eq!(
13683                    fit::<u32>(&[value]).is_ok(),
13684                    u32::try_from(value).is_ok(),
13685                    "{value} as u32"
13686                );
13687            }
13688        }
13689    }
13690
13691    /// All three block layouts come back as the same values in the same order.
13692    ///
13693    /// Blocks outside the page are what every file this build writes holds. Blocks that say where
13694    /// they are but sit inside the page behind the order are format 26, and blocks behind one
13695    /// another with only their ends recorded are older still. Nothing in the writer produces the
13696    /// last two any more, so the only way to find out whether the reader still understands those
13697    /// files is to write them here. The
13698    /// bytes go straight into a file with no directory around them, because what is under test is
13699    /// [`open_global_dictionary`], which is handed a page and a file and asks the directory for
13700    /// nothing.
13701    ///
13702    /// Three thousand values so that there are three payload blocks and a partial fourth, which is
13703    /// what makes the last block the one place where a length and an end disagree about what they
13704    /// are counting.
13705    #[test]
13706    fn a_dictionary_reads_the_same_whether_its_blocks_say_where_they_are() {
13707        let spellings = (0..3_000)
13708            .map(|index| format!("value {index:08} {}", "y".repeat(index % 40)))
13709            .collect::<Vec<_>>();
13710        let mut read = Vec::new();
13711        for layout in ["outside", "inside", "behind"] {
13712            let mut dictionary = GlobalDictionary::new();
13713            for text in &spellings {
13714                dictionary.code(text).expect("a code for every spelling");
13715            }
13716            dictionary.finish_blocks().expect("the last block encodes");
13717            let order = dictionary.ranked(None).expect("a sorted order");
13718            // Where the blocks go if they start at `from` and follow one another.
13719            let laid = |from: u64| {
13720                let mut at = from;
13721                dictionary
13722                    .blocks
13723                    .iter()
13724                    .map(|block| {
13725                        let place =
13726                            Placed { start: at, length: block.len() as u64, hash: checksum(block) };
13727                        at += block.len() as u64;
13728                        place
13729                    })
13730                    .collect::<Vec<_>>()
13731            };
13732            let payload = dictionary.blocks.concat();
13733            let scattered = layout != "behind";
13734            let (bytes, encoded, offset, length) = if layout == "outside" {
13735                let mut bytes = vec![0; HEADER as usize];
13736                bytes.extend_from_slice(&payload);
13737                let encoded = encode_global_dictionary(&dictionary, &order, &laid(HEADER), true)
13738                    .expect("an encoding");
13739                let offset = bytes.len() as u64;
13740                bytes.extend_from_slice(&encoded.index);
13741                bytes.extend_from_slice(&encoded.ranks);
13742                bytes.extend_from_slice(&encoded.grams);
13743                let length = encoded.index.len() + encoded.ranks.len() + encoded.grams.len();
13744                (bytes, encoded, offset, length)
13745            } else {
13746                // The index is the same length wherever the blocks are, so a first pass says where
13747                // the page ends and the second writes the places that follow it.
13748                let first = encode_global_dictionary(&dictionary, &order, &laid(0), scattered)
13749                    .expect("an encoding");
13750                let body = (first.index.len() + first.ranks.len() + first.grams.len()) as u64;
13751                let encoded = encode_global_dictionary(&dictionary, &order, &laid(body), scattered)
13752                    .expect("an encoding");
13753                let mut bytes = encoded.index.clone();
13754                bytes.extend_from_slice(&encoded.ranks);
13755                bytes.extend_from_slice(&encoded.grams);
13756                bytes.extend_from_slice(&payload);
13757                let length = bytes.len();
13758                (bytes, encoded, 0, length)
13759            };
13760            let path = path(&format!("blocks-{layout}"));
13761            fs::write(&path, &bytes).expect("the dictionary is written on its own");
13762            let file = Arc::new(File::open(&path).expect("it opens again"));
13763            let page = Page {
13764                offset,
13765                length: u32::try_from(length).expect("a test dictionary is small"),
13766                hash: checksum(&encoded.index),
13767            };
13768            let opened =
13769                open_global_dictionary(file, page, &LogicalType::Varchar, TEXT_KEEP_BUDGET)
13770                    .expect("a dictionary laid out either way opens");
13771            let mut swept: Vec<Vec<u8>> = Vec::new();
13772            let mut at = 0;
13773            while at < opened.len() {
13774                at = opened
13775                    .sweep_text(at, opened.len(), &mut |_index: usize, text: &[u8]| {
13776                        swept.push(text.to_vec());
13777                        Ok(())
13778                    })
13779                    .expect("a sweep reads");
13780            }
13781            fs::remove_file(&path).expect("clean up");
13782            read.push(swept);
13783        }
13784        let wanted =
13785            spellings.iter().map(|text| text.as_bytes().to_vec()).collect::<Vec<Vec<u8>>>();
13786        assert_eq!(read[0], wanted, "the blocks outside the page hold the values");
13787        assert_eq!(read[1], read[0], "the blocks inside the page hold the same values");
13788        assert_eq!(read[2], read[0], "the blocks behind one another hold the same values");
13789    }
13790
13791    /// A dictionary at its budget sweeps without keeping, and still answers what it answered.
13792    ///
13793    /// The budget is a quarter of a gigabyte in a running database, which is a fine size for a real
13794    /// column and no size at all for a test, so this opens the same dictionary a second time with a
13795    /// budget of zero. That is the shape of the hundred million row case: `URL` fills the budget
13796    /// somewhere in the middle of itself and everything past that point is read and dropped, which
13797    /// costs the decode again and holds none of it.
13798    #[test]
13799    fn a_dictionary_at_its_budget_sweeps_without_keeping() {
13800        let path = path("dictionary-budget");
13801        let spellings = (0..2_500)
13802            .map(|index| Value::Varchar(format!("value {index:08} {}", "y".repeat(index % 40))))
13803            .collect::<Vec<_>>();
13804        let mut writer =
13805            Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
13806                .expect("new file");
13807        for part in spellings.chunks(1_024) {
13808            writer
13809                .append(
13810                    &Chunk::new(vec![
13811                        Vector::from_values(LogicalType::Varchar, part).expect("strings"),
13812                    ])
13813                    .expect("one column"),
13814                )
13815                .expect("stripe written");
13816        }
13817        writer.finish().expect("commit");
13818
13819        let reader = Reader::open(&path).expect("valid directory");
13820        let page = reader.table.dictionaries[0].expect("a string column has one");
13821        let file = Arc::clone(&reader.file);
13822        let starved = open_global_dictionary(file, page, &LogicalType::Varchar, 0)
13823            .expect("a dictionary opens whatever it may keep");
13824
13825        let resting = starved.footprint();
13826        let mut swept: Vec<Vec<u8>> = Vec::new();
13827        let mut at = 0;
13828        while at < starved.len() {
13829            at = starved
13830                .sweep_text(at, starved.len(), &mut |_index: usize, text: &[u8]| {
13831                    swept.push(text.to_vec());
13832                    Ok(())
13833                })
13834                .expect("a sweep reads");
13835        }
13836        assert_eq!(swept.len(), spellings.len(), "a starved sweep still reads every value");
13837        assert_eq!(starved.footprint(), resting, "and keeps no block it decoded");
13838
13839        let generous = reader.dictionary(0).expect("read").expect("a string column has one");
13840        let read = (0..generous.len())
13841            .map(|code| generous.try_bytes_at(code).expect("read").expect("a value").to_vec())
13842            .collect::<Vec<_>>();
13843        assert_eq!(swept, read, "a starved sweep answers what a point read answers");
13844        fs::remove_file(path).expect("remove scratch file");
13845    }
13846
13847    #[test]
13848    fn damaged_membership_cannot_skip_a_string_page() {
13849        let path = path("damaged-membership");
13850        let mut writer = Writer::create(
13851            &path,
13852            "items",
13853            vec![
13854                Field::required("id", LogicalType::Integer),
13855                Field::new("text", LogicalType::Varchar),
13856            ],
13857        )
13858        .expect("new file");
13859        writer.append(&sample()).expect("stripe written");
13860        writer.finish().expect("commit");
13861
13862        let reader = Reader::open(&path).expect("valid directory");
13863        let membership = reader.table.stripes[0].memberships.get(1).expect("string membership");
13864        let mut file = OpenOptions::new().write(true).open(&path).expect("open membership page");
13865        file.seek(SeekFrom::Start(membership.offset)).expect("membership start");
13866        file.write_all(&[255]).expect("damage membership");
13867        let error = reader.skips_codes(0, 1, &[3]).expect_err("corruption must not skip rows");
13868        assert!(error.message().contains("membership page checksum differs"), "{error}");
13869        fs::remove_file(path).expect("remove scratch file");
13870    }
13871
13872    #[test]
13873    fn membership_delta_stream_is_sorted_exact_and_bounded() {
13874        let unique = unique_codes(&[900, 4, 4, 72, 9, u32::MAX]);
13875        assert_eq!(unique, [4, 9, 72, 900, u32::MAX]);
13876        let encoded = encode_membership(&unique);
13877        assert_eq!(
13878            decode_membership(&encoded).expect("valid membership"),
13879            [4, 9, 72, 900, u32::MAX]
13880        );
13881        // A stripe's index is the union of its parts', so a code in two of them is in it once and
13882        // the result is still one ascending run of deltas.
13883        let merged = merged_codes(vec![vec![4, 900], vec![9, 900, u32::MAX], vec![72]]);
13884        assert_eq!(merged, [4, 9, 72, 900, u32::MAX]);
13885        assert_eq!(
13886            decode_membership(&encode_membership(&merged)).expect("valid membership"),
13887            unique
13888        );
13889        assert!(decode_membership(&[1, 0x80]).is_err(), "a truncated varint is invalid");
13890        assert!(
13891            decode_membership(&[1, 0xff, 0xff, 0xff, 0xff, 0x10]).is_err(),
13892            "a value past u32 is invalid"
13893        );
13894    }
13895
13896    #[test]
13897    fn a_global_dictionary_may_be_larger_than_one_column_page() {
13898        let dictionary = Page {
13899            offset: HEADER,
13900            length: u32::try_from(MAX_PAGE + 1).expect("the page bound fits on disk"),
13901            hash: 0,
13902        };
13903        let table = Table {
13904            name: "items".to_owned(),
13905            fields: vec![Field::new("text", LogicalType::Varchar)],
13906            stripes: Vec::new(),
13907            rows: 0,
13908            dictionaries: vec![Some(dictionary)],
13909            dictionary_payloads: Vec::new(),
13910            distincts: vec![None],
13911            frequencies: vec![None],
13912            pair_frequencies: Vec::new(),
13913            frequency_texts: Vec::new(),
13914            host_groups: None,
13915            clustering: None,
13916            generation: 1,
13917            sections: Vec::new(),
13918        };
13919        let directory = encode_directory(&table).expect("directory");
13920        let file_size = dictionary.offset + u64::from(dictionary.length) + 1;
13921
13922        let decoded = decode_directory(&directory, file_size).expect("large lazy dictionary");
13923        assert_eq!(decoded.dictionaries[0].expect("dictionary").length, dictionary.length);
13924    }
13925
13926    #[test]
13927    fn a_column_with_one_value_everywhere_costs_almost_nothing_a_row() {
13928        let path = path("constant-codes");
13929        let mut writer =
13930            Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
13931                .expect("new file");
13932        let empty = vec![Value::Varchar(String::new()); 1024];
13933        for _ in 0..4 {
13934            let column = Vector::from_values(LogicalType::Varchar, &empty).expect("strings");
13935            writer.append(&Chunk::new(vec![column]).expect("one column")).expect("a part");
13936        }
13937        writer.finish().expect("commit");
13938
13939        let reader = Reader::open(&path).expect("valid directory");
13940        let pages = reader.layout().columns.first().expect("one column").pages;
13941        // This column used to cost four bytes a row, 16,384 of them, the same as a column of four
13942        // thousand distinct URLs would. The cascade calls each part a constant, so what is left is
13943        // a tag, a count and the value, and the row count stops being what drives the number.
13944        assert!(pages < 256, "{pages} bytes of pages for 4,096 rows of one value");
13945        let read = reader.read(3, &[0]).expect("the last part back");
13946        assert_eq!(read.value_at(0, 0), Value::Varchar(String::new()));
13947        assert_eq!(read.value_at(1023, 0), Value::Varchar(String::new()));
13948        fs::remove_file(path).expect("remove scratch file");
13949    }
13950
13951    #[test]
13952    fn a_cascade_value_too_wide_for_its_column_is_refused_rather_than_cut() {
13953        // What a damaged page looks like from here: the cascade decoded, so the bytes are not
13954        // truncated, but the values do not belong to the column the directory says they do.
13955        let over = vec![i64::from(i32::MAX) + 1];
13956        let error = narrowed(&LogicalType::Integer, over).expect_err("a page that disagrees");
13957        assert!(format!("{error}").contains("not of its type"), "{error}");
13958        assert!(narrowed(&LogicalType::BigInt, vec![i64::MIN]).is_ok(), "bigint holds all of i64");
13959        assert!(narrowed(&LogicalType::Varchar, vec![0]).is_err(), "strings are not integers");
13960    }
13961
13962    #[test]
13963    fn a_code_stream_the_cascade_cannot_shrink_is_left_alone() {
13964        // A shift register rather than a run, because an arithmetic run is the one wide shape the
13965        // cascade does shrink. This is what a column with tens of millions of distinct values hands
13966        // over: full width codes with no order to them.
13967        let mut state: u32 = 0x9e37_79b9;
13968        let spread: Vec<u32> = (0..1024)
13969            .map(|_| {
13970                state ^= state << 13;
13971                state ^= state >> 17;
13972                state ^= state << 5;
13973                state
13974            })
13975            .collect();
13976        assert_eq!(encoded_codes(&spread).expect("no failure"), None);
13977        let near: Vec<u32> = (0..1024).collect();
13978        let coded = encoded_codes(&near).expect("no failure").expect("counting up is packable");
13979        assert!(coded.len() < near.len() * 4, "{} bytes for a run of 1,024", coded.len());
13980    }
13981
13982    /// The columns of a stripe are encoded on whichever thread got to them, so the one thing that
13983    /// must not depend on which thread that was is the file. Two writes of the same rows are
13984    /// compared byte for byte rather than value for value, because a dictionary that two columns
13985    /// somehow shared would still read back correctly and would hand out its codes in the order the
13986    /// threads happened to run in, which is exactly what this is here to catch.
13987    #[test]
13988    fn two_writes_of_the_same_rows_give_the_same_bytes() {
13989        fn written(path: &PathBuf) {
13990            let fields = (0..40)
13991                .map(|column| {
13992                    let ty =
13993                        if column % 4 == 0 { LogicalType::Varchar } else { LogicalType::BigInt };
13994                    Field::new(format!("c{column}"), ty)
13995                })
13996                .collect::<Vec<_>>();
13997            let mut writer = Writer::create(path, "wide", fields).expect("new file");
13998            for part in 0..70_u64 {
13999                let columns = (0..40)
14000                    .map(|column| {
14001                        let values = (0..64_u64)
14002                            .map(|row| {
14003                                let seed = part.wrapping_mul(31).wrapping_add(row);
14004                                if column % 4 == 0 {
14005                                    Value::Varchar(format!("v{}", seed % 17))
14006                                } else {
14007                                    Value::BigInt(i64::try_from(seed % 97).expect("small"))
14008                                }
14009                            })
14010                            .collect::<Vec<_>>();
14011                        let ty = if column % 4 == 0 {
14012                            LogicalType::Varchar
14013                        } else {
14014                            LogicalType::BigInt
14015                        };
14016                        Vector::from_values(ty, &values).expect("a column")
14017                    })
14018                    .collect::<Vec<_>>();
14019                writer.append(&Chunk::new(columns).expect("forty columns")).expect("a part");
14020            }
14021            writer.finish().expect("commit");
14022        }
14023
14024        let first = path("repeatable-one");
14025        let second = path("repeatable-two");
14026        written(&first);
14027        written(&second);
14028        let left = fs::read(&first).expect("the first file");
14029        let right = fs::read(&second).expect("the second file");
14030        assert_eq!(left.len(), right.len(), "two writes of the same rows differ in length");
14031        assert!(left == right, "two writes of the same rows differ in their bytes");
14032
14033        // And the rows are still there, since a pair of identically wrong files would pass the
14034        // comparison above on its own.
14035        let reader = Reader::open(&first).expect("valid directory");
14036        assert_eq!(reader.table().rows(), 70 * 64);
14037        let read = reader.read(0, &[0, 1]).expect("the first part back");
14038        assert_eq!(read.value_at(0, 0), Value::Varchar("v0".to_owned()));
14039        assert_eq!(read.value_at(0, 1), Value::BigInt(0));
14040        fs::remove_file(first).expect("remove scratch file");
14041        fs::remove_file(second).expect("remove scratch file");
14042    }
14043
14044    /// Three tables of different shapes in one file, read back by name.
14045    fn three_tables(path: &PathBuf) {
14046        let writer = Writer::create(
14047            path,
14048            "region",
14049            vec![
14050                Field::new("r_key", LogicalType::Integer),
14051                Field::new("r_name", LogicalType::Varchar),
14052            ],
14053        )
14054        .expect("new file");
14055        let mut writer = writer;
14056        writer
14057            .append(
14058                &Chunk::new(vec![
14059                    Vector::from_values(
14060                        LogicalType::Integer,
14061                        &[Value::Integer(0), Value::Integer(1)],
14062                    )
14063                    .expect("keys"),
14064                    Vector::from_values(
14065                        LogicalType::Varchar,
14066                        &[Value::Varchar("AFRICA".to_owned()), Value::Varchar("ASIA".to_owned())],
14067                    )
14068                    .expect("names"),
14069                ])
14070                .expect("two columns"),
14071            )
14072            .expect("a part");
14073        let mut writer = writer
14074            .next("empty", vec![Field::new("nothing", LogicalType::BigInt)])
14075            .expect("a second table");
14076        writer
14077            .append(
14078                &Chunk::new(vec![
14079                    Vector::from_values(LogicalType::BigInt, &[Value::BigInt(7)]).expect("a row"),
14080                ])
14081                .expect("one column"),
14082            )
14083            .expect("a part");
14084        let mut writer =
14085            writer.next("wide", vec![Field::new("n", LogicalType::BigInt)]).expect("a third table");
14086        for part in 0..70_i64 {
14087            let values = (0..64).map(|row| Value::BigInt(part * 64 + row)).collect::<Vec<_>>();
14088            writer
14089                .append(
14090                    &Chunk::new(vec![
14091                        Vector::from_values(LogicalType::BigInt, &values).expect("a column"),
14092                    ])
14093                    .expect("one column"),
14094                )
14095                .expect("a part");
14096        }
14097        writer.finish().expect("commit");
14098    }
14099
14100    #[test]
14101    fn three_tables_in_one_file_read_back_by_name() {
14102        let file = path("three-tables");
14103        three_tables(&file);
14104        let catalog = Catalog::open(&file).expect("a committed catalog");
14105        assert_eq!(catalog.names().collect::<Vec<_>>(), ["region", "empty", "wide"]);
14106
14107        let region = catalog.table("region").expect("the first table");
14108        assert_eq!(region.table().rows(), 2);
14109        assert_eq!(
14110            region.read(0, &[1]).expect("names").value_at(1, 0),
14111            Value::Varchar("ASIA".to_owned())
14112        );
14113
14114        let wide = catalog.table("wide").expect("the third table");
14115        assert_eq!(wide.table().rows(), 70 * 64);
14116        assert_eq!(wide.read(0, &[0]).expect("the first part").value_at(0, 0), Value::BigInt(0));
14117
14118        // The middle table is reached without the one after it having been touched, which is what
14119        // a directory per table buys over one directory of everything.
14120        let empty = catalog.table("empty").expect("the second table");
14121        assert_eq!(empty.table().rows(), 1);
14122        assert_eq!(empty.read(0, &[0]).expect("the row").value_at(0, 0), Value::BigInt(7));
14123
14124        fs::remove_file(file).expect("remove scratch file");
14125    }
14126
14127    #[test]
14128    fn a_name_the_file_does_not_hold_is_an_error_rather_than_the_first_table() {
14129        let file = path("three-tables-missing");
14130        three_tables(&file);
14131        let catalog = Catalog::open(&file).expect("a committed catalog");
14132        let error = catalog.table("nation").expect_err("no such table");
14133        assert!(error.message().contains("nation"), "{}", error.message());
14134        fs::remove_file(file).expect("remove scratch file");
14135    }
14136
14137    #[test]
14138    fn a_file_of_three_tables_will_not_open_as_one() {
14139        let file = path("three-tables-unnamed");
14140        three_tables(&file);
14141        let error = Reader::open(&file).expect_err("more than one table");
14142        assert!(error.message().contains("more than one table"), "{}", error.message());
14143        fs::remove_file(file).expect("remove scratch file");
14144    }
14145
14146    /// One column per storage width, because the width is what decides how many bytes a row costs.
14147    #[test]
14148    fn decimals_of_every_storage_width_round_trip() {
14149        let file = path("decimals");
14150        let widths = [(4_u8, 2_u8), (9, 2), (18, 4), (38, 6)];
14151        let fields = widths
14152            .iter()
14153            .enumerate()
14154            .map(|(index, (width, scale))| {
14155                Field::new(
14156                    format!("d{index}"),
14157                    LogicalType::decimal(*width, *scale).expect("a decimal type"),
14158                )
14159            })
14160            .collect::<Vec<_>>();
14161        let mut writer = Writer::create(&file, "money", fields).expect("new file");
14162        let rows: [i128; 3] = [-1234, 0, 999];
14163        let columns = widths
14164            .iter()
14165            .map(|(width, scale)| {
14166                let values = rows
14167                    .iter()
14168                    .map(|unscaled| Value::Decimal {
14169                        unscaled: *unscaled,
14170                        width: *width,
14171                        scale: *scale,
14172                    })
14173                    .collect::<Vec<_>>();
14174                Vector::from_values(
14175                    LogicalType::decimal(*width, *scale).expect("a decimal type"),
14176                    &values,
14177                )
14178                .expect("a decimal column")
14179            })
14180            .collect::<Vec<_>>();
14181        writer.append(&Chunk::new(columns).expect("four columns")).expect("a part");
14182        writer.finish().expect("commit");
14183
14184        let reader = Reader::open(&file).expect("a committed file");
14185        for (index, (width, scale)) in widths.iter().enumerate() {
14186            assert_eq!(
14187                reader.table().fields()[index].ty,
14188                LogicalType::decimal(*width, *scale).expect("a decimal type"),
14189                "column {index} came back as another type"
14190            );
14191            let column = reader.read(0, &[index]).expect("the column");
14192            for (row, unscaled) in rows.iter().enumerate() {
14193                assert_eq!(
14194                    column.value_at(row, 0),
14195                    Value::Decimal { unscaled: *unscaled, width: *width, scale: *scale },
14196                    "column {index} row {row}"
14197                );
14198            }
14199        }
14200        fs::remove_file(file).expect("remove scratch file");
14201    }
14202
14203    #[test]
14204    fn two_tables_of_one_name_are_refused_before_anything_is_committed() {
14205        let file = path("two-of-a-name");
14206        let writer = Writer::create(&file, "t", vec![Field::new("a", LogicalType::BigInt)])
14207            .expect("new file");
14208        let error = writer
14209            .next("t", vec![Field::new("a", LogicalType::BigInt)])
14210            .expect_err("the same name twice");
14211        assert!(error.message().contains("same name"), "{}", error.message());
14212        fs::remove_file(file).expect("remove scratch file");
14213    }
14214
14215    #[test]
14216    fn opening_the_catalog_reads_no_table_directory() {
14217        let file = path("catalog-only");
14218        three_tables(&file);
14219        let catalog = Catalog::open(&file).expect("a committed catalog");
14220        // The header and one slot, and nothing under it. The third table's directory covers seventy
14221        // stripes and reading it here would be the whole point of the two levels thrown away.
14222        assert_eq!(catalog.opening.reads, 2, "opening the catalog read more than the slot");
14223        assert_eq!(catalog.names().len(), 3);
14224        fs::remove_file(file).expect("remove scratch file");
14225    }
14226
14227    /// The checksum answers what it has always answered, at every length its branches split on.
14228    ///
14229    /// This is a compatibility test rather than a correctness one. Nothing about the hash has to be
14230    /// any particular function, but a file already on disk carries the answers the version that
14231    /// wrote it gave, so a change here is a change that makes every stored file fail to verify. The
14232    /// lengths are the ones the code makes decisions about: nothing, under a block, a block exactly,
14233    /// a block and a word, a word and a half word, and a half word and a byte.
14234    ///
14235    /// The empty answer is the published xxHash64 vector for an empty input at seed zero, which is
14236    /// also a check that this is the function it says it is.
14237    #[test]
14238    fn the_checksum_answers_what_it_has_always_answered() {
14239        let bytes: Vec<u8> =
14240            (0..1000_u32).map(|at| (at.wrapping_mul(31).wrapping_add(7) % 251) as u8).collect();
14241        for (length, expected) in [
14242            (0, 0xef46_db37_51d8_e999),
14243            (1, 0xa96c_7f0c_e858_bbb7),
14244            (3, 0x56e6_9576_32a4_87f9),
14245            (4, 0xc60d_15b1_e3ff_8f04),
14246            (5, 0x8088_1585_8624_dd4e),
14247            (7, 0xafbe_fc3d_6c6f_9a8e),
14248            (8, 0x3da5_c7aa_2696_83e0),
14249            (9, 0x465e_c429_b13c_3892),
14250            (15, 0xdee8_9d8a_065a_6233),
14251            (16, 0x1330_489a_7767_9c80),
14252            (31, 0x3391_303d_485e_846e),
14253            (32, 0x40b7_aff7_5d45_bbc8),
14254            (33, 0x4997_cae4_951c_17a5),
14255            (39, 0x5807_28fd_5c14_5739),
14256            (40, 0xf95c_f6f5_c08a_3d3b),
14257            (63, 0x2944_b4da_fc69_b206),
14258            (64, 0xbb76_f6ef_19bd_5a1b),
14259            (65, 0x814e_0c65_4a9f_d640),
14260            (127, 0x00de_aab1_31cf_f89b),
14261            (1000, 0x9e33_00c1_cde3_c58d),
14262        ] {
14263            assert_eq!(checksum(&bytes[..length]), expected, "the checksum of {length} bytes");
14264        }
14265        assert_eq!(checksum(b"the quick brown fox jumps over the lazy dog"), 0xed71_4233_c5a9_a792);
14266    }
14267    /// A declared order survives the file, and a table that declared none stays as it was.
14268    ///
14269    /// The second half is the one worth a test. The clustering section is written only when there
14270    /// is a declaration, so a file of two tables where one is clustered exercises both the present
14271    /// and the absent branch of the decoder in one directory, which is where a length bug would
14272    /// show up as one table reading the other's bytes.
14273    #[test]
14274    fn a_declared_order_comes_back_out_of_the_file() {
14275        let path = path("clustered");
14276        let shipped = vec![
14277            Field::new("key", LogicalType::BigInt),
14278            Field::new("line", LogicalType::Integer),
14279            Field::new("shipdate", LogicalType::Date),
14280        ];
14281        let plain = vec![Field::new("a", LogicalType::Integer)];
14282        let stage_zero = Clustering::new(vec![2, 0, 1], Width::Month, &shipped).expect("valid");
14283
14284        let mut writer = Writer::create(&path, "lineitem", shipped)
14285            .expect("new file")
14286            .declare(stage_zero.clone())
14287            .expect("the columns are the table's");
14288        let column = |ty: LogicalType, values: &[Value]| {
14289            Vector::from_values(ty, values).expect("the values match the type")
14290        };
14291        writer
14292            .append(
14293                &Chunk::new(vec![
14294                    column(
14295                        LogicalType::BigInt,
14296                        &[Value::BigInt(0), Value::BigInt(1), Value::BigInt(2), Value::BigInt(3)],
14297                    ),
14298                    column(
14299                        LogicalType::Integer,
14300                        &[
14301                            Value::Integer(1),
14302                            Value::Integer(1),
14303                            Value::Integer(1),
14304                            Value::Integer(1),
14305                        ],
14306                    ),
14307                    column(
14308                        LogicalType::Date,
14309                        &[Value::Date(0), Value::Date(1), Value::Date(2), Value::Date(3)],
14310                    ),
14311                ])
14312                .expect("three columns"),
14313            )
14314            .expect("four rows");
14315        let mut writer = writer.next("nation", plain).expect("a second table");
14316        writer
14317            .append(
14318                &Chunk::new(vec![column(LogicalType::Integer, &[Value::Integer(7)])])
14319                    .expect("one column"),
14320            )
14321            .expect("one row");
14322        writer.finish().expect("commit");
14323
14324        let catalog = Catalog::open(&path).expect("reopen");
14325        let lineitem = catalog.table("lineitem").expect("the clustered table");
14326        assert_eq!(lineitem.table().clustering(), Some(&stage_zero));
14327        let nation = catalog.table("nation").expect("the plain table");
14328        assert_eq!(nation.table().clustering(), None, "nobody declared one here");
14329
14330        // And the rows are still the rows, because the section goes on the end of the directory
14331        // and the easy way to break that is to leave the cursor somewhere the next read trusts.
14332        assert_eq!(lineitem.table().rows(), 4);
14333        assert_eq!(nation.table().rows(), 1);
14334        fs::remove_file(&path).ok();
14335    }
14336
14337    /// A declaration naming a column the table does not have is refused where it is made.
14338    #[test]
14339    fn a_declaration_off_the_end_of_the_table_never_reaches_the_file() {
14340        let path = path("clustered-bad");
14341        let writer = Writer::create(&path, "items", vec![Field::new("a", LogicalType::Integer)])
14342            .expect("new file");
14343        let four =
14344            (0..4).map(|at| Field::new(format!("c{at}"), LogicalType::Integer)).collect::<Vec<_>>();
14345        let wrong = Clustering::new(vec![3], Width::Exact, &four).expect("valid against four");
14346        assert!(writer.declare(wrong).is_err(), "the table has one column, not four");
14347        fs::remove_file(&path).ok();
14348    }
14349
14350    /// The sorted order is the byte order, whatever the values do before they differ.
14351    ///
14352    /// The values here are the shape the sort is built for and the shape a comparison sort is worst
14353    /// at: a common scheme, a handful of hosts, and a path that only decides the pair thirty bytes
14354    /// in. They also cover what the bucketing has to get right at the edges, which is a value that
14355    /// has run out where another carries on, the empty value, and enough entries to take the range
14356    /// down through several passes and out the bottom into the comparison that finishes it.
14357    #[test]
14358    fn the_dictionary_order_is_the_byte_order_however_deep_the_values_agree() {
14359        let mut values = vec![String::new(), "http://".to_owned()];
14360        for host in 0..7 {
14361            for path in 0..30 {
14362                values.push(format!("http://example{host}.test/page/{path:04}/index.html"));
14363                values.push(format!("http://example{host}.test/page/{path:04}"));
14364            }
14365        }
14366        values.push("http://example0.test/page/0000/index.htmlx".to_owned());
14367
14368        let mut dictionary = GlobalDictionary::new();
14369        for value in &values {
14370            dictionary.code(value).expect("a code for every value");
14371        }
14372        dictionary.finish_blocks().expect("the last block encodes");
14373        let ranked = dictionary.ranked(None).expect("a sorted order");
14374        assert_eq!(ranked.len(), values.len(), "one entry a distinct value");
14375
14376        let spellings = dictionary_values(&dictionary);
14377        let seen = ranked
14378            .iter()
14379            .map(|&(_, code)| {
14380                String::from_utf8(spellings[code as usize].clone()).expect("text in, text out")
14381            })
14382            .collect::<Vec<_>>();
14383        let mut wanted = values.clone();
14384        wanted.sort_unstable();
14385        assert_eq!(seen, wanted, "the order is the order the bytes give");
14386
14387        for &(carried, code) in &ranked {
14388            let value = &spellings[code as usize];
14389            assert_eq!(carried, head(value), "the head belongs to the value it is filed with");
14390        }
14391    }
14392
14393    /// Picking the commonest entries leaves exactly what sorting all of them and cutting left.
14394    ///
14395    /// The counts here are deliberately full of ties, including a tie that straddles the cut, which
14396    /// is where a partition and a sort can disagree if the comparison they are given is not total.
14397    #[test]
14398    fn the_commonest_entries_are_the_ones_a_full_sort_would_have_kept() {
14399        let entry =
14400            |value: u32, count: u64| FrequencyEntry { value: FrequencyValue::Code(value), count };
14401        let mut all = (0..FREQUENCY_ENTRIES as u32 * 3)
14402            .map(|code| entry(code, u64::from(code % 7) + 1))
14403            .collect::<Vec<_>>();
14404        all.push(FrequencyEntry { value: FrequencyValue::Null, count: 4 });
14405
14406        let mut sorted = all.clone();
14407        sorted.sort_unstable_by(|left, right| {
14408            right.count.cmp(&left.count).then_with(|| frequency_order(left.value, right.value))
14409        });
14410        let wanted_omitted = sorted[FREQUENCY_ENTRIES].count;
14411        sorted.truncate(FREQUENCY_ENTRIES);
14412
14413        let mut picked = all.clone();
14414        let omitted = keep_most_frequent(&mut picked);
14415        assert_eq!(omitted, wanted_omitted, "the largest count that did not make the cut");
14416        assert_eq!(picked.len(), FREQUENCY_ENTRIES, "the cut is where it says it is");
14417        assert!(
14418            picked
14419                .iter()
14420                .zip(&sorted)
14421                .all(|(one, two)| one.value == two.value && one.count == two.count),
14422            "the same entries in the same order"
14423        );
14424
14425        let mut short = all[..FREQUENCY_ENTRIES - 1].to_vec();
14426        let omitted = keep_most_frequent(&mut short);
14427        assert_eq!(omitted, 0, "nothing is omitted when everything fits");
14428        assert!(short.windows(2).all(|pair| pair[0].count >= pair[1].count), "still in order");
14429    }
14430
14431    /// A dictionary too small to bucket, and one with nothing in it, come back in order too.
14432    #[test]
14433    fn a_short_dictionary_sorts_without_a_bucketing_pass() {
14434        let empty = GlobalDictionary::new();
14435        assert!(empty.ranked(None).expect("an empty order").is_empty(), "nothing in, nothing out");
14436
14437        let mut dictionary = GlobalDictionary::new();
14438        for value in ["pear", "apple", "", "apples", "app"] {
14439            dictionary.code(value).expect("a code for every value");
14440        }
14441        dictionary.finish_blocks().expect("the one block encodes");
14442        let spellings = dictionary_values(&dictionary);
14443        let seen = dictionary
14444            .ranked(None)
14445            .expect("a sorted order")
14446            .iter()
14447            .map(|&(_, code)| spellings[code as usize].clone())
14448            .collect::<Vec<_>>();
14449        let wanted: Vec<Vec<u8>> =
14450            [&b""[..], b"app", b"apple", b"apples", b"pear"].iter().map(|v| v.to_vec()).collect();
14451        assert_eq!(seen, wanted, "shorter first where one runs out inside another");
14452    }
14453}