Skip to main content

rudb_native/
lib.rs

1//! Rudb's single-file columnar snapshot format.
2//!
3//! A committed directory names independently readable column pages. It has two levels: a catalog
4//! directory naming every table in the file, which is what a footer slot points at and what opening
5//! a database reads, and one directory per table under it holding that table's stripes, pages and
6//! statistics. One slot write publishes all of them, so a commit is atomic across tables.
7//!
8//! This version handles scalar columns; the file header has two generation slots so an unfinished
9//! replacement directory cannot hide the last complete one. See
10//! `spec/storage-v3/12-many-tables-in-one-file.md`.
11//!
12//! # Parts and stripes
13//!
14//! A part is one appended chunk, which is a thousand rows, and it is the unit a scan decodes and
15//! hands to the pipeline. A stripe is sixty four parts, and it is the unit the directory describes
16//! and the unit the file is laid out in: one page per column per stripe, holding that column's
17//! sixty four part payloads end to end.
18//!
19//! The two are separate because they are sized by different pressures. A part wants to be small
20//! because it is a vector and vectors live in cache. A stripe wants to be large because everything
21//! the directory holds is per stripe and the directory is one buffer that has to be read and
22//! decoded before a single row can be answered. A hundred million rows of the hundred and five
23//! column ClickBench table is ninety seven thousand parts, and a directory with a page entry and a
24//! pair of bounds per part per column is several hundred megabytes, which is what made that load
25//! fail before this split existed. Sixty four parts to a stripe divides that by sixty four.
26//!
27//! Where the parts of a page start is not in the directory either, for the same reason. Each
28//! stripe writes one index page holding a length and a checksum per part per column, and a reader
29//! preads the sixty four entries belonging to the column it wants. A scan reads the whole column
30//! page once and slices it; a sparse row fetch reads the index entries and then only the part it
31//! needs.
32
33#![forbid(unsafe_code)]
34
35use std::borrow::Cow;
36use std::cmp::Ordering;
37use std::collections::{BTreeMap, HashMap, VecDeque};
38use std::fs::{File, OpenOptions};
39use std::io::{Read, Seek, SeekFrom};
40use std::mem::{size_of, size_of_val};
41use std::path::Path;
42use std::slice;
43use std::sync::atomic::{AtomicBool, AtomicUsize, Ordering as Atomic};
44use std::sync::{Arc, Condvar, Mutex, OnceLock, PoisonError, Weak};
45
46use rudb_common::bounds::{self, Bound, Op, scaled_as};
47use rudb_common::{Clustering, Error, Field, LogicalType, PhysicalType, Result, Value, Width};
48use rudb_encoding::{bitpack, chooser, integer, string};
49use rudb_metrics::{LoadProfile, Stage};
50use rudb_storage::sieve::Sieve;
51use rudb_storage::{Probe, Range, Zone};
52use rudb_vector::string::StringColumn;
53use rudb_vector::validity::Validity;
54use rudb_vector::{Buffer, Chunk, Data, Packed, TextSource, Vector, search_below};
55
56mod distinct;
57pub mod graph;
58pub mod host;
59mod prepare;
60use prepare::Lent;
61pub mod section;
62pub mod stats;
63mod zones;
64
65pub use prepare::{Merged, Merger, Paged, Prepared, Preparer};
66pub use section::Section;
67pub use zones::{Common, Stripes, ascending, distincts};
68
69const MAGIC: &[u8; 8] = b"RUDBNV10";
70const DIRECTORY: &[u8; 8] = b"RUDBDI10";
71const CATALOG: &[u8; 8] = b"RUDBCA10";
72const NONZERO_COUNTS: &[u8; 8] = b"RUDBNZ10";
73const AGGREGATE_SUMS: &[u8; 8] = b"RUDBAG10";
74const DISTINCT_COUNTS: &[u8; 8] = b"RUDBDC10";
75const INTEGER_EXTREMES: &[u8; 8] = b"RUDBEX10";
76const COMPLETE_FREQUENCIES: &[u8; 8] = b"RUDBFQ10";
77const MAX_CATALOG_FREQUENCIES: usize = 64;
78const FORMAT: u32 = 29;
79
80/// Formats this build can open.
81///
82/// More than one, for the first time, and the reason is spec/graph/10-milestones.md's G1 exit
83/// criterion: a build with the section table in it has to open a file written before the section
84/// table existed, unchanged and without a rewrite. Formats 22 and 23 are those files, and both read
85/// as a table with an empty section table, which is exactly what section 3.1 says a table with no
86/// graph sections is.
87///
88/// All three of the older ones are readable for the same reason. What took the format from 22 to 23
89/// was tags for fourteen more column types, and a file written before that has none of them in it,
90/// so nothing in an older file is a tag this build cannot read. What took it from 23 to 24 is the
91/// section table, which a file written before it simply does not have. What takes it from 24 to 25
92/// is the view section on the end of the catalog, which an older file does not have either, and a
93/// catalog that ends where the tables end reads as a catalog with no views in it. What takes it
94/// from 25 to 26 is that a global dictionary's payload blocks now say where they are, and a file
95/// written before that has them behind one another, which [`open_global_dictionary`] reads by
96/// turning the ends it finds into the same places the newer files name outright. What takes it
97/// from 26 to 27 is that those blocks are written into the file as the load goes, between the
98/// stripes, rather than behind the dictionary's index at the end, so the dictionary's page is the
99/// index and the sorted order and nothing else. A format 26 file has its blocks inside the page,
100/// and the reader tells the two apart by whether the page has room left over for them.
101///
102/// Format 28 adds per-payload-block substring signatures to global string dictionaries. Older
103/// files have no signatures and use the ordinary exact string filter. Format 29 makes each
104/// signature four times as wide, which a dictionary says with [`DICTIONARY_WIDE_GRAMS`], and a
105/// format 28 file is read with the narrow ones it has.
106///
107/// This is not a general compatibility promise. Seven formats are readable because there was a
108/// specific reason for each, and the list shrinks again the moment the older ones stop being worth
109/// carrying.
110const READABLE: &[u32] = &[22, 23, 24, 25, 26, 27, 28, FORMAT];
111
112const HEADER: u64 = 80;
113const SLOT_BYTES: usize = 28;
114const MAX_PAGE: usize = 256 * 1024 * 1024;
115const MAX_DIRECTORY: usize = 128 * 1024 * 1024;
116const FREQUENCIES_V2: &[u8; 8] = b"RUDBFQ2\0";
117const FREQUENCIES: &[u8; 8] = b"RUDBFQ3\0";
118/// Inline spellings for string entries in the bounded frequency synopsis.
119///
120/// A planner usually asks about one literal such as the empty string. Without this block it opens
121/// a multi-million-value global dictionary and visits the payload blocks of every retained entry
122/// merely to compare that literal with at most 512 heavy hitters. The spellings are already in
123/// memory while the writer sorts the dictionary, so storing this bounded copy makes planning a
124/// directory read and leaves the dictionary unopened.
125const FREQUENCY_TEXTS: &[u8; 8] = b"RUDBFT1\0";
126/// Certified host aggregate state for the version-one anchored replacement expression.
127const HOST_GROUPS: &[u8; 8] = b"RUDBHG1\0";
128/// Exact leading counts for a bounded pair of dictionary-backed grouping keys.
129///
130/// This is a separate optional directory block rather than another frequency format. Readers that
131/// predate it still understand every earlier directory, and a table without a pair worth keeping
132/// writes no block at all.
133const PAIR_FREQUENCIES: &[u8; 8] = b"RUDBPF1\0";
134/// The clustering declaration, written after the frequencies and only when there is one.
135///
136/// No format bump for this, which is the convention the frequency section set in #728: a new
137/// optional trailing section with its own magic leaves every file that does not use it byte for
138/// byte what it was, and the version is bumped for a change to a layout that already exists, as
139/// #1029 did. A file with no declaration is the same bytes this build wrote yesterday.
140///
141/// The width byte in this block gained a fifth value for #1285, for a declaration that leaves the
142/// bucket to the row count, and that did not bump the format either. It is the one case where the
143/// reasoning needs saying out loud, because it is a new value in a layout that already exists
144/// rather than a new section. A build without it reading one of these says `clustering width
145/// tag differs` and refuses the table, which is what that message was written for. Bumping the
146/// format instead would have made every file this build writes unreadable to an older one, whether
147/// it has a declaration in it or not, to warn about a case that only arises when it does.
148const CLUSTERING: &[u8; 8] = b"RUDBCL1\0";
149/// The graph section table, written after the clustering declaration and written even when empty.
150///
151/// Same convention and the same reason as the block above it, with one difference: this one is
152/// always there, so a file written by this build says which sections it has rather than leaving a
153/// reader to infer it from where the bytes ran out. Section 3.1 of the graph spec is what makes
154/// that safe to add without a format bump, because a table with no sections answers every query
155/// the way it did before, only without the graph path.
156const SECTIONS: &[u8; 8] = b"RUDBSE1\0";
157/// How many bytes of each column's global dictionary live outside its page, written only when any do.
158///
159/// From format 27 a dictionary's payload blocks are written into the file while the load runs, so
160/// they sit between the stripes and the dictionary's page covers only its index and sorted order.
161/// Nothing needs the total to read the file, because the index names every block. It is here for
162/// what a file costs a column, which [`Reader::layout`] and the statistics budget both report, and
163/// which would otherwise lose most of the bytes of every large string column.
164const DICTIONARY_PAYLOADS: &[u8; 8] = b"RUDBDP1\0";
165
166/// The most sections one table's directory may name.
167///
168/// A relationship contributes at most three sections, so this bounds a table at a few thousand
169/// relationships, which is far past anything a schema has. The bound is here so that a torn
170/// directory naming four billion of them is refused at decode rather than turned into an
171/// allocation, the same reason the extent count has one.
172const MAX_SECTIONS: usize = 4096;
173const FREQUENCY_CANDIDATES: usize = 32_768;
174const FREQUENCY_ENTRIES: usize = 512;
175const FREQUENCY_BUILD_RANK: usize = 10;
176const FREQUENCY_ORDINALS: usize = 131_072;
177const MAX_PAIR_FREQUENCIES: usize = 1024;
178/// The most exact heavy-hitter text one column may copy into the directory.
179///
180/// A column with unusually large leading values keeps the old code-only synopsis instead. The
181/// optimization must never turn a valid load into a directory-size failure.
182const FREQUENCY_TEXT_BUDGET: usize = 1024 * 1024;
183/// The most threads the two per column passes at the end of a commit are spread over.
184///
185/// A table like `hits` has ninety numeric columns, so on a machine with more cores than this the
186/// cap is what decides how long the frequencies take rather than the columns are. It is here at all
187/// because each worker holds a candidate table and a decoded part, and a hundred of those at once
188/// on a narrow machine would be worse than waiting.
189const MAX_FREQUENCY_WORKERS: usize = 32;
190
191/// How many threads the passes at the end of a commit are spread over on this machine.
192fn close_workers() -> usize {
193    std::thread::available_parallelism().map_or(1, usize::from).min(MAX_FREQUENCY_WORKERS)
194}
195
196/// How many decoded bytes the global dictionaries closing at the same time may hold between them.
197///
198/// Closing a dictionary decodes every value it holds, sorts them and drops them, and #1356 took the
199/// columns one at a time so that five of them decoded at once were not the peak of a load. On the
200/// ClickBench `hits` 10M load that made the dictionaries 1.9 s of a 6.3 s load on the 32 core box,
201/// with `Referer`, `Title` and `URL` each most of a second on their own. A column is taken while the
202/// ones already closing leave room for it under this, and always when nothing else is closing, so
203/// every column of `hits` at 10M rows closes at once and `URL` at 100M, which is past this alone,
204/// still closes on its own.
205const CLOSE_DICTIONARY_BYTES: usize = 1 << 30;
206
207/// The most threads one stripe's encode is spread over.
208///
209/// Higher than the frequency cap because this is the load itself rather than a pass at the end of
210/// it, and the work is one column of sixty four parts, which is large enough that a thread that
211/// takes one is not a thread that was started for nothing. A machine with more cores than this has
212/// the rest of them on the Parquet read, which is still one thread and is the other half of #808.
213const MAX_ENCODE_WORKERS: usize = 32;
214
215/// How much a writer appends before it asks the kernel to start writing it to the device.
216///
217/// Without it every byte of a load waits in the page cache for the sync at the commit, and that
218/// sync was 1.3 to 1.7 s of a ClickBench `hits` 10M load of 8 to 9 s on the 32 core box. With it
219/// the device writes while the load is still encoding. Thirty two megabytes is a few stripes of
220/// `hits`, big enough that the call costs nothing next to the write, and small enough that what
221/// is left for the commit is one stretch.
222const WRITEBACK_STRETCH: u64 = 32 << 20;
223
224/// The most bytes one column of one part may spend on a membership sieve.
225///
226/// A part is a thousand rows, so a filter sized for every one of them being distinct is about
227/// thirteen hundred bytes and this never binds in practice. It is here so that a part that somehow
228/// arrives much wider than a vector cannot put an unbounded index in the file. What does bind is the
229/// rule in `Writer::encode_pages` that a sieve may not be as large as the part it indexes, which is a cap
230/// per column rather than one number for the whole file.
231const SIEVE_BUDGET: usize = 8 * 1024;
232
233/// The most bytes one end of a per part range may spend on a string.
234///
235/// A bound is allowed to be wider than the truth and never narrower, so a long string is cut down to
236/// this many bytes for the low end and cut down and then stepped up for the high end. The reason for
237/// a cap at all is that there are nine hundred and seventy four parts of a hundred and five columns
238/// in a million rows of ClickBench and `URL` runs to hundreds of bytes, so keeping every end whole
239/// would put more in the directory than the skipping is worth. Twenty four bytes is past the point
240/// where two URLs of the same site still look alike.
241const PART_BOUND_BYTES: usize = 24;
242
243fn io(error: std::io::Error) -> Error {
244    Error::io(error.to_string())
245}
246
247fn invalid(message: &str) -> Error {
248    Error::invalid_input(format!("invalid rudb native file: {message}"))
249}
250
251/// Adds a sequence of byte counts without an overflow the caller has to think about.
252fn sum(counts: impl Iterator<Item = u64>) -> u64 {
253    counts.fold(0, u64::saturating_add)
254}
255
256/// One column's span out of a per column list, or zero when the list is shorter than the column.
257fn span_bytes(spans: &[Span], at: usize) -> u64 {
258    spans.get(at).map_or(0, |span| u64::from(span.length))
259}
260
261/// One column's page out of a per column list, or zero when that column has no page at all.
262fn page_bytes(pages: &[Option<Page>], at: usize) -> u64 {
263    pages.get(at).and_then(Option::as_ref).map_or(0, Page::bytes)
264}
265
266/// Everything one column's global dictionary costs the file, its page and the blocks outside it.
267fn dictionary_bytes(table: &Table, at: usize) -> u64 {
268    page_bytes(&table.dictionaries, at)
269        .saturating_add(table.dictionary_payloads.get(at).copied().unwrap_or(0))
270}
271
272/// The xxHash64 of `bytes`, which is what every span this format stores is checked against.
273///
274/// It walks the input as chunks rather than as offsets into it, and that is the only thing about it
275/// worth a comment. The offset form reads `bytes[at..at + 8]`, and neither the slicing nor the
276/// `try_into` behind it can be proved in range by a compiler that does not know where `at` stopped,
277/// so each of the four lanes paid for a bounds check and a length check on every thirty two bytes.
278/// A chunk carries its own length, so both fold away and the loop is the multiplies and rotates it
279/// was meant to be. That loop runs over every byte of every span a query reads, which on ClickBench
280/// 8 is about five percent of the query.
281fn checksum(bytes: &[u8]) -> u64 {
282    seeded_checksum(bytes, 0)
283}
284
285/// A hundred and twenty eight bit name for `bytes`, as two xxHash64 walks under different seeds,
286/// with the format this build writes folded in so that a name made by one format is never taken
287/// for the name of a file in another.
288///
289/// For a caller outside this crate that has to name a file by what went into it, which is what a
290/// Parquet mirror's key is. See the global dictionary's use of the same pair for the arithmetic.
291#[must_use]
292pub fn content_name(bytes: &[u8]) -> u128 {
293    let seed = u64::from(FORMAT);
294    u128::from(seeded_checksum(bytes, seed)) << 64 | u128::from(seeded_checksum(bytes, !seed))
295}
296
297/// [`content_name`] of bytes that arrive in pieces, which gives the same name as the pieces joined.
298///
299/// A Parquet mirror is named for the file's footer, and that is 930 KB on the ten million row
300/// ClickBench file. Read whole to be hashed it is a freed megabyte in every process that opens the
301/// mirror, which the allocator keeps. Read a window at a time it is a window.
302#[derive(Debug, Clone)]
303pub struct ContentNamer {
304    seeds: [u64; 2],
305    lanes: [[u64; 4]; 2],
306    held: [u8; 32],
307    filled: usize,
308    length: u64,
309}
310
311impl Default for ContentNamer {
312    fn default() -> Self {
313        let seed = u64::from(FORMAT);
314        let seeds = [seed, !seed];
315        let lanes = seeds.map(|seed| {
316            [
317                seed.wrapping_add(XXH_P1).wrapping_add(XXH_P2),
318                seed.wrapping_add(XXH_P2),
319                seed,
320                seed.wrapping_sub(XXH_P1),
321            ]
322        });
323        Self { seeds, lanes, held: [0; 32], filled: 0, length: 0 }
324    }
325}
326
327impl ContentNamer {
328    /// Takes the next piece.
329    pub fn update(&mut self, mut bytes: &[u8]) {
330        self.length += bytes.len() as u64;
331        if self.filled > 0 {
332            let take = (32 - self.filled).min(bytes.len());
333            self.held[self.filled..self.filled + take].copy_from_slice(&bytes[..take]);
334            self.filled += take;
335            bytes = &bytes[take..];
336            if self.filled < 32 {
337                return;
338            }
339            let block = self.held;
340            self.lanes.iter_mut().for_each(|lanes| checksum_block(lanes, &block));
341            self.filled = 0;
342        }
343        let mut blocks = bytes.chunks_exact(32);
344        for block in blocks.by_ref() {
345            self.lanes.iter_mut().for_each(|lanes| checksum_block(lanes, block));
346        }
347        let rest = blocks.remainder();
348        self.held[..rest.len()].copy_from_slice(rest);
349        self.filled = rest.len();
350    }
351
352    /// The name of everything taken so far.
353    #[must_use]
354    pub fn finish(&self) -> u128 {
355        let rest = &self.held[..self.filled];
356        let [first, second] = [0, 1].map(|at| {
357            if self.length < 32 {
358                checksum_tail(self.seeds[at].wrapping_add(XXH_P5).wrapping_add(self.length), rest)
359            } else {
360                finish_checksum(self.lanes[at], rest, self.length)
361            }
362        });
363        u128::from(first) << 64 | u128::from(second)
364    }
365}
366
367/// The xxHash64 of `bytes` started from `seed`, which is the same walk with a different beginning.
368///
369/// A seed is here for one caller: a global dictionary decides whether two values are the same by
370/// their hashes rather than by their bytes, and one sixty four bit hash is not enough to do that
371/// with. Twenty million distinct values collide on sixty four bits about once in a hundred thousand
372/// loads, which for a wrong answer is far too often. Two hashes of the same value under different
373/// seeds are independent, so the pair is a hundred and twenty eight bits and the same arithmetic
374/// puts that at around one in 1e24.
375fn seeded_checksum(bytes: &[u8], seed: u64) -> u64 {
376    // Asked for before the loop rather than after it, because a `ChunksExact` settles what it
377    // cannot divide when it is built and hands back the same tail whether it has been walked or not.
378    let mut blocks = bytes.chunks_exact(32);
379    let rest = blocks.remainder();
380    if bytes.len() < 32 {
381        return checksum_tail(seed.wrapping_add(XXH_P5).wrapping_add(bytes.len() as u64), rest);
382    }
383    let mut lanes = [
384        seed.wrapping_add(XXH_P1).wrapping_add(XXH_P2),
385        seed.wrapping_add(XXH_P2),
386        seed,
387        seed.wrapping_sub(XXH_P1),
388    ];
389    for block in blocks.by_ref() {
390        checksum_block(&mut lanes, block);
391    }
392    finish_checksum(lanes, rest, bytes.len() as u64)
393}
394
395const XXH_P1: u64 = 11_400_714_785_074_694_791;
396const XXH_P2: u64 = 14_029_467_366_897_019_727;
397const XXH_P3: u64 = 1_609_587_929_392_839_161;
398const XXH_P4: u64 = 9_650_029_242_287_828_579;
399const XXH_P5: u64 = 2_870_177_450_012_600_261;
400
401fn checksum_round(state: u64, word: u64) -> u64 {
402    state.wrapping_add(word.wrapping_mul(XXH_P2)).rotate_left(31).wrapping_mul(XXH_P1)
403}
404
405fn checksum_word(chunk: &[u8]) -> u64 {
406    u64::from_le_bytes(chunk.try_into().expect("eight checksum bytes"))
407}
408
409/// One thirty two byte block into the four lanes.
410fn checksum_block(lanes: &mut [u64; 4], block: &[u8]) {
411    for (lane, chunk) in lanes.iter_mut().zip(block.chunks_exact(8)) {
412        *lane = checksum_round(*lane, checksum_word(chunk));
413    }
414}
415
416/// The lanes after every whole block, folded together with what was left over and the length.
417fn finish_checksum(lanes: [u64; 4], rest: &[u8], length: u64) -> u64 {
418    let merge = |state: u64, lane: u64| {
419        (state ^ checksum_round(0, lane)).wrapping_mul(XXH_P1).wrapping_add(XXH_P4)
420    };
421    let [one, two, three, four] = lanes;
422    let combined = one
423        .rotate_left(1)
424        .wrapping_add(two.rotate_left(7))
425        .wrapping_add(three.rotate_left(12))
426        .wrapping_add(four.rotate_left(18));
427    let hash = merge(merge(merge(merge(combined, one), two), three), four);
428    checksum_tail(hash.wrapping_add(length), rest)
429}
430
431/// The fewer than thirty two bytes after the last whole block, and the final mix.
432fn checksum_tail(mut hash: u64, mut rest: &[u8]) -> u64 {
433    let mut words = rest.chunks_exact(8);
434    for chunk in words.by_ref() {
435        hash ^= checksum_round(0, checksum_word(chunk));
436        hash = hash.rotate_left(27).wrapping_mul(XXH_P1).wrapping_add(XXH_P4);
437    }
438    rest = words.remainder();
439    if rest.len() >= 4 {
440        let (head, tail) = rest.split_at(4);
441        let quarter = u32::from_le_bytes(head.try_into().expect("four checksum bytes"));
442        hash ^= u64::from(quarter).wrapping_mul(XXH_P1);
443        hash = hash.rotate_left(23).wrapping_mul(XXH_P2).wrapping_add(XXH_P3);
444        rest = tail;
445    }
446    for &byte in rest {
447        hash ^= u64::from(byte).wrapping_mul(XXH_P5);
448        hash = hash.rotate_left(11).wrapping_mul(XXH_P1);
449    }
450    hash ^= hash >> 33;
451    hash = hash.wrapping_mul(XXH_P2);
452    hash ^= hash >> 29;
453    hash = hash.wrapping_mul(XXH_P3);
454    hash ^ (hash >> 32)
455}
456
457/// The checksum of `length` bytes of `file` from `offset`, read [`DIRECTORY_WINDOW`] at a time.
458///
459/// The same xxHash64 as [`checksum`], carried across reads rather than over one buffer, so that a
460/// directory can be checked without all of it being in memory at once. The four lanes take whole
461/// thirty two byte blocks, and a read that ends partway through one keeps the tail for the next.
462fn file_checksum(file: &File, offset: u64, length: usize) -> Result<u64> {
463    walk_checksummed(file, offset, length, DIRECTORY_WINDOW, |_| Ok(()))
464}
465
466/// Reads `length` bytes at `offset` a window at a time, hands each window to `each`, and answers
467/// the checksum of all of them.
468///
469/// `window` is a multiple of thirty two, so every window but the last is whole blocks of the hash
470/// and nothing has to be carried from one read to the next.
471fn walk_checksummed(
472    file: &File,
473    offset: u64,
474    length: usize,
475    window: usize,
476    mut each: impl FnMut(&[u8]) -> Result<()>,
477) -> Result<u64> {
478    debug_assert!(window % 32 == 0 && window > 0, "a window is whole blocks of the hash");
479    if length < 32 {
480        let mut bytes = vec![0; length];
481        read_at(file, offset, &mut bytes)?;
482        each(&bytes)?;
483        return Ok(checksum(&bytes));
484    }
485    let mut lanes = [XXH_P1.wrapping_add(XXH_P2), XXH_P2, 0, 0_u64.wrapping_sub(XXH_P1)];
486    let mut buffer = vec![0; window.min(length)];
487    let mut read = 0;
488    let (mut whole, mut filled) = (0, 0);
489    while read < length {
490        filled = buffer.len().min(length - read);
491        read_at(file, offset + read as u64, &mut buffer[..filled])?;
492        read += filled;
493        each(&buffer[..filled])?;
494        whole = filled / 32 * 32;
495        for block in buffer[..whole].chunks_exact(32) {
496            checksum_block(&mut lanes, block);
497        }
498    }
499    Ok(finish_checksum(lanes, &buffer[whole..filled], length as u64))
500}
501
502#[derive(Debug, Clone, Copy)]
503struct Slot {
504    offset: u64,
505    length: u32,
506    generation: u64,
507    hash: u64,
508}
509
510impl Slot {
511    fn bytes(self) -> [u8; SLOT_BYTES] {
512        let mut result = [0; SLOT_BYTES];
513        result[..8].copy_from_slice(&self.offset.to_le_bytes());
514        result[8..12].copy_from_slice(&self.length.to_le_bytes());
515        result[12..20].copy_from_slice(&self.generation.to_le_bytes());
516        result[20..28].copy_from_slice(&self.hash.to_le_bytes());
517        result
518    }
519
520    fn read(bytes: &[u8]) -> Self {
521        Self {
522            offset: u64::from_le_bytes(bytes[..8].try_into().expect("eight bytes")),
523            length: u32::from_le_bytes(bytes[8..12].try_into().expect("four bytes")),
524            generation: u64::from_le_bytes(bytes[12..20].try_into().expect("eight bytes")),
525            hash: u64::from_le_bytes(bytes[20..28].try_into().expect("eight bytes")),
526        }
527    }
528}
529
530#[derive(Debug, Clone, Copy)]
531struct Page {
532    offset: u64,
533    length: u32,
534    hash: u64,
535}
536
537impl Page {
538    /// How much of the file this page takes, for [`Reader::layout`].
539    fn bytes(&self) -> u64 {
540        u64::from(self.length)
541    }
542}
543
544#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
545enum FrequencyValue {
546    Null,
547    Integer(i128),
548    Code(u32),
549}
550
551/// A table keyed by the sixty four bits of the values the numeric frequency pass counts.
552///
553/// Every integer of every numeric column goes through one of these at least once when a table
554/// closes, and with the standard hasher that was a fifth of the close on its own, all of it SipHash
555/// guarding against an attacker who would have to choose the rows of the file being written.
556type FrequencyMap<V> = HashMap<u64, V, Spread>;
557
558/// The first pass of [`Writer::numeric_frequency`]: a Misra-Gries candidate table keyed by a value's
559/// sixty four bits, with the null counted beside it.
560#[derive(Debug, Default)]
561struct Candidates {
562    counts: FrequencyMap<u32>,
563    nulls: u32,
564    decrements: u64,
565}
566
567impl Candidates {
568    /// Counts `times` rows of `bits` and ends in the state `times` rows counted one at a time would.
569    ///
570    /// A value already held, or one there is room to hold, takes the whole run at once, because
571    /// every row after the first would find it held. A value the full table turns away goes a row
572    /// at a time, because each of its rows decrements every candidate and one of those decrements
573    /// can free the place the next row takes.
574    fn add(&mut self, bits: Option<u64>, mut times: u32) {
575        while times > 0 {
576            let held = match bits {
577                Some(bits) => self.counts.get_mut(&bits),
578                None if self.nulls != 0 => Some(&mut self.nulls),
579                None => None,
580            };
581            if let Some(count) = held {
582                *count = count.saturating_add(times);
583                return;
584            }
585            if self.counts.len() + usize::from(self.nulls != 0) < FREQUENCY_CANDIDATES {
586                match bits {
587                    Some(bits) => {
588                        self.counts.insert(bits, times);
589                    }
590                    None => self.nulls = times,
591                }
592                return;
593            }
594            self.counts.retain(|_, count| {
595                *count -= 1;
596                *count != 0
597            });
598            self.nulls = self.nulls.saturating_sub(1);
599            self.decrements = self.decrements.saturating_add(1);
600            times -= 1;
601        }
602    }
603}
604
605/// Equal rows in a row, gathered so they are counted once.
606#[derive(Debug, Default)]
607struct Run {
608    bits: Option<u64>,
609    times: u32,
610}
611
612impl Run {
613    /// Adds one row, and hands back the run it ended if it was not the same value.
614    fn push(&mut self, bits: Option<u64>) -> Option<(Option<u64>, u32)> {
615        if self.times != 0 && self.bits == bits && self.times < u32::MAX {
616            self.times += 1;
617            return None;
618        }
619        let ended = self.take();
620        self.bits = bits;
621        self.times = 1;
622        ended
623    }
624
625    /// The run being gathered, if there is one, leaving none.
626    fn take(&mut self) -> Option<(Option<u64>, u32)> {
627        let times = std::mem::take(&mut self.times);
628        (times != 0).then_some((self.bits, times))
629    }
630}
631
632/// Builds the hasher for [`FrequencyMap`].
633#[derive(Debug, Default, Clone, Copy)]
634struct Spread;
635
636impl std::hash::BuildHasher for Spread {
637    type Hasher = SpreadHasher;
638
639    fn build_hasher(&self) -> SpreadHasher {
640        SpreadHasher(0)
641    }
642}
643
644/// Folds each word in with a full width multiply whose two halves are xored together.
645///
646/// A plain multiply leaves the low bits of the hash as poor as the low bits of the key, and the
647/// table picks its bucket from the low bits, so a timestamp column, whose values are all multiples
648/// of a million microseconds, would pile into a sixty fourth of the buckets. Folding the high half
649/// of the product back in is what gives the low bits the whole word.
650#[derive(Debug)]
651struct SpreadHasher(u64);
652
653impl SpreadHasher {
654    fn mix(&mut self, word: u64) {
655        let product = u128::from(self.0 ^ word) * 0x9E37_79B9_7F4A_7C15_u128;
656        self.0 = (product as u64) ^ ((product >> 64) as u64);
657    }
658}
659
660impl std::hash::Hasher for SpreadHasher {
661    fn write(&mut self, bytes: &[u8]) {
662        for part in bytes.chunks(8) {
663            let mut word = [0; 8];
664            word[..part.len()].copy_from_slice(part);
665            self.mix(u64::from_le_bytes(word));
666        }
667    }
668
669    fn write_u32(&mut self, value: u32) {
670        self.mix(u64::from(value));
671    }
672
673    fn write_u64(&mut self, value: u64) {
674        self.mix(value);
675    }
676
677    fn write_i128(&mut self, value: i128) {
678        self.mix(value as u64);
679        self.mix((value >> 64) as u64);
680    }
681
682    fn write_isize(&mut self, value: isize) {
683        self.mix(value as u64);
684    }
685
686    fn finish(&self) -> u64 {
687        self.0
688    }
689}
690
691#[derive(Debug, Clone)]
692struct FrequencyEntry {
693    value: FrequencyValue,
694    count: u64,
695}
696
697/// Exact leading frequencies for one column.
698///
699/// Values outside `entries` occur at most `omitted_max` times. This lets a count-descending TopN
700/// use the synopsis only when its last winner is strictly above every omitted value.
701#[derive(Debug, Clone)]
702struct FrequencySummary {
703    entries: Vec<FrequencyEntry>,
704    omitted_max: u64,
705    ordinals: Vec<u64>,
706    ordinal_entries: Vec<u16>,
707}
708
709#[derive(Debug, Clone)]
710struct PairFrequencyEntry {
711    first_entry: u16,
712    second: Option<u32>,
713    count: u64,
714}
715
716/// Exact leading counts for one numeric frequency anchor and one stable string code space.
717///
718/// `omitted_max` covers both first-key values outside the numeric synopsis and pairs below the
719/// retained prefix. A TopN may therefore use the entries only when its boundary strictly exceeds
720/// this number.
721#[derive(Debug, Clone)]
722struct PairFrequencySummary {
723    first: u16,
724    second: u16,
725    entries: Vec<PairFrequencyEntry>,
726    omitted_max: u64,
727}
728
729/// One column's frequency synopsis, in memory or left where it is in the file.
730///
731/// A writer holds what it counted. A reader leaves every synopsis in the file and reads one back
732/// when a query asks about its column, because they are the largest thing in a directory once they
733/// are decoded, forty eight bytes an entry and nearly twenty thousand entries over `hits`, and
734/// most queries ask about none of them. Where one sits is found at open, by reading it through and
735/// checking it, so a torn synopsis is still refused when the table is opened.
736#[derive(Debug, Clone)]
737enum Frequencies {
738    Held(FrequencySummary),
739    /// Where the synopsis sits, and whether it was written with the value of each ordinal, which
740    /// is what the directory's frequency magic says and the synopsis itself does not.
741    Stored {
742        span: Span,
743        values: bool,
744    },
745}
746
747/// The values one column's frequency synopsis lists, with a bound on everything it left out.
748///
749/// What [`Reader::frequency_prefix`] answers. The counts are exact, and `omitted_max` is how many
750/// rows any value not in the list can hold, which is zero when nothing was left out at all.
751#[derive(Debug, Clone)]
752pub struct FrequencyPrefix {
753    /// Every value the synopsis lists, with the number of rows holding it, count descending.
754    pub entries: Vec<(Value, u64)>,
755    /// How many rows the most common value outside the list holds, and zero for a complete list.
756    pub omitted_max: u64,
757}
758
759/// Sparse row ordinals covered by a numeric frequency candidate set.
760#[derive(Debug, Clone, PartialEq)]
761pub struct FrequencyOccurrences {
762    /// Upper bound for the frequency of every value absent from the fetched rows.
763    pub omitted_max: u64,
764    /// Table-wide row ordinals in ascending order.
765    pub ordinals: Vec<u64>,
766    /// The retained heavy-hitter values named by `anchor_indices`.
767    pub anchors: Vec<Value>,
768    /// The index in `anchors` at each ordinal, or empty for a legacy FQ2 directory.
769    pub anchor_indices: Vec<u16>,
770}
771
772/// Exact grouped counts for a pair of values, in descending count order.
773pub type PairFrequencyCounts = Vec<(Vec<Value>, u64)>;
774
775/// Where one column's page for one stripe sits in the file.
776///
777/// A column page has no checksum of its own because every part inside it carries one, and the
778/// stripe's index page holds those. Checking a part on the way out of the page covers exactly the
779/// bytes a reader is about to decode, and covers them once whether the reader took the whole page
780/// or pulled one part out of the middle of it.
781#[derive(Debug, Clone, Copy, Default)]
782struct Span {
783    offset: u64,
784    length: u32,
785}
786
787/// One optional page for each column of a stripe, holding only the pages that are there.
788///
789/// A stripe has three of these, the membership, sieve and part range pages. As a
790/// `Vec<Option<Page>>` each was thirty two bytes a column whether the page was there or not, and
791/// over the ten million rows of `hits` that is half a megabyte at open for 7171 pages out of 16380
792/// slots. Kept sparse and packed, a page that is there is twenty four bytes and one that is not is
793/// nothing.
794#[derive(Debug, Clone, Default)]
795struct Pages {
796    columns: usize,
797    held: Box<[StripePage]>,
798}
799
800/// A page and the column it is for, packed so that the column sits where the padding was.
801#[derive(Debug, Clone, Copy)]
802struct StripePage {
803    offset: u64,
804    hash: u64,
805    length: u32,
806    column: u32,
807}
808
809impl Pages {
810    /// The pages of `columns` columns, one slot each in column order.
811    fn from_slots(slots: Vec<Option<Page>>) -> Result<Self> {
812        let mut held = Vec::with_capacity(slots.iter().flatten().count());
813        for (column, page) in slots.iter().enumerate() {
814            if let Some(page) = page {
815                let column =
816                    u32::try_from(column).map_err(|_| invalid("too many columns for a page"))?;
817                held.push(StripePage {
818                    offset: page.offset,
819                    hash: page.hash,
820                    length: page.length,
821                    column,
822                });
823            }
824        }
825        Ok(Self { columns: slots.len(), held: held.into_boxed_slice() })
826    }
827
828    /// The page of one column, if it has one.
829    fn get(&self, column: usize) -> Option<Page> {
830        let at = self.held.binary_search_by_key(&column, |placed| placed.column as usize).ok()?;
831        let placed = self.held[at];
832        Some(Page { offset: placed.offset, length: placed.length, hash: placed.hash })
833    }
834
835    /// One slot per column, in column order, the way the directory writes them.
836    fn slots(&self) -> impl Iterator<Item = Option<Page>> + '_ {
837        (0..self.columns).map(|column| self.get(column))
838    }
839
840    /// How much of the file one column's page takes, or zero when it has none.
841    fn bytes(&self, column: usize) -> u64 {
842        self.get(column).map_or(0, |page| page.bytes())
843    }
844}
845
846/// One independently readable stripe of a table.
847#[derive(Debug, Clone)]
848pub struct Stripe {
849    rows: usize,
850    /// Rows in each part, in source order. Kept in the directory so that mapping a row ordinal to a
851    /// part, which every sparse fetch does, never reads the file.
852    parts: Vec<u32>,
853    /// The index page: one section per column, holding a length and a checksum for every part and
854    /// then a checksum of the section itself, so that a reader can pread one column's section and
855    /// still know it is intact.
856    index: Span,
857    pages: Vec<Span>,
858    memberships: Pages,
859    /// One page per column holding the membership sieve of every part of the stripe, for the
860    /// columns that have one. A column whose parts all declined a sieve has no page at all.
861    sieves: Pages,
862    /// One page per column holding the two ends and the null count of every part of the stripe.
863    ///
864    /// The stripe's own `zone` below covers sixty four times as many rows, and on a column that is
865    /// not the one the rows are ordered by that is the difference between skipping half the file and
866    /// skipping all but three percent of it. On ClickBench 24 the cutoff the answer settles at
867    /// leaves eight stripes of sixteen alive and thirty parts of nine hundred and seventy four.
868    ///
869    /// A page per column rather than one page for the stripe, so that a query that compares one
870    /// column reads the ends of that column and not of the hundred and four beside it. Read lazily
871    /// for the same reason, like the sieves.
872    part_ranges: Pages,
873    zone: Zone,
874}
875
876impl Stripe {
877    /// Number of rows in this stripe.
878    #[must_use]
879    pub fn rows(&self) -> usize {
880        self.rows
881    }
882
883    /// Number of parts in this stripe.
884    #[must_use]
885    pub fn parts(&self) -> usize {
886        self.parts.len()
887    }
888
889    /// The two ends and the null count of every column over the whole stripe.
890    ///
891    /// In the directory and so in memory, which is what makes it the one a planner can ask. The
892    /// finer ones are a page per column per stripe in the file, read by [`Reader::skips`] when a
893    /// scan wants to know which parts to open.
894    #[must_use]
895    pub fn zone(&self) -> &Zone {
896        &self.zone
897    }
898}
899
900/// The committed table directory.
901#[derive(Debug, Clone)]
902pub struct Table {
903    name: String,
904    fields: Vec<Field>,
905    stripes: Vec<Stripe>,
906    rows: usize,
907    dictionaries: Vec<Option<Page>>,
908    /// Bytes of each column's dictionary payload that are outside its page, which is all of them
909    /// from format 27 and none of them before. See [`DICTIONARY_PAYLOADS`].
910    ///
911    /// Empty rather than a row of zeros on a table that has none, and read with `get` for that
912    /// reason, so that a table built by hand in a test does not have to know about it.
913    dictionary_payloads: Vec<u64>,
914    frequencies: Vec<Option<Frequencies>>,
915    pair_frequencies: Vec<PairFrequencySummary>,
916    /// String spellings aligned with each column's frequency entries.
917    ///
918    /// Empty for files written before `RUDBFT1`. A `None` entry is the null frequency entry; every
919    /// code entry in a column named by the block has its exact bytes here.
920    frequency_texts: Vec<Vec<Option<Vec<u8>>>>,
921    /// Exact candidate host aggregates and an upper bound for every omitted host.
922    host_groups: Option<host::HostSummary>,
923    /// How many distinct values each column holds, for the columns that know.
924    ///
925    /// A dictionary entry is made the first time a value is seen and nothing ever removes one, so
926    /// the size of the dictionary is the number of distinct values in the column. That is the whole
927    /// story for a column with no null in it, and the wrong number by one for a column with a null
928    /// in it, because a null row is written as the code for the empty string and makes an entry the
929    /// dictionary would not otherwise have. The writer knows which case it is, since it counts the
930    /// non-null rows that use each code while it builds the frequency summary, and the reader cannot
931    /// work it out from the dictionary alone. So the writer settles it here.
932    distincts: Vec<Option<u64>>,
933    /// The order the rows of this table are meant to be stored in, if anybody declared one.
934    ///
935    /// A declaration and not a measurement. Nothing here checks that the stripes actually arrived
936    /// in this order, and the reason it is worth storing anyway is that the order is the only thing
937    /// about a table that a rewrite destroys without anybody noticing. The fragment ranges prune on
938    /// whatever order the rows came in, so a table loaded sorted prunes and the same table after a
939    /// checkpoint that did not know to keep the order quietly stops pruning and nothing says why.
940    clustering: Option<Clustering>,
941    /// The file generation of the commit that last wrote this table's column pages.
942    ///
943    /// This is what spec/graph/03-the-file-format.md section 3.2 calls the table generation, and
944    /// the definition is deliberately about the pages rather than about the directory. A graph
945    /// section is a restatement of a column in terms of row ids, so what invalidates one is the
946    /// rows being renumbered, and nothing else. Adding a second table to the file, or attaching a
947    /// section to this one, commits a new file generation without touching a single row of this
948    /// table, and a definition that moved with those would declare every section in the file stale
949    /// for no reason.
950    ///
951    /// Zero on a table written before format 23, where nothing recorded it. Real generations start
952    /// at one, so zero can never match a section's stamp, and a table from format 22 has no
953    /// sections for it to match anyway.
954    generation: u64,
955    /// The graph sections this table carries, per spec/graph/03-the-file-format.md section 3.2.
956    ///
957    /// Empty for every table written before the section table existed, and empty is not a
958    /// degraded state: section 3.1 says deleting every graph section from a file changes no answer,
959    /// only the time, so a table with none here answers every query the same way and slower. That
960    /// is what lets this field arrive without a migration.
961    sections: Vec<Section>,
962}
963
964impl Table {
965    /// The SQL table name held by this snapshot.
966    #[must_use]
967    pub fn name(&self) -> &str {
968        &self.name
969    }
970
971    /// Columns in their SQL order.
972    #[must_use]
973    pub fn fields(&self) -> &[Field] {
974        &self.fields
975    }
976
977    /// Committed row count.
978    #[must_use]
979    pub fn rows(&self) -> usize {
980        self.rows
981    }
982
983    /// Independently readable stripes.
984    #[must_use]
985    pub fn stripes(&self) -> &[Stripe] {
986        &self.stripes
987    }
988
989    /// The order the rows are meant to be stored in, if this table was declared with one.
990    #[must_use]
991    pub fn clustering(&self) -> Option<&Clustering> {
992        self.clustering.as_ref()
993    }
994
995    /// The generation every section of this table is judged against.
996    ///
997    /// See the field. A caller deciding whether to read a section asks [`Section::usable`] with
998    /// this.
999    #[must_use]
1000    pub fn generation(&self) -> u64 {
1001        self.generation
1002    }
1003
1004    /// Every graph section this table names, including the kinds this build does not know.
1005    ///
1006    /// Including them is the point. A caller that wants only the ones it can use asks
1007    /// [`Section::usable`], and a caller rewriting the directory carries the rest through, so a
1008    /// file opened by an older build and written again does not silently lose a section that build
1009    /// had no name for.
1010    #[must_use]
1011    pub fn sections(&self) -> &[Section] {
1012        &self.sections
1013    }
1014}
1015
1016/// One table's line in the catalog directory.
1017///
1018/// The small level of the two. It holds what opening a database needs and nothing else: the name to
1019/// bind, the shape to plan against, the row count, and where the table's own directory sits. A file
1020/// of eight tables is eight of these, and reading them costs the same whether the tables hold a
1021/// thousand rows or a billion.
1022///
1023/// The name, the fields and the row count are repeated here rather than pointed at inside the table
1024/// directory, which is the entire point of having two levels. A catalog that pointed at them would
1025/// have to read every table directory at open to answer what tables there are, which is the cost
1026/// this level exists to avoid.
1027#[derive(Debug, Clone)]
1028struct Entry {
1029    name: String,
1030    fields: Vec<Field>,
1031    rows: usize,
1032    /// Where this table's own directory sits, with the checksum it was committed under.
1033    directory: Page,
1034    /// Exact non-null, nonzero integer counts certified by the catalog checksum.
1035    nonzero: Vec<Option<u64>>,
1036    /// Exact sum and non-null count for signed integer columns.
1037    aggregates: Vec<Option<(i128, u64)>>,
1038    /// Exact non-null distinct values when the writer finished counting the column.
1039    distincts: Vec<Option<u64>>,
1040    /// Exact integer or date bounds; the inner `None` means every row is null.
1041    extremes: Vec<StoredIntegerExtremes>,
1042    /// Complete bounded numeric frequencies, including NULL when present.
1043    frequencies: Vec<StoredNumericFrequencies>,
1044}
1045
1046type StoredIntegerExtremes = Option<Option<(i128, i128)>>;
1047type StoredNumericFrequencies = Option<NumericFrequencies>;
1048
1049/// One view's line in the catalog directory.
1050///
1051/// A view has no pages, so unlike a table it is entirely here and there is no second level under it.
1052/// What it is made of is text: the body the binder binds again at every reference, and the whole
1053/// statement written back out, which is what `duckdb_views()` reports and nothing else reads.
1054///
1055/// The columns are a cache and they are written down anyway, which is worth saying out loud because
1056/// a cache in a file looks like a mistake. It is what the pin does. Create a view on a file, open
1057/// the file again in another process, and `duckdb_views()` answers `column_count` and `is_bound`
1058/// true without anything having bound the body, so the list survived the write. Not writing it
1059/// would answer null and false there, and the only way back would be to bind every view at open,
1060/// which is the thing the cache exists to avoid.
1061#[derive(Debug, Clone, PartialEq, Eq)]
1062pub struct ViewEntry {
1063    /// The view's own name, without the schema, the way a table entry holds its name.
1064    pub name: String,
1065    /// The query the view stands for, as the text that was written.
1066    pub sql: String,
1067    /// The whole `CREATE VIEW` written back out.
1068    pub statement: String,
1069    /// The column names the statement gave, which rename a prefix of what the body produces.
1070    pub aliases: Vec<String>,
1071    /// The columns the last bind of the body produced.
1072    pub columns: Vec<Field>,
1073}
1074
1075/// Where one column's bytes went, taken from the directory rather than by reading pages.
1076#[derive(Debug, Clone)]
1077pub struct ColumnLayout {
1078    /// The column's name, so a report does not have to carry the field list beside this.
1079    pub name: String,
1080    /// The type, spelled the way the catalog spells it.
1081    pub kind: String,
1082    /// Every stripe's page of this column added up, which is the encoded data itself.
1083    pub pages: u64,
1084    /// Every stripe's exact code membership page for this column.
1085    pub memberships: u64,
1086    /// Every stripe's membership sieve page for this column.
1087    pub sieves: u64,
1088    /// Every stripe's per part range page for this column.
1089    pub part_ranges: u64,
1090    /// The table wide dictionary of this column, if it has one.
1091    pub dictionary: u64,
1092}
1093
1094impl ColumnLayout {
1095    /// Everything this column costs, which is what the file would lose if the column went.
1096    #[must_use]
1097    pub fn total(&self) -> u64 {
1098        self.pages
1099            .saturating_add(self.memberships)
1100            .saturating_add(self.sieves)
1101            .saturating_add(self.part_ranges)
1102            .saturating_add(self.dictionary)
1103    }
1104}
1105
1106/// Where a whole file's bytes went.
1107///
1108/// Every number here comes out of the committed directory, so taking it costs one directory read
1109/// however large the file is. That is the point: a 45 GB table has to be able to say where it went
1110/// without being read, or nobody will ask.
1111///
1112/// The parts that are not a column are kept apart rather than shared out over the columns. The
1113/// stripe index page holds a section per column and could be split, and the directory and the
1114/// header cannot be, so splitting one of the three and not the others would read as if the columns
1115/// accounted for everything. They do not, and the gap is the thing worth looking at.
1116#[derive(Debug, Clone)]
1117pub struct Layout {
1118    /// The size of the file on disk.
1119    pub file: u64,
1120    /// Committed rows.
1121    pub rows: usize,
1122    /// Committed stripes.
1123    pub stripes: usize,
1124    /// Committed parts, which is how many chunks a scan reads.
1125    pub parts: usize,
1126    /// One entry per column, in the table's column order.
1127    pub columns: Vec<ColumnLayout>,
1128    /// Every stripe's index page, which carries a length and a checksum for every part of every
1129    /// column and is charged per stripe rather than per column.
1130    pub indexes: u64,
1131    /// The committed directory itself, the one that was read to build this.
1132    pub directory: u64,
1133    /// The fixed header, which holds the magic, the format and the two directory slots.
1134    pub header: u64,
1135}
1136
1137impl Layout {
1138    /// Everything the columns cost together.
1139    #[must_use]
1140    pub fn columns_total(&self) -> u64 {
1141        self.columns.iter().map(ColumnLayout::total).fold(0, u64::saturating_add)
1142    }
1143
1144    /// What the file holds that this does not account for.
1145    ///
1146    /// A committed file is written once and never rewritten in place, so an earlier directory and
1147    /// the pages of an earlier snapshot are still in it. That is the honest place for them: they
1148    /// are bytes on disk that no column owns.
1149    #[must_use]
1150    pub fn unaccounted(&self) -> u64 {
1151        self.file
1152            .saturating_sub(self.columns_total())
1153            .saturating_sub(self.indexes)
1154            .saturating_sub(self.directory)
1155            .saturating_sub(self.header)
1156    }
1157}
1158
1159/// How one part of one column is stored, which is one row of `pragma_storage_info`.
1160///
1161/// Everything here is read off the file rather than worked out from the schema, because the whole
1162/// question this answers is what the encoder chose, and the encoder chooses per part. Two files
1163/// holding the same rows in a different order give different answers and that difference is the
1164/// reason to ask.
1165///
1166/// The encoding costs a read of the column's page, so this is not free the way [`Layout`] is. It is
1167/// one read per column per stripe rather than one per part, because a part is a few kilobytes out
1168/// of a page that is a quarter of a megabyte.
1169#[derive(Debug, Clone)]
1170pub struct StoredPart {
1171    /// Which stripe the part belongs to.
1172    pub stripe: usize,
1173    /// Which part of that stripe it is, counting from zero inside the stripe.
1174    pub part: usize,
1175    /// The table wide row number the part starts at.
1176    pub row: usize,
1177    /// How many rows it holds.
1178    pub rows: usize,
1179    /// What the encoder made of it, as a line of text like `DICT(PACKED, PACKED)`.
1180    pub encoding: String,
1181    /// The stored bytes of the part, which is what it costs in the file.
1182    pub bytes: u64,
1183    /// Where in the file the column page holding this part starts.
1184    pub page: u64,
1185    /// Where in that page the part starts.
1186    pub offset: u64,
1187    /// The smallest value the part holds, when the stored ranges say.
1188    pub low: Option<Value>,
1189    /// The largest, same.
1190    pub high: Option<Value>,
1191    /// How many of its rows are null, when the stored ranges say.
1192    pub nulls: Option<usize>,
1193}
1194
1195/// Seeds the second hash a global dictionary tells its values apart by.
1196///
1197/// Any value that is not zero does, since zero is the seed [`checksum`] already uses and the point
1198/// is only that the two hashes of one value are not the same number. This one is the fractional part
1199/// of the golden ratio in sixty four bits, which is the constant everything else here is built out
1200/// of and is as good a nothing-up-my-sleeve number as any.
1201const DICTIONARY_CHECK_SEED: u64 = 11_400_714_819_323_198_485;
1202
1203/// One column's table wide dictionary while the load is running.
1204///
1205/// The thing to understand about this is what it does not hold. A dictionary of `URL` at a hundred
1206/// million ClickBench rows has about eighteen million distinct values and 1.3 GB of bytes in them,
1207/// and five columns like it are twelve of the seventeen gigabytes a load of `hits` peaks at. So the
1208/// bytes are not kept. A value's bytes go into [`GlobalDictionary::filling`], and when that reaches
1209/// [`TEXT_PAYLOAD_VALUES`] values the block is sealed, handed out at the end of the merge that
1210/// sealed it to be encoded with the stripe's pages, and never seen in that form again. What is left
1211/// is the encoded block, which is two to three times smaller, and that is the same bytes the file
1212/// is going to hold anyway.
1213///
1214/// Two things needed the raw bytes and neither needs them now. Deciding whether a value has been
1215/// seen before was a hash lookup and then a comparison of the bytes, and is now a hash lookup and a
1216/// comparison of a second hash under a different seed, which is [`DICTIONARY_CHECK_SEED`] and the
1217/// argument for why that is sound. Sorting the values at the end needed all of them at once, and
1218/// now reads the blocks back through [`GlobalDictionary::decoded`] one column at a time, which is
1219/// one column's bytes rather than every column's.
1220///
1221/// The offsets going block relative comes free with it, and takes the four gigabyte wall with it.
1222/// They were `u32` into a per column payload, so a column could not hold more than four gigabytes of
1223/// values however much memory the machine had, and `URL` and `Referer` are within a small factor of
1224/// that at a hundred million rows. A `u32` into a block of 1,024 values is not a bound anything real
1225/// reaches. The stored form is unchanged, because [`encode_offsets`] was already subtracting a per
1226/// block base before writing.
1227#[derive(Debug)]
1228struct GlobalDictionary {
1229    primary: HashMap<u64, u32>,
1230    collisions: HashMap<u64, Vec<u32>>,
1231    /// Every value's hash under [`DICTIONARY_CHECK_SEED`], in code order.
1232    checks: Vec<u64>,
1233    /// Where every value ends inside the payload block it is in, in code order.
1234    ends: Vec<u32>,
1235    counts: Vec<u64>,
1236    nulls: u64,
1237    /// The values of the block being filled, back to back.
1238    filling: Vec<u8>,
1239    /// One conservative four-byte substring signature per encoded payload block, in block order.
1240    ///
1241    /// Made where the block is encoded rather than where it is sealed, because sealing is under the
1242    /// writer's lock and every byte of every value going through [`gram_bits`] was 2.9 of the 14
1243    /// seconds the 10m ClickBench load spent on the 32 core box.
1244    grams: Vec<[u8; TEXT_GRAM_BYTES]>,
1245    /// Blocks that have filled and not been handed out to be encoded yet, each with its block number.
1246    ///
1247    /// Empty except inside the merge that filled them, and while the column is still too small to
1248    /// settle a shape on.
1249    waiting: Vec<(usize, Vec<u8>)>,
1250    /// Blocks kept raw to settle a shape on, spread across the column, each with its number.
1251    ///
1252    /// At most [`PAYLOAD_SAMPLE_BLOCKS`] of them and so at most a few megabytes. Spread rather than
1253    /// taken off the front for the reason [`settle_shape`] gives, and kept rather than read back
1254    /// because reading back is a decode and this is a sample of a column that is still growing.
1255    sample: Vec<(usize, Vec<u8>)>,
1256    /// How far apart the blocks in `sample` are, which doubles every time there are too many.
1257    stride: usize,
1258    /// What the blocks encoded so far were encoded with, once the column is big enough to settle it.
1259    shape: Option<chooser::Settled>,
1260    /// How many blocks had filled when that shape was settled.
1261    settled: usize,
1262    /// The blocks that are encoded and not yet in the file, in block order, following `placed`.
1263    ///
1264    /// Empty between stripes, because [`Writer::place_blocks`] writes them the moment they come
1265    /// back. Only a dictionary that never meets a writer, which is a test's, keeps them here.
1266    blocks: Vec<Vec<u8>>,
1267    /// Blocks that came back encoded ahead of a block before them, by block number.
1268    ///
1269    /// Two stripes merged one after the other can have their pages built in the other order, and a
1270    /// block cannot go into `blocks` until every block before it is there. They wait here until the
1271    /// gap closes, which is at most until the stripe merged just before this one is written.
1272    early: BTreeMap<usize, EncodedBlock>,
1273    /// Where every block already written to the file is, in block order.
1274    placed: Vec<Placed>,
1275}
1276
1277/// Where one payload block of a global dictionary is in the file, and its checksum.
1278#[derive(Debug, Clone, Copy)]
1279struct Placed {
1280    start: u64,
1281    length: u64,
1282    hash: u64,
1283}
1284
1285/// Sorted `(head, code)` entries and the decoded bytes and block bases they were sorted over.
1286type RankedDictionary = (Vec<(u64, u32)>, Vec<u8>, Vec<u64>);
1287
1288impl GlobalDictionary {
1289    fn new() -> Self {
1290        Self {
1291            primary: HashMap::new(),
1292            collisions: HashMap::new(),
1293            checks: Vec::new(),
1294            ends: Vec::new(),
1295            counts: Vec::new(),
1296            nulls: 0,
1297            filling: Vec::new(),
1298            grams: Vec::new(),
1299            waiting: Vec::new(),
1300            sample: Vec::new(),
1301            stride: 1,
1302            shape: None,
1303            settled: 0,
1304            blocks: Vec::new(),
1305            early: BTreeMap::new(),
1306            placed: Vec::new(),
1307        }
1308    }
1309
1310    /// How many distinct values this dictionary holds, which is one past its largest code.
1311    fn values(&self) -> usize {
1312        self.ends.len()
1313    }
1314
1315    /// About how many bytes closing this dictionary holds at once: every value decoded, and a
1316    /// sort entry and a code for each.
1317    fn closing_bytes(&self) -> usize {
1318        let values = self.values();
1319        let decoded = (0..values.div_ceil(TEXT_PAYLOAD_VALUES))
1320            .map(|block| self.ends[((block + 1) * TEXT_PAYLOAD_VALUES).min(values) - 1] as usize)
1321            .sum::<usize>();
1322        decoded.saturating_add(values.saturating_mul(size_of::<(u64, u32)>() + size_of::<u32>()))
1323    }
1324
1325    /// How many blocks are encoded, written or not, which is the number the next one has to have.
1326    fn encoded(&self) -> usize {
1327        self.placed.len() + self.blocks.len()
1328    }
1329
1330    #[cfg(test)]
1331    fn code(&mut self, text: &str) -> Result<u32> {
1332        let bytes = text.as_bytes();
1333        self.code_hashed(bytes, checksum(bytes), seeded_checksum(bytes, DICTIONARY_CHECK_SEED))
1334    }
1335
1336    /// The code for a value whose two hashes the caller already has.
1337    ///
1338    /// A stripe prepared outside the writer's lock hashed every value it holds while it was coding
1339    /// them, and merging it into this dictionary is one of these a distinct value rather than two
1340    /// hashes of every row. See [`prepare`].
1341    fn code_hashed(&mut self, text: &[u8], hash: u64, check: u64) -> Result<u32> {
1342        if let Some(&code) = self.primary.get(&hash) {
1343            if self.checks.get(code as usize) == Some(&check) {
1344                return Ok(code);
1345            }
1346            if let Some(codes) = self.collisions.get(&hash) {
1347                if let Some(code) =
1348                    codes.iter().copied().find(|&code| self.checks[code as usize] == check)
1349                {
1350                    return Ok(code);
1351                }
1352            }
1353            let code = self.insert(text, check)?;
1354            self.collisions.entry(hash).or_default().push(code);
1355            return Ok(code);
1356        }
1357        let code = self.insert(text, check)?;
1358        self.primary.insert(hash, code);
1359        Ok(code)
1360    }
1361
1362    fn insert(&mut self, text: &[u8], check: u64) -> Result<u32> {
1363        let code = u32::try_from(self.ends.len())
1364            .map_err(|_| invalid("global dictionary has too many values"))?;
1365        self.filling.extend_from_slice(text);
1366        self.ends.push(
1367            u32::try_from(self.filling.len())
1368                .map_err(|_| invalid("a global dictionary value exceeds 4 GiB"))?,
1369        );
1370        self.checks.push(check);
1371        self.counts.push(0);
1372        if self.ends.len() % TEXT_PAYLOAD_VALUES == 0 {
1373            self.seal();
1374        }
1375        Ok(code)
1376    }
1377
1378    /// Closes the block being filled and puts it in the queue to be encoded.
1379    ///
1380    /// Also keeps a copy of it if it lands on the sample's stride, and halves the sample when that
1381    /// has left too many, which is what keeps the kept blocks spread evenly over however much of the
1382    /// column exists rather than bunched at whichever end was cheap to remember.
1383    fn seal(&mut self) {
1384        let at = self.ends.len().div_ceil(TEXT_PAYLOAD_VALUES) - 1;
1385        let bytes = std::mem::take(&mut self.filling);
1386        if at % self.stride == 0 {
1387            self.sample.push((at, bytes.clone()));
1388            if self.sample.len() > PAYLOAD_SAMPLE_BLOCKS {
1389                self.stride *= 2;
1390                let stride = self.stride;
1391                self.sample.retain(|(at, _)| at % stride == 0);
1392            }
1393        }
1394        self.waiting.push((at, bytes));
1395    }
1396
1397    /// The values of one block, as slices into the bytes the block was filled with.
1398    fn slices<'a>(&self, at: usize, bytes: &'a [u8]) -> Vec<&'a [u8]> {
1399        block_values(self.block_ends(at), bytes)
1400    }
1401
1402    /// Where every value of one block ends, relative to the block.
1403    fn block_ends(&self, at: usize) -> &[u32] {
1404        let first = (at * TEXT_PAYLOAD_VALUES).min(self.ends.len());
1405        let last = (first + TEXT_PAYLOAD_VALUES).min(self.ends.len());
1406        &self.ends[first..last]
1407    }
1408
1409    /// Takes every waiting block out to be encoded somewhere else, if the column has a shape to
1410    /// encode them with.
1411    ///
1412    /// This is what keeps the encoding out of the writer's lock. A block needs its bytes, where its
1413    /// values end and the shape, and nothing else of the dictionary, so it goes out with a copy of
1414    /// the four kilobytes of ends it has and comes back through [`GlobalDictionary::take_back`].
1415    fn hand_out(&mut self, column: usize) -> Vec<Unencoded> {
1416        let Some(shape) = &self.shape else { return Vec::new() };
1417        let waiting = std::mem::take(&mut self.waiting);
1418        waiting
1419            .into_iter()
1420            .map(|(at, bytes)| Unencoded {
1421                column,
1422                at,
1423                ends: self.block_ends(at).to_vec(),
1424                bytes,
1425                shape: shape.clone(),
1426            })
1427            .collect()
1428    }
1429
1430    /// Takes back one block that was handed out, and moves every block that is now next in line
1431    /// into `blocks`.
1432    fn take_back(&mut self, at: usize, block: EncodedBlock) -> Result<()> {
1433        if at < self.encoded() || self.early.insert(at, block).is_some() {
1434            return Err(Error::internal("a dictionary block came back twice"));
1435        }
1436        while let Some(block) = self.early.remove(&self.encoded()) {
1437            self.push_block(block);
1438        }
1439        Ok(())
1440    }
1441
1442    /// Appends the next encoded block and its signature.
1443    fn push_block(&mut self, (bytes, grams): EncodedBlock) {
1444        self.blocks.push(bytes);
1445        self.grams.push(*grams);
1446    }
1447
1448    /// Settles the shape the waiting blocks are about to be encoded with, if there is enough column
1449    /// to settle one on.
1450    ///
1451    /// Settled again once the column has grown fourfold, because the sample it was settled on then
1452    /// covered a quarter of what exists now and a dictionary in first seen order does not look the
1453    /// same at both ends. Blocks already encoded keep the shape they were encoded with. They can,
1454    /// because a block says what it is: nothing reading one asks the column what shape to expect.
1455    fn settle(&mut self) -> Result<()> {
1456        if self.sample.len() < PAYLOAD_SAMPLE_BLOCKS {
1457            return Ok(());
1458        }
1459        let complete = self.ends.len() / TEXT_PAYLOAD_VALUES;
1460        if self.shape.is_some() && complete < self.settled.saturating_mul(4) {
1461            return Ok(());
1462        }
1463        let sample =
1464            self.sample.iter().map(|(at, bytes)| self.slices(*at, bytes)).collect::<Vec<_>>();
1465        self.shape = Some(string::with_symbols(settle_shape(&sample)?, &sample));
1466        self.settled = complete;
1467        Ok(())
1468    }
1469
1470    /// Seals the part block at the end of the load, if there is one.
1471    fn seal_rest(&mut self) {
1472        // Asked of the values rather than of the bytes, because a block of empty strings has values
1473        // in it and no bytes, and a column of nulls is exactly that.
1474        if self.ends.len() % TEXT_PAYLOAD_VALUES != 0 {
1475            self.seal();
1476        }
1477    }
1478
1479    /// Encodes the waiting block at `at`, with the settled shape when there is one and by trying
1480    /// everything when the column was too small to settle one.
1481    fn encode_waiting(&self, at: usize) -> Result<EncodedBlock> {
1482        let (block, bytes) = &self.waiting[at];
1483        let values = self.slices(*block, bytes);
1484        let encoded = match &self.shape {
1485            Some(shape) => string::encode_with(&values, shape)?,
1486            None => string::encode(&values)?,
1487        };
1488        Ok((encoded, block_grams(&values)))
1489    }
1490
1491    /// [`finish_dictionaries`] for one dictionary on this thread, for the tests that hold one.
1492    #[cfg(test)]
1493    fn finish_blocks(&mut self) -> Result<()> {
1494        self.seal_rest();
1495        let made = (0..self.waiting.len())
1496            .map(|at| self.encode_waiting(at))
1497            .collect::<Result<Vec<_>>>()?;
1498        for ((at, _), block) in std::mem::take(&mut self.waiting).into_iter().zip(made) {
1499            if self.encoded() != at {
1500                return Err(Error::internal("a dictionary block was encoded out of order"));
1501            }
1502            self.push_block(block);
1503        }
1504        Ok(())
1505    }
1506
1507    /// Every value of this dictionary read back out of its encoded blocks, as the bytes back to back
1508    /// and where each block starts in them.
1509    ///
1510    /// This is the one place the whole column is in memory at once and the reason [`Writer::close`]
1511    /// takes the columns one at a time rather than across threads. One column's values is 1.3 GB on
1512    /// the worst ClickBench column, and five columns of that at once is the peak this was all meant
1513    /// to remove.
1514    ///
1515    /// The blocks are spread over threads instead. Each block's decoded length is already known from
1516    /// the ends of its values, so the answer is laid out before anything is decoded and every thread
1517    /// decodes its own run of blocks straight into its own part of it. On the 10m ClickBench sample
1518    /// this was a second of the close for `URL` alone, on one core of thirty two, and the close is
1519    /// what a load waits on once its stripes are written.
1520    ///
1521    /// The blocks already written are read back out of `file`, so what a thread holds beyond the
1522    /// answer is one encoded block. They were written moments or minutes ago and are almost always
1523    /// still in the page cache, so this is a copy rather than a read of the disk.
1524    fn decoded(&self, file: Option<&File>) -> Result<(Vec<u8>, Vec<u64>)> {
1525        let count = self.placed.len() + self.blocks.len();
1526        if count != self.values().div_ceil(TEXT_PAYLOAD_VALUES) {
1527            return Err(invalid("global dictionary blocks do not cover its values"));
1528        }
1529        let mut bases = Vec::with_capacity(count);
1530        let mut total = 0_usize;
1531        for block in 0..count {
1532            bases.push(total as u64);
1533            let last = ((block + 1) * TEXT_PAYLOAD_VALUES).min(self.values()) - 1;
1534            total = total
1535                .checked_add(self.ends[last] as usize)
1536                .ok_or_else(|| invalid("global dictionary does not fit in memory"))?;
1537        }
1538        let mut flat = vec![0_u8; total];
1539        let mut outs = Vec::with_capacity(count);
1540        let mut rest = flat.as_mut_slice();
1541        for block in 0..count {
1542            let end = bases.get(block + 1).map_or(total, |&base| base as usize);
1543            let (out, after) = rest.split_at_mut(end - bases[block] as usize);
1544            outs.push((block, out));
1545            rest = after;
1546        }
1547        let one = |run: &mut [(usize, &mut [u8])]| -> Result<()> {
1548            let mut stored = Vec::new();
1549            for (block, out) in run {
1550                let encoded = match self.placed.get(*block) {
1551                    Some(place) => {
1552                        let file = file.ok_or_else(|| {
1553                            Error::internal("a written dictionary block has no file")
1554                        })?;
1555                        let length = usize::try_from(place.length).map_err(|_| {
1556                            invalid("global dictionary block does not fit in memory")
1557                        })?;
1558                        stored.resize(length, 0);
1559                        read_at(file, place.start, &mut stored)?;
1560                        if checksum(&stored) != place.hash {
1561                            return Err(invalid(
1562                                "a global dictionary block did not read back as written",
1563                            ));
1564                        }
1565                        stored.as_slice()
1566                    }
1567                    None => &self.blocks[*block - self.placed.len()],
1568                };
1569                let decoded = string::decode_flat(encoded)?;
1570                if decoded.bytes().len() != out.len() {
1571                    return Err(invalid(
1572                        "a global dictionary block is not the length its ends say",
1573                    ));
1574                }
1575                out.copy_from_slice(decoded.bytes());
1576            }
1577            Ok(())
1578        };
1579        // Sixteen blocks a thread at the least, because a thread costs about what decoding a few
1580        // blocks does and most columns have one or two.
1581        let workers = close_workers().min(count / 16).max(1);
1582        if workers <= 1 {
1583            one(&mut outs)?;
1584        } else {
1585            let per = count.div_ceil(workers);
1586            std::thread::scope(|scope| {
1587                outs.chunks_mut(per)
1588                    .map(|run| scope.spawn(|| one(run)))
1589                    .collect::<Vec<_>>()
1590                    .into_iter()
1591                    .try_for_each(|handle| {
1592                        handle.join().map_err(|_| {
1593                            Error::internal("a global dictionary decode worker panicked")
1594                        })?
1595                    })
1596            })?;
1597        }
1598        drop(outs);
1599        Ok((flat, bases))
1600    }
1601
1602    /// Where the value at `code` sits in the bytes [`GlobalDictionary::decoded`] handed back.
1603    ///
1604    /// A block's first value starts at the block, and every other value starts where the one before
1605    /// it ended, which is what makes 1,024 values 1,024 numbers rather than 1,025.
1606    fn value_span(ends: &[u32], bases: &[u64], code: usize) -> (usize, usize) {
1607        let Some(&base) = bases.get(code / TEXT_PAYLOAD_VALUES) else { return (0, 0) };
1608        let Some(&end) = ends.get(code) else { return (0, 0) };
1609        let base = base as usize;
1610        let from = if code % TEXT_PAYLOAD_VALUES == 0 { 0 } else { ends[code - 1] as usize };
1611        (base + from, base + end as usize)
1612    }
1613
1614    /// This dictionary's values in sorted order, each as the first eight bytes of the value and the
1615    /// code that holds it, so entry `rank` describes the value that sits at `rank` when the values
1616    /// are sorted by their bytes.
1617    ///
1618    /// Codes themselves stay in first appearance order, which is what lets the writer hand one out
1619    /// the moment it sees a value rather than waiting for the last stripe, and which also keeps a
1620    /// stripe's codes close together because the data is clustered. This is what puts the values
1621    /// back in order for anything that needs it, and it is separate from the codes so that getting
1622    /// it costs a sort of the distinct values at the end rather than a rewrite of every code page.
1623    ///
1624    /// The order is the byte order of the values and nothing else. The heads are attached after the
1625    /// sort rather than sorted on, because padding with zero on the right is order preserving for
1626    /// byte strings and so sorting by head and then by bytes lands in the same place as sorting by
1627    /// bytes: a shorter value differs from a longer one that starts the same way at a position
1628    /// where the shorter one has run out, and zero is below every byte that could be there.
1629    ///
1630    /// The heads are kept because a reader searching this order wants a comparison it can make out
1631    /// of the index alone. What they buy there depends entirely on the column and is much less than
1632    /// it looks on the columns that cost the most, which [`sort_by_value`] measures.
1633    fn ranked_with_values(&self, file: Option<&File>) -> Result<RankedDictionary> {
1634        let (flat, bases) = self.decoded(file)?;
1635        let value = |code: u32| {
1636            let (from, to) = Self::value_span(&self.ends, &bases, code as usize);
1637            flat.get(from..to).unwrap_or_default()
1638        };
1639        let mut codes = (0..self.values() as u32).collect::<Vec<_>>();
1640        sort_by_value_across(&mut codes, value, close_workers());
1641        let order = codes.into_iter().map(|code| (head(value(code)), code)).collect();
1642        Ok((order, flat, bases))
1643    }
1644
1645    #[cfg(test)]
1646    fn ranked(&self, file: Option<&File>) -> Result<Vec<(u64, u32)>> {
1647        self.ranked_with_values(file).map(|(order, _, _)| order)
1648    }
1649}
1650
1651/// Appends pages and commits a new directory.
1652///
1653/// One writer covers a whole file rather than one table. [`Writer::next`] closes the table it is on
1654/// and opens another over the same file, and [`Writer::finish`] commits every table it has closed in
1655/// one generation. That is what makes a checkpoint atomic across tables: there is one slot write at
1656/// the end of it and a reader sees every table at the generation before it or every table at the
1657/// generation after it.
1658#[derive(Debug)]
1659pub struct Writer {
1660    file: File,
1661    /// Where the next write goes, counted here rather than asked of the file.
1662    ///
1663    /// The file's own cursor is not ours. Building the numeric frequencies reads pages back through
1664    /// [`read_at`], and a positional read is only positional about where it reads from: `pread`
1665    /// leaves the cursor alone, and the call Windows has for it moves the cursor to the end of what
1666    /// it read. A writer that asked the file where it was would then write the directory over a
1667    /// page it had already written, which is what it did.
1668    at: u64,
1669    /// How far into the file the kernel has been asked to start writing, see [`WRITEBACK_STRETCH`].
1670    written_back: u64,
1671    table: Table,
1672    generation: u64,
1673    /// The first and the last source position in every stripe, in the order the stripes were
1674    /// written.
1675    order: Vec<((u64, u64), (u64, u64))>,
1676    next_order: u64,
1677    dictionaries: Vec<Option<GlobalDictionary>>,
1678    /// Which columns still have a global dictionary, shared with every [`Preparer`] this writer
1679    /// hands out so that a stripe prepared after a column lost its dictionary is not coded for it.
1680    coded: Arc<[AtomicBool]>,
1681    /// One per column, folding the rows into a summary and a sketch as they go past.
1682    ///
1683    /// `None` for a column with no hash rule, which is the interval and the nested types. See
1684    /// [`stats::Gather`] for why the statistics are built here rather than by reading the file back
1685    /// once it is committed.
1686    gathers: Vec<Option<stats::Gather>>,
1687    /// The dictionaries and the statistics while a [`Merger`] has them, which is from
1688    /// [`Writer::merger`] until the table is closed. `dictionaries` and `gathers` are empty then.
1689    lent: Option<Arc<Lent>>,
1690    pending: Vec<PendingChunk>,
1691    /// The tables already closed in this generation, in the order they were written.
1692    closed: Vec<Entry>,
1693    /// The views the next commit writes down, which [`Writer::with_views`] sets.
1694    ///
1695    /// Carried forward from the committed generation by [`Writer::open`], so a writer that was only
1696    /// opened to append a table does not have to know about views to avoid dropping them.
1697    views: Vec<ViewEntry>,
1698    /// Where the stages this writer runs are charged, which [`Writer::with_profile`] sets.
1699    ///
1700    /// The writer runs the page builder, the dictionary blocks, the writes and the publish, and it
1701    /// charges them once per stripe and once per worker, never per chunk. See
1702    /// `rudb_metrics::LoadProfile` for why that is the grain.
1703    profile: Option<Arc<LoadProfile>>,
1704}
1705
1706/// A chunk that has arrived and is waiting for the rest of its stripe.
1707///
1708/// The rows are kept rather than the pages they encode to, which is the whole of #808's first half.
1709/// Encoding on arrival put every column of every part on the thread that called `append_at`, and
1710/// that thread is the only one the load has. Encoding at the flush instead means a stripe's worth
1711/// of work is on the table at once, and a stripe splits by column into a hundred and five pieces
1712/// that share nothing.
1713#[derive(Debug)]
1714struct PendingChunk {
1715    order: (u64, u64),
1716    chunk: Chunk,
1717}
1718
1719/// What the writer still needs of a part once its columns are encoded: where in the source it came
1720/// from, how many rows it has and how large those rows were.
1721///
1722/// A stripe waiting for the writer's lock carries these rather than its chunks, so its rows are
1723/// freed as soon as they are encoded and not after the stripe is written. See [`prepare`].
1724#[derive(Debug, Clone, Copy)]
1725struct Part {
1726    order: (u64, u64),
1727    rows: usize,
1728    footprint: usize,
1729}
1730
1731impl Part {
1732    fn of(pending: &PendingChunk) -> Self {
1733        Self {
1734            order: pending.order,
1735            rows: pending.chunk.len(),
1736            footprint: pending.chunk.footprint(),
1737        }
1738    }
1739}
1740
1741/// One column's share of a stripe, which is what one encode worker produces.
1742///
1743/// Indexed by part, so a stripe is a column of these and the write loop reads down one of them.
1744/// That is also the order the loop wanted: `flush_pending` walks a column at a time and lays its
1745/// parts next to each other, and it used to reach across a row of parts to do it.
1746#[derive(Debug)]
1747struct ColumnStripe {
1748    pages: Vec<Vec<u8>>,
1749    codes: Vec<Option<Vec<u32>>>,
1750    sieves: Vec<Option<Sieve>>,
1751    ranges: Vec<Range>,
1752}
1753
1754/// Roughly what encoding a column of this type costs, for ordering the encode queue.
1755///
1756/// Only the order matters and only roughly. A string column hashes and copies every value into a
1757/// dictionary and is in a different class from everything else, and among the fixed widths the wide
1758/// ones carry more bytes through the cascade than the narrow ones. Anything finer than that would
1759/// be a cost model, and the queue already absorbs a wrong guess: it only has to avoid finishing on
1760/// a column nobody else can help with.
1761fn weight(ty: &LogicalType) -> usize {
1762    match ty {
1763        LogicalType::Varchar | LogicalType::Blob | LogicalType::Bit => 64,
1764        LogicalType::HugeInt
1765        | LogicalType::UHugeInt
1766        | LogicalType::Uuid
1767        | LogicalType::Interval => 16,
1768        LogicalType::BigInt
1769        | LogicalType::UBigInt
1770        | LogicalType::Timestamp
1771        | LogicalType::Time
1772        | LogicalType::TimeTz
1773        | LogicalType::TimestampTz
1774        | LogicalType::TimestampS
1775        | LogicalType::TimestampMs
1776        | LogicalType::TimestampNs
1777        | LogicalType::Double
1778        | LogicalType::Decimal { .. } => 8,
1779        LogicalType::Integer | LogicalType::UInteger | LogicalType::Date | LogicalType::Float => 4,
1780        LogicalType::SmallInt | LogicalType::USmallInt => 2,
1781        _ => 1,
1782    }
1783}
1784
1785/// Parts in one stripe.
1786///
1787/// Sixty four thousand rows is the smallest stripe that keeps the ClickBench directory in single
1788/// digit megabytes at a hundred million rows, and it puts a four byte column's page at a quarter of
1789/// a megabyte, which is the size a sequential read wants. Larger stripes buy a smaller directory
1790/// and cost a sparse fetch, which has to read a page index before it can reach one part.
1791pub const STRIPE_PARTS: usize = 64;
1792
1793/// How many rows the writer wants to see before it decides whether a varchar column gets to keep
1794/// its global dictionary.
1795///
1796/// See [`prepare::drops_dictionary`]. A stripe is up to [`STRIPE_PARTS`] parts, so most tables give it
1797/// far more than this and it binds only on a table that is smaller than one stripe. A handful of
1798/// rows says nothing about whether a column repeats itself, and the answer that costs nothing when
1799/// the sample is that small is the one the writer has always given, which is to keep the dictionary.
1800const DICTIONARY_DECIDE_ROWS: usize = 4_096;
1801
1802/// Out of ten. A varchar column loses its dictionary when more than this many rows in ten of the
1803/// first stripe held a value that stripe had not seen before.
1804///
1805/// See [`prepare::drops_dictionary`]. Nine and not five, because the properties a dictionary buys are
1806/// worth keeping everywhere they are real. On ClickBench the widest string column is `Referer` at
1807/// 0.131 of its first stripe and every other one is below that, so nothing there is near this and
1808/// every one of them keeps its dictionary, which is what a group by on codes wants. On TPC-H
1809/// `o_comment` and `c_comment` are at 0.97 and are what this catches.
1810///
1811/// `l_comment` sits at 0.883 and so keeps its dictionary. Eight was built and measured rather than
1812/// argued about, and it is not a clear win: it takes `select l_comment from lineitem` from 4.335 G
1813/// instructions to 3.473 G and the file from 280.2 MB to 260.4 MB, and it takes a `like` over the
1814/// same column from 3.29 G to 4.27 G, because a dictionary runs the predicate once a distinct value
1815/// and there are 3.6 M of those to 6.0 M rows. The 22 query suite came out 6.91 s against 7.07 s in
1816/// favour of nine. So nine stays until there is a reason to prefer one of those shapes. See #1137.
1817const DICTIONARY_DISTINCT_IN_TEN: usize = 9;
1818
1819/// Bytes one part takes in a stripe's index page: four for the length, eight for the checksum.
1820const INDEX_ENTRY: usize = size_of::<u32>() + size_of::<u64>();
1821
1822/// Bytes one column's section of a stripe's index page takes, including its own trailing checksum.
1823fn index_section(parts: usize) -> Result<usize> {
1824    parts
1825        .checked_mul(INDEX_ENTRY)
1826        .and_then(|bytes| bytes.checked_add(size_of::<u64>()))
1827        .ok_or_else(|| invalid("index page length overflow"))
1828}
1829
1830impl Writer {
1831    /// Opens a committed file and starts a table in the generation after the one it holds.
1832    ///
1833    /// The tables already in the file are carried forward by name and by directory pointer, and
1834    /// their pages are not read. Nothing in the file is overwritten: the new table's pages and the
1835    /// new catalog go on the end, past the catalog the committed generation points at, and the one
1836    /// write that is not an append is the slot in the header that [`Writer::finish`] does last.
1837    ///
1838    /// That slot is the other one. A file committed at generation 1 is named by the slot at 16 and
1839    /// generation 2 writes the one at 44, so until the last four bytes of the commit land the file
1840    /// still reads as the generation before it, and a slot torn across a write fails its checksum
1841    /// and the reader falls back to the one beside it. This is what the second slot has always been
1842    /// for.
1843    ///
1844    /// # Errors
1845    ///
1846    /// If the file has no valid committed directory, is not this build's format, repeats the name
1847    /// of a table already in it that holds rows, has a field with no scalar encoding, or cannot be
1848    /// written.
1849    pub fn open(
1850        path: impl AsRef<Path>,
1851        name: impl Into<String>,
1852        fields: Vec<Field>,
1853    ) -> Result<Self> {
1854        for field in &fields {
1855            type_tag(&field.ty)?;
1856        }
1857        let name = name.into();
1858        let path = path.as_ref();
1859        let (_, size, slot, bytes, _) = slot_bytes(path)?;
1860        let (mut closed, views) = decode_catalog(&bytes, size)?;
1861        // A table already in the file under this name is only in the way if it holds rows. One that
1862        // holds none has no pages for this generation to carry and no reader that could lose
1863        // anything, so the table being started here takes its place in the catalog rather than
1864        // colliding with it, and `finish` writes the new entry where the old one was.
1865        //
1866        // That is not a corner. It is the shape every loading script writes: the schema goes in one
1867        // statement and the rows go in the next, and a checkpoint between them commits the empty
1868        // table. Before this, the second statement had to build the whole table in memory because
1869        // the first had already put the name in the file, which is how a load of a table larger
1870        // than memory became a load that needed memory the size of the table.
1871        if let Some(at) = closed.iter().position(|held| held.name == name) {
1872            if closed[at].rows > 0 {
1873                return Err(invalid("two tables in one native file have the same name"));
1874            }
1875            closed.remove(at);
1876        }
1877        // The generation of the slot whose bytes checksummed, and not the highest number in the
1878        // header. A slot torn across a write can hold any number at all, and taking that one would
1879        // be choosing which slot to overwrite from a value nothing has vouched for, which is how a
1880        // half written commit gets to destroy the one good copy beside it.
1881        let generation = slot
1882            .generation
1883            .checked_add(1)
1884            .ok_or_else(|| invalid("native file generation overflow"))?;
1885        let file = OpenOptions::new().write(true).read(true).open(path).map_err(io)?;
1886        Ok(Self {
1887            file,
1888            // The end of the file, so that the committed generation's catalog stays where its slot
1889            // says it is and keeps naming a file a reader can still open.
1890            at: size,
1891            written_back: size,
1892            dictionaries: fields
1893                .iter()
1894                .map(|field| (field.ty == LogicalType::Varchar).then(GlobalDictionary::new))
1895                .collect(),
1896            coded: fields
1897                .iter()
1898                .map(|field| AtomicBool::new(field.ty == LogicalType::Varchar))
1899                .collect(),
1900            gathers: fields.iter().map(|field| stats::Gather::new(&field.ty, generation)).collect(),
1901            lent: None,
1902            table: Table {
1903                name,
1904                dictionaries: vec![None; fields.len()],
1905                dictionary_payloads: Vec::new(),
1906                distincts: vec![None; fields.len()],
1907                fields,
1908                stripes: Vec::new(),
1909                rows: 0,
1910                frequencies: Vec::new(),
1911                pair_frequencies: Vec::new(),
1912                frequency_texts: Vec::new(),
1913                host_groups: None,
1914                clustering: None,
1915                generation,
1916                sections: Vec::new(),
1917            },
1918            generation,
1919            order: Vec::new(),
1920            next_order: 0,
1921            pending: Vec::with_capacity(STRIPE_PARTS),
1922            closed,
1923            views,
1924            profile: None,
1925        })
1926    }
1927
1928    /// Creates a new v10 file and its first table.
1929    ///
1930    /// # Errors
1931    ///
1932    /// If the file exists, a field has no scalar encoding, or the path cannot be written.
1933    pub fn create(
1934        path: impl AsRef<Path>,
1935        name: impl Into<String>,
1936        fields: Vec<Field>,
1937    ) -> Result<Self> {
1938        for field in &fields {
1939            type_tag(&field.ty)?;
1940        }
1941        let file =
1942            OpenOptions::new().write(true).read(true).create_new(true).open(path).map_err(io)?;
1943        let mut header = [0; HEADER as usize];
1944        header[..8].copy_from_slice(MAGIC);
1945        header[8..12].copy_from_slice(&FORMAT.to_le_bytes());
1946        write_at(&file, 0, &header)?;
1947        Ok(Self {
1948            file,
1949            at: HEADER,
1950            written_back: HEADER,
1951            dictionaries: fields
1952                .iter()
1953                .map(|field| (field.ty == LogicalType::Varchar).then(GlobalDictionary::new))
1954                .collect(),
1955            coded: fields
1956                .iter()
1957                .map(|field| AtomicBool::new(field.ty == LogicalType::Varchar))
1958                .collect(),
1959            gathers: fields.iter().map(|field| stats::Gather::new(&field.ty, 1)).collect(),
1960            lent: None,
1961            table: Table {
1962                name: name.into(),
1963                dictionaries: vec![None; fields.len()],
1964                dictionary_payloads: Vec::new(),
1965                distincts: vec![None; fields.len()],
1966                fields,
1967                stripes: Vec::new(),
1968                rows: 0,
1969                frequencies: Vec::new(),
1970                pair_frequencies: Vec::new(),
1971                frequency_texts: Vec::new(),
1972                host_groups: None,
1973                clustering: None,
1974                generation: 1,
1975                sections: Vec::new(),
1976            },
1977            generation: 1,
1978            order: Vec::new(),
1979            next_order: 0,
1980            pending: Vec::with_capacity(STRIPE_PARTS),
1981            closed: Vec::new(),
1982            views: Vec::new(),
1983            profile: None,
1984        })
1985    }
1986
1987    /// Creates a new file that holds no table at all, committed and ready to open.
1988    ///
1989    /// A database somebody dropped the last table out of is still a database, and until this there
1990    /// was no way to write one down. Every other way into this file goes through a table, because
1991    /// [`Writer::create`] takes the first one and [`Writer::finish`] commits the one it is on, so a
1992    /// catalog with nothing in it could be read and not written. The format already allowed it: the
1993    /// catalog is a count and that many entries, and a count of nought encodes and decodes the same
1994    /// way every other count does, which is why nothing here is a version change.
1995    ///
1996    /// It hands back nothing rather than a writer, because a writer with no table is a writer with
1997    /// nothing to append to. A file that is going to hold a table is [`Writer::create`], and one
1998    /// that is going to have a table added to it later is [`Writer::open`], which reads what this
1999    /// wrote the same way it reads any other generation.
2000    ///
2001    /// It takes the views anyway, because a database with no table can still have views in it. A
2002    /// view over `range` or over another view names no table, so dropping the last table out of a
2003    /// database does not have to leave the catalog with nothing worth writing down.
2004    ///
2005    /// # Errors
2006    ///
2007    /// If the file exists or the path cannot be written.
2008    pub fn empty(path: impl AsRef<Path>, views: &[ViewEntry]) -> Result<()> {
2009        let file =
2010            OpenOptions::new().write(true).read(true).create_new(true).open(path).map_err(io)?;
2011        let mut header = [0; HEADER as usize];
2012        header[..8].copy_from_slice(MAGIC);
2013        header[8..12].copy_from_slice(&FORMAT.to_le_bytes());
2014        write_at(&file, 0, &header)?;
2015        let catalog = encode_catalog(&[], views)?;
2016        write_at(&file, HEADER, &catalog)?;
2017        // The same two syncs in the same order as [`Writer::finish`], and for the same reason. The
2018        // catalog is on the disk before the slot names it, so a file this is interrupted in the
2019        // middle of is a header with no valid slot rather than a slot pointing at nothing.
2020        file.sync_all().map_err(io)?;
2021        let slot = Slot {
2022            offset: HEADER,
2023            length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
2024            generation: 1,
2025            hash: checksum(&catalog),
2026        };
2027        write_at(&file, slot_offset(1), &slot.bytes())?;
2028        file.sync_all().map_err(io)?;
2029        Ok(())
2030    }
2031
2032    /// Closes the table this writer is on and starts another one in the same file.
2033    ///
2034    /// Nothing is published here. The closed table's directory is written so that the bytes are on
2035    /// disk and its span is known, and the catalog that names it is only written by
2036    /// [`Writer::finish`], so a crash between two tables leaves the previous generation intact.
2037    ///
2038    /// # Errors
2039    ///
2040    /// If the name repeats a table already closed, a field has no scalar encoding, or the table
2041    /// being closed cannot be written.
2042    pub fn next(mut self, name: impl Into<String>, fields: Vec<Field>) -> Result<Self> {
2043        for field in &fields {
2044            type_tag(&field.ty)?;
2045        }
2046        let name = name.into();
2047        let entry = self.close()?;
2048        if self.closed.iter().chain(std::iter::once(&entry)).any(|held| held.name == name) {
2049            return Err(invalid("two tables in one native file have the same name"));
2050        }
2051        let Self { file, at, generation, mut closed, views, .. } = self;
2052        closed.push(entry);
2053        Ok(Self {
2054            file,
2055            written_back: at,
2056            at,
2057            generation,
2058            closed,
2059            views,
2060            profile: None,
2061            dictionaries: fields
2062                .iter()
2063                .map(|field| (field.ty == LogicalType::Varchar).then(GlobalDictionary::new))
2064                .collect(),
2065            coded: fields
2066                .iter()
2067                .map(|field| AtomicBool::new(field.ty == LogicalType::Varchar))
2068                .collect(),
2069            gathers: fields.iter().map(|field| stats::Gather::new(&field.ty, generation)).collect(),
2070            lent: None,
2071            table: Table {
2072                name,
2073                dictionaries: vec![None; fields.len()],
2074                dictionary_payloads: Vec::new(),
2075                distincts: vec![None; fields.len()],
2076                fields,
2077                stripes: Vec::new(),
2078                rows: 0,
2079                frequencies: Vec::new(),
2080                pair_frequencies: Vec::new(),
2081                frequency_texts: Vec::new(),
2082                host_groups: None,
2083                clustering: None,
2084                generation,
2085                sections: Vec::new(),
2086            },
2087            order: Vec::new(),
2088            next_order: 0,
2089            pending: Vec::with_capacity(STRIPE_PARTS),
2090        })
2091    }
2092
2093    /// Sets the views the next commit writes down, replacing whatever was carried forward.
2094    ///
2095    /// It replaces rather than adds because the caller has the whole catalog in front of it and the
2096    /// writer does not. A view that was dropped is a view that is not in the list any more, and
2097    /// there is no other way for the writer to hear about that, since nothing else it is told about
2098    /// mentions views at all.
2099    ///
2100    /// A writer that is never told anything writes back the views it read at [`Writer::open`], so a
2101    /// checkpoint that only had a table to append does not quietly drop them.
2102    #[must_use]
2103    pub fn with_views(mut self, views: Vec<ViewEntry>) -> Self {
2104        self.views = views;
2105        self
2106    }
2107
2108    /// Charges the stages this writer runs to `profile`.
2109    ///
2110    /// For the table being written now. [`Writer::next`] starts the next table without one,
2111    /// because a second table's stripes charged to the first table's load would be a profile of
2112    /// neither.
2113    #[must_use]
2114    pub fn with_profile(mut self, profile: Arc<LoadProfile>) -> Self {
2115        self.profile = Some(profile);
2116        self
2117    }
2118
2119    /// Records the order this table's rows are meant to be stored in.
2120    ///
2121    /// The declaration goes in the table directory and comes back out of
2122    /// [`Table::clustering`]. Nothing here sorts anything, and nothing here checks that the rows
2123    /// handed to [`Writer::append`] arrive in the order this claims. That is deliberate for now:
2124    /// the thing that was missing was a place to write the order down, and a loader that honours
2125    /// the declaration is the next piece rather than this one.
2126    ///
2127    /// The declaration applies to the table the writer is currently on, so it is set after
2128    /// [`Writer::next`] rather than once for the file.
2129    ///
2130    /// # Errors
2131    ///
2132    /// If the declaration names a column this table does not have.
2133    pub fn declare(mut self, clustering: Clustering) -> Result<Self> {
2134        // Rebuilt against this table's own column count rather than trusted, because the caller
2135        // built it against a catalog entry and the two could have drifted.
2136        self.table.clustering = Some(Clustering::new(
2137            clustering.columns().to_vec(),
2138            clustering.width(),
2139            &self.table.fields,
2140        )?);
2141        Ok(self)
2142    }
2143
2144    /// Appends bytes at the end of the file and moves the writer's own offset past them.
2145    ///
2146    /// Every write in here goes through this, so that [`Writer::at`] is the only answer to where
2147    /// anything is and the file's cursor is never consulted for it.
2148    fn put(&mut self, bytes: &[u8]) -> Result<()> {
2149        write_at(&self.file, self.at, bytes)?;
2150        self.at = self
2151            .at
2152            .checked_add(bytes.len() as u64)
2153            .ok_or_else(|| invalid("native file length overflow"))?;
2154        if self.at - self.written_back >= WRITEBACK_STRETCH {
2155            rudb_io::start_writeback(&self.file, self.written_back, self.at - self.written_back);
2156            self.written_back = self.at;
2157        }
2158        Ok(())
2159    }
2160
2161    /// Writes one chunk as independently readable column pages.
2162    ///
2163    /// # Errors
2164    ///
2165    /// If its width or types differ from the declared table, or a page exceeds its bound.
2166    pub fn append(&mut self, chunk: &Chunk) -> Result<()> {
2167        let order = (self.next_order, 0);
2168        self.next_order = self.next_order.saturating_add(1);
2169        self.append_at(order, chunk)
2170    }
2171
2172    /// Writes one chunk and records its source position for directory ordering.
2173    ///
2174    /// Pages may be encoded by parallel pipeline instances and reach the file in completion order.
2175    /// The stripe they land in is sorted by this key at commit, and [`Self::finish`] rejects a
2176    /// sequence whose parts do not come out in source order once the stripes are sorted, because a
2177    /// stripe groups whatever arrived together and cannot put a late part back where it belongs.
2178    ///
2179    /// # Errors
2180    ///
2181    /// The same as [`Self::append`].
2182    pub fn append_at(&mut self, order: (u64, u64), chunk: &Chunk) -> Result<()> {
2183        if chunk.is_empty() {
2184            return Ok(());
2185        }
2186        self.admit(chunk)?;
2187        if self.pending.last().is_some_and(|last| last.order > order) {
2188            self.flush_pending()?;
2189        }
2190        // Cloned rather than encoded, and a clone of a chunk that owns its buffers is a copy of
2191        // them. Sixty four parts of a hundred and five columns is tens of megabytes held for the
2192        // length of a stripe and a few seconds of memory traffic over a whole ClickBench load,
2193        // against the hundreds of seconds of encode this is what lets off one thread.
2194        self.pending.push(PendingChunk { order, chunk: chunk.clone() });
2195        if self.pending.len() == STRIPE_PARTS {
2196            self.flush_pending()?;
2197        }
2198        Ok(())
2199    }
2200
2201    /// Writes a run of chunks as one stripe of its own.
2202    ///
2203    /// [`Self::append_at`] decides where a stripe ends by watching the orders go past, which works
2204    /// when one caller hands over every chunk in source order and does not when several do. A
2205    /// writer being fed by more than one pipeline instance sees the orders interleave, and a stripe
2206    /// that ends every time two of them cross is a stripe of one or two parts.
2207    ///
2208    /// So the grouping moves to the caller. Whoever is buffering hands over a run it already knows
2209    /// is contiguous and in order, and gets a stripe holding exactly that run. The orders still
2210    /// have to come out in source order once the stripes are sorted, which [`Self::finish`] checks,
2211    /// so the runs from different callers may interleave with each other but may not overlap.
2212    ///
2213    /// # Errors
2214    ///
2215    /// The same as [`Self::append`], and if the run is longer than [`STRIPE_PARTS`].
2216    pub fn append_stripe(&mut self, parts: Vec<((u64, u64), Chunk)>) -> Result<()> {
2217        if parts.len() > STRIPE_PARTS {
2218            return Err(invalid("a stripe was handed more parts than it holds"));
2219        }
2220        // Whatever an earlier caller left behind is its own stripe rather than the front of this
2221        // one, because the two runs are from different places in the source and a stripe is a run.
2222        self.flush_pending()?;
2223        for (order, chunk) in parts {
2224            if chunk.is_empty() {
2225                continue;
2226            }
2227            self.admit(&chunk)?;
2228            self.pending.push(PendingChunk { order, chunk });
2229        }
2230        self.flush_pending()
2231    }
2232
2233    /// Checks a chunk against the declared table and counts its rows in.
2234    fn admit(&mut self, chunk: &Chunk) -> Result<()> {
2235        if chunk.width() != self.table.fields.len() {
2236            return Err(invalid("chunk width differs from table schema"));
2237        }
2238        for (index, field) in self.table.fields.iter().enumerate() {
2239            if chunk.column(index)?.logical_type() != &field.ty {
2240                return Err(invalid("chunk type differs from table schema"));
2241            }
2242        }
2243        self.table.rows = self
2244            .table
2245            .rows
2246            .checked_add(chunk.len())
2247            .ok_or_else(|| invalid("row count overflow"))?;
2248        Ok(())
2249    }
2250
2251    /// One column's parts of a stripe as pages, for a column with no global dictionary.
2252    fn encode_pages(columns: &[&Vector]) -> Result<ColumnStripe> {
2253        let mut stripe = ColumnStripe {
2254            pages: Vec::with_capacity(columns.len()),
2255            codes: Vec::with_capacity(columns.len()),
2256            sieves: Vec::with_capacity(columns.len()),
2257            ranges: Vec::with_capacity(columns.len()),
2258        };
2259        let mut settling = Settling::default();
2260        for &column in columns {
2261            let bytes = encode(column, &mut settling)?;
2262            if bytes.len() > MAX_PAGE {
2263                return Err(invalid("column page exceeds the configured bound"));
2264            }
2265            // The range is built first because the sieve reads it rather than walking the column a
2266            // second time to find out how wide it is.
2267            let range = Range::of(column);
2268            // A sieve at least as large as the part it indexes is not written. A reader reads the
2269            // sieve to decide whether to read the part, so when the sieve is the larger of the two
2270            // it has already spent more than the read it is trying to avoid, and that holds even if
2271            // it rejects every time. It is a necessary condition rather than the whole rule, which
2272            // is that a sieve pays when its bytes are under the rejection rate times the part's,
2273            // but the rejection rate depends on what a query probes for and the writer does not
2274            // know that. The necessary half needs two numbers that are both in hand here.
2275            //
2276            // A column with a global dictionary gets none, because it already has an exact
2277            // membership index per stripe. Those do not come through here. See [`prepare`].
2278            let sieve =
2279                Sieve::of(column, &range, SIEVE_BUDGET).filter(|sieve| sieve.len() < bytes.len());
2280            stripe.pages.push(bytes);
2281            stripe.codes.push(None);
2282            stripe.sieves.push(sieve);
2283            stripe.ranges.push(range);
2284        }
2285        Ok(stripe)
2286    }
2287
2288    /// Writes every encoded dictionary block that is not in the file yet and forgets its bytes.
2289    ///
2290    /// This is what keeps a load from holding its dictionaries' payload. The blocks land between
2291    /// stripes wherever the writer is, which is fine because the index says where each one is.
2292    fn place_blocks(&mut self) -> Result<()> {
2293        if let Some(lent) = self.lent.clone() {
2294            return self.place_lent_blocks(&lent);
2295        }
2296        let mut dictionaries = std::mem::take(&mut self.dictionaries);
2297        let placed = dictionaries.iter_mut().flatten().try_for_each(|dictionary| {
2298            for block in std::mem::take(&mut dictionary.blocks) {
2299                let start = self.at;
2300                self.put(&block)?;
2301                dictionary.placed.push(Placed {
2302                    start,
2303                    length: block.len() as u64,
2304                    hash: checksum(&block),
2305                });
2306            }
2307            Ok(())
2308        });
2309        self.dictionaries = dictionaries;
2310        placed
2311    }
2312
2313    /// [`Writer::place_blocks`] while a [`Merger`] has the dictionaries.
2314    ///
2315    /// A column whose merge is running is passed over rather than waited for, because the writer's
2316    /// lock is held here and a merge of `URL` can take tens of milliseconds. Its blocks go out with
2317    /// a later stripe, or at the close.
2318    fn place_lent_blocks(&mut self, lent: &Lent) -> Result<()> {
2319        for column in lent.columns() {
2320            let Ok(mut held) = column.try_lock() else { continue };
2321            let Some(dictionary) = held.dictionary.as_mut() else { continue };
2322            for block in std::mem::take(&mut dictionary.blocks) {
2323                let start = self.at;
2324                self.put(&block)?;
2325                dictionary.placed.push(Placed {
2326                    start,
2327                    length: block.len() as u64,
2328                    hash: checksum(&block),
2329                });
2330            }
2331        }
2332        Ok(())
2333    }
2334
2335    /// Takes the dictionaries and the statistics back from the [`Merger`] that has them.
2336    ///
2337    /// A merge that starts after this is refused, since whatever it merged would be lost.
2338    fn reclaim(&mut self) -> Result<()> {
2339        let Some(lent) = self.lent.take() else { return Ok(()) };
2340        let (dictionaries, gathers) = lent.reclaim()?;
2341        self.dictionaries = dictionaries;
2342        self.gathers = gathers;
2343        Ok(())
2344    }
2345
2346    /// Writes the buffered parts as one stripe, each column's parts contiguous on disk.
2347    ///
2348    /// The same four steps a caller holding this writer behind a lock takes, with nobody else
2349    /// waiting between them. See [`prepare`].
2350    fn flush_pending(&mut self) -> Result<()> {
2351        if self.pending.is_empty() {
2352            return Ok(());
2353        }
2354        let held = std::mem::take(&mut self.pending);
2355        let prepared = self.preparer().prepare_held(held)?;
2356        let merged = self.merge_held(prepared)?;
2357        let paged = merged.pages()?;
2358        self.write_paged(paged)
2359    }
2360
2361    /// Writes one stripe whose pages are built, each column's parts contiguous on disk.
2362    fn write_stripe(&mut self, held: &[Part], encoded: Vec<ColumnStripe>) -> Result<()> {
2363        let width = self.table.fields.len();
2364        let parts = held.len();
2365        if encoded.len() != width {
2366            return Err(Error::internal("a stripe came to the writer with the wrong columns"));
2367        }
2368        let profile = self.profile.clone();
2369        if let Some(profile) = &profile {
2370            let rows = held.iter().map(|part| part.rows as u64).sum();
2371            let raw = held.iter().map(|part| part.footprint as u64).sum();
2372            let pages =
2373                encoded.iter().flat_map(|stripe| &stripe.pages).map(|page| page.len() as u64).sum();
2374            profile.moved(Stage::Pages, raw, pages, rows);
2375        }
2376        // Before a byte of the stripe is written, so that the blocks the stripe's pages were built
2377        // with, and any that were waiting on them, are let go of now rather than a stripe later.
2378        let timing = profile.as_deref().map(|profile| profile.span(Stage::Dictionary));
2379        let before = self.at;
2380        self.place_blocks()?;
2381        drop(timing);
2382        if let Some(profile) = &profile {
2383            profile.moved(Stage::Dictionary, 0, self.at - before, 0);
2384        }
2385        let timing = profile.as_deref().map(|profile| profile.span(Stage::Write));
2386        let before = self.at;
2387        let mut pages = Vec::with_capacity(width);
2388        let mut memberships = vec![None; width];
2389        let mut ranges = Vec::with_capacity(width);
2390        let mut index = Vec::with_capacity(width.saturating_mul(index_section(parts)?));
2391        for stripe in &encoded {
2392            let offset = self.at;
2393            let section = index.len();
2394            let mut length = 0_usize;
2395            for bytes in &stripe.pages {
2396                write_at(&self.file, self.at + length as u64, bytes)?;
2397                put_u32(
2398                    &mut index,
2399                    u32::try_from(bytes.len()).map_err(|_| invalid("part length overflow"))?,
2400                );
2401                put_u64(&mut index, checksum(bytes));
2402                length = length
2403                    .checked_add(bytes.len())
2404                    .ok_or_else(|| invalid("column page length overflow"))?;
2405            }
2406            let hash = checksum(&index[section..]);
2407            put_u64(&mut index, hash);
2408            if length > MAX_PAGE {
2409                return Err(invalid("column page exceeds the configured bound"));
2410            }
2411            self.at = self
2412                .at
2413                .checked_add(length as u64)
2414                .ok_or_else(|| invalid("native file length overflow"))?;
2415            pages.push(Span {
2416                offset,
2417                length: u32::try_from(length).map_err(|_| invalid("page length overflow"))?,
2418            });
2419            ranges.push(merged_range(stripe.ranges.iter().cloned()));
2420        }
2421        for (membership, stripe) in memberships.iter_mut().zip(&encoded) {
2422            if stripe.codes.iter().all(Option::is_none) {
2423                continue;
2424            }
2425            let lists = stripe
2426                .codes
2427                .iter()
2428                .map(|codes| codes.clone().unwrap_or_default())
2429                .collect::<Vec<_>>();
2430            let bytes = encode_membership(&merged_codes(lists));
2431            let offset = self.at;
2432            self.put(&bytes)?;
2433            *membership = Some(Page {
2434                offset,
2435                length: u32::try_from(bytes.len())
2436                    .map_err(|_| invalid("membership page length overflow"))?,
2437                hash: checksum(&bytes),
2438            });
2439        }
2440        let mut sieves = vec![None; width];
2441        for (page, stripe) in sieves.iter_mut().zip(&encoded) {
2442            if stripe.sieves.iter().all(Option::is_none) {
2443                continue;
2444            }
2445            let bytes = encode_sieves(stripe.sieves.iter())?;
2446            let offset = self.at;
2447            self.put(&bytes)?;
2448            *page = Some(Page {
2449                offset,
2450                length: u32::try_from(bytes.len())
2451                    .map_err(|_| invalid("sieve page length overflow"))?,
2452                hash: checksum(&bytes),
2453            });
2454        }
2455        // A stripe of one part has the same rows in it as that part, so its own bounds are already
2456        // the part's and a page here would say what the directory says. Everywhere else the page is
2457        // written unless it comes to more than the column it indexes, which is the rule the sieves
2458        // go by and for the same reason: a reader reads this to decide whether to read the column,
2459        // so a page larger than the column has spent more than the read it is avoiding.
2460        let mut part_ranges = vec![None; width];
2461        if parts > 1 {
2462            for ((page, stripe), span) in part_ranges.iter_mut().zip(&encoded).zip(&pages) {
2463                let bytes = encode_part_ranges(&stripe.ranges)?;
2464                if bytes.len() >= span.length as usize {
2465                    continue;
2466                }
2467                let offset = self.at;
2468                self.put(&bytes)?;
2469                *page = Some(Page {
2470                    offset,
2471                    length: u32::try_from(bytes.len())
2472                        .map_err(|_| invalid("part range page length overflow"))?,
2473                    hash: checksum(&bytes),
2474                });
2475            }
2476        }
2477        let offset = self.at;
2478        self.put(&index)?;
2479        let index = Span {
2480            offset,
2481            length: u32::try_from(index.len())
2482                .map_err(|_| invalid("index page length overflow"))?,
2483        };
2484        let mut rows = 0_usize;
2485        let mut lengths = Vec::with_capacity(parts);
2486        let mut span = None;
2487        for part in held {
2488            rows = rows.checked_add(part.rows).ok_or_else(|| invalid("row count overflow"))?;
2489            lengths.push(u32::try_from(part.rows).map_err(|_| invalid("part row count overflow"))?);
2490            span = Some(span.map_or((part.order, part.order), |(first, _)| (first, part.order)));
2491        }
2492        self.order.push(span.ok_or_else(|| invalid("a stripe was flushed with no parts"))?);
2493        self.table.stripes.push(Stripe {
2494            rows,
2495            parts: lengths,
2496            index,
2497            pages,
2498            memberships: Pages::from_slots(memberships)?,
2499            sieves: Pages::from_slots(sieves)?,
2500            part_ranges: Pages::from_slots(part_ranges)?,
2501            zone: Zone::from_ranges(ranges),
2502        });
2503        drop(timing);
2504        if let Some(profile) = &profile {
2505            profile.moved(Stage::Write, 0, self.at - before, rows as u64);
2506        }
2507        Ok(())
2508    }
2509
2510    /// Finds exact heavy hitters without keeping a hash table for every numeric column while the
2511    /// load is live. The pages are already in the target file, so one column at a time uses a
2512    /// bounded Misra-Gries candidate table and then recounts only those candidates.
2513    ///
2514    /// The first of those passes also counts the column's distinct values exactly, up to the cap in
2515    /// [`distinct`], which is the number a string column gets from its dictionary. It comes back
2516    /// beside the summary because a column whose heavy hitters cannot be proved can still have been
2517    /// counted.
2518    ///
2519    /// The tables are keyed by a value's sixty four bits rather than by [`FrequencyValue`], and a
2520    /// null is counted beside them. Every integer type the format stores fits in those bits, so
2521    /// within one column two values share bits only if they are the same value, and a sixteen byte
2522    /// entry keeps the whole candidate table in the second level cache where the forty eight byte
2523    /// one did not. The null takes part in the candidate table exactly as a key would: it holds a
2524    /// place while its count is above zero, and it is decremented with the rest.
2525    fn numeric_frequency(&self, column: usize) -> Result<(Option<FrequencySummary>, Option<u64>)> {
2526        let signed = match self.table.fields[column].ty {
2527            LogicalType::TinyInt
2528            | LogicalType::SmallInt
2529            | LogicalType::Integer
2530            | LogicalType::BigInt
2531            | LogicalType::Date
2532            | LogicalType::Timestamp => true,
2533            LogicalType::UTinyInt
2534            | LogicalType::USmallInt
2535            | LogicalType::UInteger
2536            | LogicalType::UBigInt => false,
2537            _ => return Ok((None, None)),
2538        };
2539        let value_of = |bits: Option<u64>| match bits {
2540            None => FrequencyValue::Null,
2541            Some(bits) if signed => FrequencyValue::Integer(i128::from(bits as i64)),
2542            Some(bits) => FrequencyValue::Integer(i128::from(bits)),
2543        };
2544        // Rows arrive a run of equal values at a time, because a sorted column is runs and a flag
2545        // column is mostly one value, so a run is counted and inserted once rather than per row.
2546        let mut first = Candidates::default();
2547        let mut distinct = distinct::ExactDistinct::new();
2548        let mut run = Run::default();
2549        self.visit_numeric(column, signed, |_, bits| {
2550            if let Some((bits, times)) = run.push(bits) {
2551                first.add(bits, times);
2552            }
2553            if run.times == 1 {
2554                if let Some(bits) = bits {
2555                    distinct.insert(bits);
2556                }
2557            }
2558        })?;
2559        if let Some((bits, times)) = run.take() {
2560            first.add(bits, times);
2561        }
2562        let Candidates { counts: candidates, nulls, decrements } = first;
2563        let (exact, null_count) = if decrements == 0 {
2564            let exact = candidates
2565                .into_iter()
2566                .map(|(bits, count)| (bits, u64::from(count)))
2567                .collect::<FrequencyMap<_>>();
2568            (exact, (nulls != 0).then_some(u64::from(nulls)))
2569        } else {
2570            let mut lower = candidates.values().copied().collect::<Vec<_>>();
2571            if nulls != 0 {
2572                lower.push(nulls);
2573            }
2574            lower.sort_unstable_by(|left, right| right.cmp(left));
2575            if lower.len() < FREQUENCY_BUILD_RANK
2576                || u64::from(lower[FREQUENCY_BUILD_RANK - 1]) <= decrements
2577            {
2578                return Ok((None, distinct.count()));
2579            }
2580            let mut exact =
2581                candidates.into_keys().map(|bits| (bits, 0_u64)).collect::<FrequencyMap<_>>();
2582            let mut null_count = (nulls != 0).then_some(0_u64);
2583            let mut recount = |bits: Option<u64>, times: u32| {
2584                let held = match bits {
2585                    Some(bits) => exact.get_mut(&bits),
2586                    None => null_count.as_mut(),
2587                };
2588                if let Some(count) = held {
2589                    *count = count.saturating_add(u64::from(times));
2590                }
2591            };
2592            let mut run = Run::default();
2593            self.visit_numeric(column, signed, |_, bits| {
2594                if let Some((bits, times)) = run.push(bits) {
2595                    recount(bits, times);
2596                }
2597            })?;
2598            if let Some((bits, times)) = run.take() {
2599                recount(bits, times);
2600            }
2601            (exact, null_count)
2602        };
2603        let mut entries = exact
2604            .into_iter()
2605            .map(|(bits, count)| FrequencyEntry { value: value_of(Some(bits)), count })
2606            .chain(null_count.map(|count| FrequencyEntry { value: FrequencyValue::Null, count }))
2607            .collect::<Vec<_>>();
2608        let omitted_max = keep_most_frequent(&mut entries).max(decrements);
2609        let kept_rows = entries.iter().try_fold(0_u64, |total, entry| {
2610            total.checked_add(entry.count).filter(|&total| total <= FREQUENCY_ORDINALS as u64)
2611        });
2612        let mut ordinals = Vec::new();
2613        let mut ordinal_entries = Vec::new();
2614        if let Some(kept_rows) = kept_rows {
2615            let mut kept = FrequencyMap::default();
2616            let mut null_kept = None;
2617            for (at, entry) in entries.iter().enumerate() {
2618                let at = u16::try_from(at)
2619                    .map_err(|_| invalid("too many retained frequency entries"))?;
2620                match entry.value {
2621                    FrequencyValue::Integer(value) => {
2622                        kept.insert(value as u64, at);
2623                    }
2624                    FrequencyValue::Null => null_kept = Some(at),
2625                    FrequencyValue::Code(_) => {}
2626                }
2627            }
2628            ordinals.reserve(usize::try_from(kept_rows).unwrap_or(FREQUENCY_ORDINALS));
2629            ordinal_entries.reserve(usize::try_from(kept_rows).unwrap_or(FREQUENCY_ORDINALS));
2630            self.visit_numeric(column, signed, |ordinal, bits| {
2631                let held = match bits {
2632                    Some(bits) => kept.get(&bits).copied(),
2633                    None => null_kept,
2634                };
2635                if let Some(entry) = held {
2636                    ordinals.push(ordinal);
2637                    ordinal_entries.push(entry);
2638                }
2639            })?;
2640        }
2641        Ok((
2642            Some(FrequencySummary { entries, omitted_max, ordinals, ordinal_entries }),
2643            distinct.count(),
2644        ))
2645    }
2646
2647    /// Hands every row of an integer column to `visit` as its ordinal and its sixty four bits, or
2648    /// `None` for a null.
2649    ///
2650    /// `signed` says which of the two readings the column has. A packed unsigned column would come
2651    /// back from `signed_block` as a base plus a code in `i64`, which wraps for a value past the top
2652    /// of `BIGINT`, so only a signed column takes the block path.
2653    fn visit_numeric(
2654        &self,
2655        column: usize,
2656        signed: bool,
2657        mut visit: impl FnMut(u64, Option<u64>),
2658    ) -> Result<()> {
2659        let ty = &self.table.fields[column].ty;
2660        let mut start = 0_u64;
2661        let mut block = Vec::new();
2662        for stripe in &self.table.stripes {
2663            let spans = read_index(&self.file, stripe, column)?;
2664            let page = stripe.pages[column];
2665            let mut bytes = vec![0; page.length as usize];
2666            read_at(&self.file, page.offset, &mut bytes)?;
2667            for (span, &rows) in spans.iter().zip(&stripe.parts) {
2668                let part = part_bytes(&bytes, *span)?;
2669                if checksum(part) != span.hash {
2670                    return Err(invalid("column page checksum differs while building frequencies"));
2671                }
2672                let rows = rows as usize;
2673                let vector = decode(ty, rows, part, None)?;
2674                // Every signed layout a numeric column decodes to, which is every column of `hits`,
2675                // comes out as one run of `i64` and is walked as a slice. The row path below is for
2676                // the unsigned types and anything else that cannot be handed over that way.
2677                if signed && vector.signed_block(&mut block) && block.len() == rows {
2678                    if vector.none_null() {
2679                        for (row, &value) in block.iter().enumerate() {
2680                            visit(start.saturating_add(row as u64), Some(value as u64));
2681                        }
2682                    } else {
2683                        for (row, &value) in block.iter().enumerate() {
2684                            let bits = (!vector.is_null_at(row)).then_some(value as u64);
2685                            visit(start.saturating_add(row as u64), bits);
2686                        }
2687                    }
2688                    start = start.saturating_add(rows as u64);
2689                    continue;
2690                }
2691                // row at a time: frequency construction visits decoded values to update bounded candidates.
2692                for row in 0..rows {
2693                    let bits = if vector.is_null_at(row) {
2694                        None
2695                    } else {
2696                        // An unsigned column has no signed reading, and the documented fallback is
2697                        // the value itself. Every width the format stores fits in sixty four bits,
2698                        // so nothing is lost on the way through.
2699                        let widened = match vector.signed_at(row) {
2700                            Some(value) => Some(value as u64),
2701                            None => match vector.value_at(row) {
2702                                Value::UTinyInt(value) => Some(u64::from(value)),
2703                                Value::USmallInt(value) => Some(u64::from(value)),
2704                                Value::UInteger(value) => Some(u64::from(value)),
2705                                Value::UBigInt(value) => Some(value),
2706                                _ => None,
2707                            },
2708                        };
2709                        Some(widened.ok_or_else(|| {
2710                            invalid("numeric frequency page did not contain an integer value")
2711                        })?)
2712                    };
2713                    visit(start.saturating_add(row as u64), bits);
2714                }
2715                start = start.saturating_add(rows as u64);
2716            }
2717        }
2718        Ok(())
2719    }
2720
2721    /// Builds independent numeric synopses concurrently after all column pages are committed.
2722    ///
2723    /// The columns go through a queue rather than being cut into equal runs, because they are not
2724    /// equally expensive and they are not shuffled. A `BIGINT` column carries eight times the bytes
2725    /// of a `TINYINT` through the decode, and a run of them sits together in a schema the way it
2726    /// sits together in `hits`, so a worker that was handed the wrong six columns finishes long
2727    /// after one that was handed the right six and the whole phase waits for it.
2728    fn numeric_frequencies(&self) -> Result<Vec<(Option<FrequencySummary>, Option<u64>)>> {
2729        let mut columns = self
2730            .table
2731            .fields
2732            .iter()
2733            .enumerate()
2734            .filter_map(|(column, field)| {
2735                matches!(
2736                    field.ty,
2737                    LogicalType::TinyInt
2738                        | LogicalType::SmallInt
2739                        | LogicalType::Integer
2740                        | LogicalType::BigInt
2741                        | LogicalType::UTinyInt
2742                        | LogicalType::USmallInt
2743                        | LogicalType::UInteger
2744                        | LogicalType::UBigInt
2745                        | LogicalType::Date
2746                        | LogicalType::Timestamp
2747                )
2748                .then_some(column)
2749            })
2750            .collect::<Vec<_>>();
2751        let workers = std::thread::available_parallelism()
2752            .map_or(1, usize::from)
2753            .min(MAX_FREQUENCY_WORKERS)
2754            .min(columns.len());
2755        let profile = self.profile.as_deref();
2756        if workers <= 1 {
2757            let _timing = profile.map(|profile| profile.span(Stage::Publish));
2758            let mut frequencies = vec![(None, None); self.table.fields.len()];
2759            for column in columns {
2760                frequencies[column] = self.numeric_frequency(column)?;
2761            }
2762            return Ok(frequencies);
2763        }
2764        // Popped from the back, so the expensive columns are the ones taken first and the cheap ones
2765        // are what is left to fill in behind them.
2766        columns.sort_by_key(|&column| weight(&self.table.fields[column].ty));
2767        let queue = Mutex::new(columns);
2768        let pieces = std::thread::scope(|scope| {
2769            (0..workers)
2770                .map(|_| {
2771                    scope.spawn(|| {
2772                        let _timing = profile.map(|profile| profile.span(Stage::Publish));
2773                        let mut mine = Vec::new();
2774                        loop {
2775                            let taken = queue
2776                                .lock()
2777                                .map_err(|_| Error::internal("a native frequency worker panicked"))?
2778                                .pop();
2779                            let Some(column) = taken else { break };
2780                            mine.push((column, self.numeric_frequency(column)?));
2781                        }
2782                        Ok(mine)
2783                    })
2784                })
2785                .collect::<Vec<_>>()
2786                .into_iter()
2787                .map(|handle| {
2788                    handle
2789                        .join()
2790                        .map_err(|_| Error::internal("a native frequency worker panicked"))?
2791                })
2792                .collect::<Result<Vec<_>>>()
2793        })?;
2794        let mut frequencies = vec![(None, None); self.table.fields.len()];
2795        for piece in pieces {
2796            for (column, summary) in piece {
2797                frequencies[column] = summary;
2798            }
2799        }
2800        Ok(frequencies)
2801    }
2802
2803    /// Reads one stable dictionary code column only at sorted table-wide row ordinals.
2804    #[allow(dead_code)]
2805    fn stable_codes_at(&self, column: usize, ordinals: &[u64]) -> Result<Option<Vec<Option<u32>>>> {
2806        if self.dictionaries.get(column).and_then(Option::as_ref).is_none() {
2807            return Ok(None);
2808        }
2809        if ordinals.windows(2).any(|pair| pair[0] >= pair[1]) {
2810            return Err(invalid("frequency ordinals are not sorted and unique"));
2811        }
2812        let mut out = Vec::with_capacity(ordinals.len());
2813        let mut wanted = 0;
2814        let mut stripe_start = 0_u64;
2815        for stripe in &self.table.stripes {
2816            let stripe_end = stripe_start.saturating_add(stripe.rows as u64);
2817            if wanted == ordinals.len() || ordinals[wanted] >= stripe_end {
2818                stripe_start = stripe_end;
2819                continue;
2820            }
2821            let spans = read_index(&self.file, stripe, column)?;
2822            let page = stripe.pages[column];
2823            let mut bytes = vec![0; page.length as usize];
2824            read_at(&self.file, page.offset, &mut bytes)?;
2825            let mut part_start = stripe_start;
2826            for (span, &rows) in spans.iter().zip(&stripe.parts) {
2827                let part_end = part_start.saturating_add(u64::from(rows));
2828                if wanted < ordinals.len() && ordinals[wanted] < part_end {
2829                    let part = part_bytes(&bytes, *span)?;
2830                    if checksum(part) != span.hash {
2831                        return Err(invalid(
2832                            "column page checksum differs while building pair frequencies",
2833                        ));
2834                    }
2835                    let upto = ordinals.partition_point(|&ordinal| ordinal < part_end);
2836                    let positions = ordinals[wanted..upto]
2837                        .iter()
2838                        .map(|&ordinal| {
2839                            usize::try_from(ordinal.saturating_sub(part_start))
2840                                .map_err(|_| invalid("frequency row offset does not fit in memory"))
2841                        })
2842                        .collect::<Result<Vec<_>>>()?;
2843                    if !decode_selected_stable_codes(rows as usize, part, &positions, &mut out)? {
2844                        return Ok(None);
2845                    }
2846                    wanted = upto;
2847                }
2848                part_start = part_end;
2849            }
2850            stripe_start = stripe_end;
2851        }
2852        if wanted != ordinals.len() {
2853            return Err(invalid("frequency ordinal is outside the table"));
2854        }
2855        Ok(Some(out))
2856    }
2857
2858    /// Derives bounded two-key leaders from numeric anchor ordinals and stable string codes.
2859    #[allow(dead_code)]
2860    fn pair_frequencies(
2861        &self,
2862        frequencies: &[Option<Frequencies>],
2863    ) -> Result<Vec<PairFrequencySummary>> {
2864        let anchors = frequencies
2865            .iter()
2866            .enumerate()
2867            .filter_map(|(column, summary)| {
2868                // A writer holds every synopsis it counted, so there is nothing stored to skip.
2869                match summary {
2870                    Some(Frequencies::Held(summary)) => Some(summary),
2871                    _ => None,
2872                }
2873                .filter(|summary| {
2874                    !summary.ordinals.is_empty()
2875                        && summary.ordinal_entries.len() == summary.ordinals.len()
2876                })
2877                .cloned()
2878                .map(|summary| (column, summary))
2879            })
2880            .collect::<Vec<_>>();
2881        let strings = self
2882            .dictionaries
2883            .iter()
2884            .enumerate()
2885            .filter_map(|(column, dictionary)| dictionary.as_ref().map(|_| column))
2886            .collect::<Vec<_>>();
2887        let mut summaries = Vec::new();
2888        for (first, anchors) in anchors {
2889            for &second in &strings {
2890                if summaries.len() == MAX_PAIR_FREQUENCIES {
2891                    return Ok(summaries);
2892                }
2893                let Some(codes) = self.stable_codes_at(second, &anchors.ordinals)? else {
2894                    continue;
2895                };
2896                if codes.len() != anchors.ordinal_entries.len() {
2897                    return Err(invalid("pair frequency columns have different lengths"));
2898                }
2899                let mut counts = HashMap::<(u16, Option<u32>), u64>::new();
2900                for (&anchor, code) in anchors.ordinal_entries.iter().zip(codes) {
2901                    *counts.entry((anchor, code)).or_default() += 1;
2902                }
2903                let mut entries = counts
2904                    .into_iter()
2905                    .map(|((first_entry, second), count)| PairFrequencyEntry {
2906                        first_entry,
2907                        second,
2908                        count,
2909                    })
2910                    .collect::<Vec<_>>();
2911                entries.sort_unstable_by(|left, right| {
2912                    right
2913                        .count
2914                        .cmp(&left.count)
2915                        .then_with(|| left.first_entry.cmp(&right.first_entry))
2916                        .then_with(|| left.second.cmp(&right.second))
2917                });
2918                let pair_omitted = entries.get(FREQUENCY_ENTRIES).map_or(0, |entry| entry.count);
2919                entries.truncate(FREQUENCY_ENTRIES);
2920                summaries.push(PairFrequencySummary {
2921                    first: u16::try_from(first)
2922                        .map_err(|_| invalid("pair frequency column index overflows"))?,
2923                    second: u16::try_from(second)
2924                        .map_err(|_| invalid("pair frequency column index overflows"))?,
2925                    entries,
2926                    omitted_max: anchors.omitted_max.max(pair_omitted),
2927                });
2928            }
2929        }
2930        Ok(summaries)
2931    }
2932
2933    /// Writes the directory of the table this writer is on and says where it went.
2934    ///
2935    /// Everything [`Writer::finish`] used to do except the two writes that publish. Pulling it out
2936    /// is what lets a second table follow a first: the bytes of a closed table are complete and
2937    /// addressable while nothing yet points at them, and the pointer is the last write of the
2938    /// commit.
2939    ///
2940    /// # Errors
2941    ///
2942    /// If directory encoding or writing fails.
2943    fn close(&mut self) -> Result<Entry> {
2944        self.reclaim()?;
2945        self.flush_pending()?;
2946        // The rest of a table is its statistics, its dictionaries and its directory. The dictionary
2947        // work is charged as its own stage, because ranking a global dictionary can be most of what
2948        // this costs, and the rest as publish.
2949        let profile = self.profile.clone();
2950        let timing = profile.as_deref().map(|profile| profile.span(Stage::Publish));
2951        let before = self.at;
2952        let mut stripes = std::mem::take(&mut self.order)
2953            .into_iter()
2954            .zip(std::mem::take(&mut self.table.stripes))
2955            .collect::<Vec<_>>();
2956        stripes.sort_by_key(|(order, _)| order.0);
2957        let mut previous: Option<(u64, u64)> = None;
2958        for ((first, last), _) in &stripes {
2959            if previous.is_some_and(|previous| previous >= *first) {
2960                return Err(invalid("chunks did not arrive in source order"));
2961            }
2962            previous = Some(*last);
2963        }
2964        self.table.stripes = stripes.into_iter().map(|(_, stripe)| stripe).collect();
2965        drop(timing);
2966        let timing = profile.as_deref().map(|profile| profile.span(Stage::Dictionary));
2967        let placing = self.at;
2968        finish_dictionaries(&mut self.dictionaries)?;
2969        self.place_blocks()?;
2970        // The numeric frequencies and the global dictionaries read what is already written and
2971        // write nothing, so they run at the same time. Each was most of a second on `hits` with the
2972        // other waiting for it, and neither keeps every core busy on its own: each is as long as
2973        // its longest column. Both charge themselves, one span to each thread that works, because
2974        // they run on threads of their own and a span on this one would see their wall time and
2975        // none of their CPU.
2976        let this = &*self;
2977        let (numeric, closed) = std::thread::scope(|scope| {
2978            let numeric = scope.spawn(|| {
2979                let (frequencies, distincts): (Vec<Option<FrequencySummary>>, Vec<_>) =
2980                    this.numeric_frequencies()?.into_iter().unzip();
2981                let frequencies = frequencies
2982                    .into_iter()
2983                    .map(|held| held.map(Frequencies::Held))
2984                    .collect::<Vec<_>>();
2985                // Pair leaders are query results, not reusable column statistics.
2986                let pairs = Vec::new();
2987                Ok::<_, Error>((frequencies, distincts, pairs))
2988            });
2989            let closed = this.close_dictionaries();
2990            let numeric =
2991                numeric.join().map_err(|_| Error::internal("the native frequency thread panicked"));
2992            (numeric, closed)
2993        });
2994        let (frequencies, distincts, pairs) = numeric??;
2995        let closed = closed?;
2996        self.table.frequencies = frequencies;
2997        self.table.distincts = distincts;
2998        self.table.pair_frequencies = pairs;
2999        self.dictionaries = Vec::new();
3000        self.table.dictionary_payloads = vec![0; self.table.fields.len()];
3001        self.table.frequency_texts = vec![Vec::new(); self.table.fields.len()];
3002        self.table.host_groups = None;
3003        for (index, closed) in closed.into_iter().enumerate() {
3004            let Some(closed) = closed else { continue };
3005            let ClosedDictionary { distinct, frequencies, texts, hosts, encoded, payload } = closed;
3006            self.table.distincts[index] = Some(distinct);
3007            self.table.frequencies[index] = Some(Frequencies::Held(frequencies));
3008            self.table.frequency_texts[index] = texts;
3009            if hosts.is_some() {
3010                self.table.host_groups = hosts;
3011            }
3012            let offset = self.at;
3013            self.put(&encoded.index)?;
3014            self.put(&encoded.ranks)?;
3015            self.put(&encoded.grams)?;
3016            self.table.dictionary_payloads[index] = payload;
3017            let length = encoded
3018                .index
3019                .len()
3020                .checked_add(encoded.ranks.len())
3021                .and_then(|len| len.checked_add(encoded.grams.len()))
3022                .ok_or_else(|| invalid("dictionary page length overflow"))?;
3023            self.table.dictionaries[index] = Some(Page {
3024                offset,
3025                length: u32::try_from(length)
3026                    .map_err(|_| invalid("dictionary page length overflow"))?,
3027                hash: checksum(&encoded.index),
3028            });
3029        }
3030        drop(timing);
3031        let timing = profile.as_deref().map(|profile| profile.span(Stage::Publish));
3032        let placed = self.at - placing;
3033        self.write_stats()?;
3034        let directory = encode_directory(&self.table)?;
3035        if directory.len() > MAX_DIRECTORY {
3036            return Err(invalid("directory exceeds the configured bound"));
3037        }
3038        let offset = self.at;
3039        self.put(&directory)?;
3040        drop(timing);
3041        if let Some(profile) = &profile {
3042            profile.moved(Stage::Dictionary, 0, placed, 0);
3043            profile.moved(Stage::Publish, 0, self.at - before - placed, 0);
3044        }
3045        Ok(Entry {
3046            name: self.table.name.clone(),
3047            fields: self.table.fields.clone(),
3048            rows: self.table.rows,
3049            nonzero: table_nonzero_counts(&self.table),
3050            aggregates: table_aggregate_sums(&self.table),
3051            distincts: self.table.distincts.clone(),
3052            extremes: table_integer_extremes(&self.table),
3053            frequencies: table_complete_numeric_frequencies(&self.table),
3054            directory: Page {
3055                offset,
3056                length: u32::try_from(directory.len())
3057                    .map_err(|_| invalid("directory length overflow"))?,
3058                hash: checksum(&directory),
3059            },
3060        })
3061    }
3062
3063    /// Every global dictionary's page and statistics, by column, as many columns at a time as
3064    /// [`CLOSE_DICTIONARY_BYTES`] allows.
3065    ///
3066    /// The largest column that fits is the one taken next, so the long ones start first and the
3067    /// short ones fill in behind them. A column that does not fit waits for one that is closing to
3068    /// finish, unless nothing is closing, in which case it goes alone.
3069    fn close_dictionaries(&self) -> Result<Vec<Option<ClosedDictionary>>> {
3070        let mut jobs = self
3071            .dictionaries
3072            .iter()
3073            .enumerate()
3074            .filter_map(|(index, dictionary)| {
3075                dictionary
3076                    .as_ref()
3077                    .map(|dictionary| (index, dictionary, dictionary.closing_bytes()))
3078            })
3079            .collect::<Vec<_>>();
3080        jobs.sort_by_key(|&(_, _, bytes)| bytes);
3081        let mut closed = (0..self.dictionaries.len()).map(|_| None).collect::<Vec<_>>();
3082        let workers = close_workers().min(jobs.len());
3083        if workers <= 1 {
3084            for (index, dictionary, _) in jobs {
3085                closed[index] = Some(self.close_dictionary(index, dictionary)?);
3086            }
3087            return Ok(closed);
3088        }
3089        // The columns not taken yet, smallest first, and the bytes the ones closing now hold.
3090        let state = Mutex::new((jobs, 0_usize));
3091        let finished = Condvar::new();
3092        let profile = self.profile.as_deref();
3093        let pieces = std::thread::scope(|scope| {
3094            (0..workers)
3095                .map(|_| {
3096                    scope.spawn(|| {
3097                        let _timing = profile.map(|profile| profile.span(Stage::Dictionary));
3098                        let mut mine = Vec::new();
3099                        loop {
3100                            let mut held = state.lock().map_err(|_| {
3101                                Error::internal("a native dictionary worker panicked")
3102                            })?;
3103                            let (index, dictionary, bytes) = loop {
3104                                let (jobs, busy) = &mut *held;
3105                                if jobs.is_empty() {
3106                                    return Ok(mine);
3107                                }
3108                                let fits = jobs.iter().rposition(|&(_, _, bytes)| {
3109                                    *busy == 0
3110                                        || busy.saturating_add(bytes) <= CLOSE_DICTIONARY_BYTES
3111                                });
3112                                if let Some(at) = fits {
3113                                    let job = jobs.remove(at);
3114                                    *busy += job.2;
3115                                    break job;
3116                                }
3117                                held = finished.wait(held).map_err(|_| {
3118                                    Error::internal("a native dictionary worker panicked")
3119                                })?;
3120                            };
3121                            drop(held);
3122                            // Given back on the way out whether the close worked, failed or
3123                            // panicked, so that a worker waiting for room is never left waiting.
3124                            let _room = Room { state: &state, finished: &finished, bytes };
3125                            mine.push((index, self.close_dictionary(index, dictionary)?));
3126                        }
3127                    })
3128                })
3129                .collect::<Vec<_>>()
3130                .into_iter()
3131                .map(|handle| {
3132                    handle
3133                        .join()
3134                        .map_err(|_| Error::internal("a native dictionary worker panicked"))?
3135                })
3136                .collect::<Result<Vec<_>>>()
3137        })?;
3138        for (index, one) in pieces.into_iter().flatten() {
3139            closed[index] = Some(one);
3140        }
3141        Ok(closed)
3142    }
3143
3144    /// One global dictionary's page and statistics, built from what is already in the file.
3145    ///
3146    /// Nothing is written here, so that [`Self::close`] can run this beside the numeric frequencies
3147    /// and put the pages down afterwards in column order, which is where they always went. The
3148    /// column's values are decoded in here and dropped before it returns, and
3149    /// [`Self::close_dictionaries`] decides how many columns are in here at once.
3150    fn close_dictionary(
3151        &self,
3152        _index: usize,
3153        dictionary: &GlobalDictionary,
3154    ) -> Result<ClosedDictionary> {
3155        let (order, flat, bases) = dictionary.ranked_with_values(Some(&self.file))?;
3156        // A code nothing counted is a code no non-null row of this column holds, which is the
3157        // empty string a null was written as and nothing else, because a code is only ever made by
3158        // a row asking for one.
3159        let distinct = dictionary.counts.iter().filter(|count| **count != 0).count() as u64;
3160        let (frequencies, texts) = code_frequency(dictionary, &flat, &bases)?;
3161        // Deriving a fixed SQL host expression at load time materializes its answer.
3162        let hosts = None;
3163        drop(flat);
3164        drop(bases);
3165        let encoded = encode_global_dictionary(dictionary, &order, &dictionary.placed, true)?;
3166        let payload = dictionary
3167            .placed
3168            .iter()
3169            .try_fold(0_u64, |sum, place| sum.checked_add(place.length))
3170            .ok_or_else(|| invalid("global dictionary payload overflow"))?;
3171        Ok(ClosedDictionary { distinct, frequencies, texts, hosts, encoded, payload })
3172    }
3173
3174    /// Writes the statistics sections for the table being closed, as far as the budget reaches.
3175    ///
3176    /// Called from [`Self::close`] after the last stripe and after the dictionaries, which is the
3177    /// first moment the table's column bytes are final and the last moment before the directory is
3178    /// encoded. Both halves matter: the budget is a share of the column bytes, and a section that
3179    /// went in after the directory would be a section the directory does not name.
3180    ///
3181    /// Nothing here can fail the write. A column whose gather came back blind gets no sections, a
3182    /// column the budget could not reach gets none, and section 3.1 says both of those plan the way
3183    /// they planned before statistics existed. The two errors that are returned are an encode
3184    /// failure and a section count past the bound, and neither is a thing a column can cause.
3185    fn write_stats(&mut self) -> Result<()> {
3186        let gathers = std::mem::take(&mut self.gathers);
3187        let rows = self.table.rows as u64;
3188        let mut payloads = Vec::new();
3189        for (column, gather) in gathers.into_iter().enumerate() {
3190            let Some(gather) = gather else { continue };
3191            // A gather that saw a different number of rows than the table committed is a gather
3192            // that missed some, and a distinct count over some of a column is the one error an
3193            // estimator cannot see coming. This has no way of happening today, since a table is
3194            // written once and every chunk goes through `flush_pending`, and that is exactly why it
3195            // is worth a line: it stays true only while that stays true.
3196            if gather.rows() != rows {
3197                continue;
3198            }
3199            let Some(stats) = gather.finish() else { continue };
3200            let mut summary = Vec::new();
3201            stats.summary.encode(&mut summary)?;
3202            let mut sketches = Vec::new();
3203            stats.sketches.encode(&mut sketches)?;
3204            payloads.push((column, summary, sketches));
3205        }
3206        if payloads.is_empty() {
3207            return Ok(());
3208        }
3209        let costs = payloads
3210            .iter()
3211            .map(|(_, summary, sketches)| summary.len() + sketches.len())
3212            .collect::<Vec<_>>();
3213        let allowance = stats::allowance(stats::column_bytes(&self.table), stats::BUDGET_SHARE);
3214        // Nothing is spent yet. A table this writer is closing is one it wrote from nothing, so the
3215        // only statistics sections it can have are the ones about to go in.
3216        let keep = stats::within(&costs, allowance, 0);
3217        for ((column, summary, sketches), _) in
3218            payloads.iter().zip(&keep).filter(|&(_, &keep)| keep)
3219        {
3220            let id = u64::try_from(*column).map_err(|_| invalid("column index overflow"))?;
3221            for (kind, bytes, header_bytes) in [
3222                // A summary is a header the whole way down: there is nothing behind it a reader
3223                // could decide not to read.
3224                (*section::SUMMARY, summary, summary.len() as u32),
3225                (*section::SKETCHES, sketches, rudb_stats::sketches::HEADER_BYTES),
3226            ] {
3227                let written = write_section(
3228                    &self.file,
3229                    &mut self.at,
3230                    &section::Attachment { kind, id, flags: 0, header_bytes, bytes },
3231                    self.generation,
3232                )?;
3233                self.table.sections.push(written);
3234            }
3235        }
3236        if self.table.sections.len() > MAX_SECTIONS {
3237            return Err(invalid("the table would name more sections than the bound allows"));
3238        }
3239        Ok(())
3240    }
3241
3242    /// Commits every table this writer has written and syncs the file before publishing its header
3243    /// slot.
3244    ///
3245    /// The table handed back is the one the writer was on, which is the last of them. Callers that
3246    /// wrote several already know the others, since they named them.
3247    ///
3248    /// # Errors
3249    ///
3250    /// If directory encoding, writing, or syncing fails.
3251    pub fn finish(mut self) -> Result<Table> {
3252        let entry = self.close()?;
3253        let profile = self.profile.take();
3254        let _timing = profile.as_deref().map(|profile| profile.span(Stage::Publish));
3255        let mut tables = std::mem::take(&mut self.closed);
3256        tables.push(entry);
3257        let catalog = encode_catalog(&tables, &self.views)?;
3258        if catalog.len() > MAX_DIRECTORY {
3259            return Err(invalid("catalog exceeds the configured bound"));
3260        }
3261        let offset = self.at;
3262        self.put(&catalog)?;
3263        if let Some(profile) = &profile {
3264            profile.moved(Stage::Publish, 0, catalog.len() as u64, 0);
3265        }
3266        // Every page and every table directory is on the disk before anything points at them. The
3267        // slot write below is what makes this generation the one a reader picks, so the order of
3268        // these two syncs is the whole of the commit.
3269        synced(&self.file, profile.as_deref())?;
3270        let slot = Slot {
3271            offset,
3272            length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
3273            generation: self.generation,
3274            hash: checksum(&catalog),
3275        };
3276        // The one write that is not an append, and the last one. It goes back over the slot in the
3277        // header, so it names its offset rather than going through `put`, and `at` does not move.
3278        // Which of the two slots it is alternates with the generation, so the one naming the
3279        // generation before this is still intact and still valid until this write lands.
3280        write_at(&self.file, slot_offset(self.generation), &slot.bytes())?;
3281        synced(&self.file, profile.as_deref())?;
3282        Ok(self.table)
3283    }
3284
3285    /// Commits a generation that changes the views and leaves every table exactly where it is.
3286    ///
3287    /// There was no way to do this before views existed, because everything that could change the
3288    /// catalog also wrote a table, so the only way to say something new about a file was to go
3289    /// through a table. A view is the first thing that can change on its own. Without this, adding
3290    /// a view to a database with eight tables in it would rewrite all eight, since the append path
3291    /// needs a table to append and the fallback is the whole file.
3292    ///
3293    /// It is the same commit as [`Writer::finish`] with nothing appended before it. The table
3294    /// entries are carried forward by directory pointer the way an append carries them, the new
3295    /// catalog goes on the end, and the slot write at the end is what publishes it.
3296    ///
3297    /// # Errors
3298    ///
3299    /// If the file has no valid committed directory, is not this build's format, or cannot be
3300    /// written.
3301    pub fn restate(path: impl AsRef<Path>, views: &[ViewEntry]) -> Result<()> {
3302        let path = path.as_ref();
3303        let (_, size, slot, bytes, _) = slot_bytes(path)?;
3304        let (closed, _) = decode_catalog(&bytes, size)?;
3305        let generation = slot
3306            .generation
3307            .checked_add(1)
3308            .ok_or_else(|| invalid("native file generation overflow"))?;
3309        let catalog = encode_catalog(&closed, views)?;
3310        if catalog.len() > MAX_DIRECTORY {
3311            return Err(invalid("catalog exceeds the configured bound"));
3312        }
3313        let file = OpenOptions::new().write(true).read(true).open(path).map_err(io)?;
3314        write_at(&file, size, &catalog)?;
3315        file.sync_all().map_err(io)?;
3316        let slot = Slot {
3317            offset: size,
3318            length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
3319            generation,
3320            hash: checksum(&catalog),
3321        };
3322        write_at(&file, slot_offset(generation), &slot.bytes())?;
3323        file.sync_all().map_err(io)?;
3324        Ok(())
3325    }
3326
3327    /// Adds exact count, sum, distinct, bound, and bounded frequency certificates to an older file without
3328    /// rewriting table pages. The old slot remains readable until the new catalog is synced.
3329    pub fn certify_summaries(path: impl AsRef<Path>) -> Result<()> {
3330        let path = path.as_ref();
3331        let (_, size, slot, bytes, _) = slot_bytes(path)?;
3332        let (mut entries, views) = decode_catalog(&bytes, size)?;
3333        let native = Catalog::open(path)?;
3334        for entry in &mut entries {
3335            let reader = native.table(&entry.name)?;
3336            entry.nonzero = reader_nonzero_counts(&reader)?;
3337            entry.aggregates = reader_aggregate_sums(&reader)?;
3338            entry.distincts = (0..entry.fields.len())
3339                .map(|column| reader.distinct_values(column))
3340                .collect::<Result<Vec<_>>>()?;
3341            entry.extremes = reader_integer_extremes(&reader)?;
3342            entry.frequencies = reader_complete_numeric_frequencies(&reader)?;
3343        }
3344        let generation = slot
3345            .generation
3346            .checked_add(1)
3347            .ok_or_else(|| invalid("native file generation overflow"))?;
3348        let catalog = encode_catalog(&entries, &views)?;
3349        if catalog.len() > MAX_DIRECTORY {
3350            return Err(invalid("catalog exceeds the configured bound"));
3351        }
3352        let file = OpenOptions::new().write(true).read(true).open(path).map_err(io)?;
3353        write_at(&file, size, &catalog)?;
3354        file.sync_all().map_err(io)?;
3355        let slot = Slot {
3356            offset: size,
3357            length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
3358            generation,
3359            hash: checksum(&catalog),
3360        };
3361        write_at(&file, slot_offset(generation), &slot.bytes())?;
3362        file.sync_all().map_err(io)?;
3363        Ok(())
3364    }
3365
3366    /// The earlier name for [`Self::certify_summaries`].
3367    pub fn certify_counts(path: impl AsRef<Path>) -> Result<()> {
3368        Self::certify_summaries(path)
3369    }
3370}
3371
3372/// Appends one run of bytes at `at` and moves it past them, answering where they went.
3373///
3374/// The append half of [`attach`], which cannot use [`Writer::put`] because it is not writing a
3375/// table. Every byte a section costs goes through here, so the offsets in an extent table come
3376/// from one place.
3377fn append(file: &File, at: &mut u64, bytes: &[u8]) -> Result<u64> {
3378    let offset = *at;
3379    write_at(file, offset, bytes)?;
3380    *at =
3381        at.checked_add(bytes.len() as u64).ok_or_else(|| invalid("native file length overflow"))?;
3382    Ok(offset)
3383}
3384
3385/// Writes one attachment's payload as extents and returns the entry that names it.
3386///
3387/// The split is by bytes, and the `first` of each extent is therefore a byte count. A section kind
3388/// whose extents should break on a row boundary instead will want to hand its extents over already
3389/// split; nothing needs that yet, and guessing at the shape of it now would be guessing.
3390fn write_section(
3391    file: &File,
3392    at: &mut u64,
3393    one: &section::Attachment<'_>,
3394    generation: u64,
3395) -> Result<Section> {
3396    // A payload of nothing is the exception, and it is not a special case so much as a different
3397    // reading of the same field: an entry with no bytes has no header to be longer than them, and
3398    // `header_bytes` is what the structure would have cost. See `Section::refused`.
3399    if !one.bytes.is_empty() && one.header_bytes as usize > one.bytes.len() {
3400        return Err(invalid("a section's header is longer than its payload"));
3401    }
3402    let mut extents = Vec::new();
3403    let mut first = 0_u64;
3404    for chunk in one.bytes.chunks(section::MAX_EXTENT as usize) {
3405        let offset = append(file, at, chunk)?;
3406        extents.push(section::Extent {
3407            offset,
3408            length: u32::try_from(chunk.len()).map_err(|_| invalid("extent length overflow"))?,
3409            hash: checksum(chunk),
3410            first,
3411        });
3412        first += chunk.len() as u64;
3413    }
3414    let mut table = Vec::with_capacity(extents.len() * section::EXTENT_BYTES);
3415    section::encode_extents(&extents, &mut table)?;
3416    // A payload of nothing is a section of no extents and no extent table, and its `extent_page`
3417    // is zero rather than the end of the file. Section 3.7 wants that entry to exist: it is how a
3418    // relationship that did not fit the budget is recorded as not built rather than forgotten.
3419    let extent_page = if table.is_empty() { 0 } else { append(file, at, &table)? };
3420    Ok(Section {
3421        kind: one.kind,
3422        id: one.id,
3423        generation,
3424        extents: u32::try_from(extents.len()).map_err(|_| invalid("too many extents"))?,
3425        extent_page,
3426        extent_bytes: u32::try_from(table.len()).map_err(|_| invalid("extent table overflow"))?,
3427        hash: checksum(&table),
3428        flags: one.flags,
3429        header_bytes: one.header_bytes,
3430    })
3431}
3432
3433/// Attaches graph sections to a table already committed in a file, without rewriting a page.
3434///
3435/// This is the second pass spec/graph/03-the-file-format.md section 3.8 asks for. A key map has to
3436/// exist before the link that uses it can be built, and it is built by reading the key column back,
3437/// so the structures of a table cannot be written during the load that wrote the table. They are
3438/// written afterwards, by this, and the file in between the two is a correct file that answers
3439/// every query more slowly.
3440///
3441/// Nothing is overwritten. The payloads, the extent tables, the new directory for this table and
3442/// the new catalog all go on the end of the file past the committed generation, and the last write
3443/// is the header slot, exactly as [`Writer::finish`] does it. So a crash anywhere in here leaves
3444/// the generation before it intact, and leaves unreferenced trailing bytes that the next commit
3445/// writes past.
3446///
3447/// An attachment replaces any section of the same kind and id, and every other section is carried
3448/// through untouched, including one whose kind this build does not know. The table's own generation
3449/// is carried through too, because attaching a section moves no row: see [`Table::generation`].
3450///
3451/// # Errors
3452///
3453/// If the file has no valid committed directory, is an older format than this build writes, holds
3454/// no table of that name, names a section whose payload cannot be written, or would end up naming
3455/// more sections than the format allows.
3456pub fn attach(
3457    path: impl AsRef<Path>,
3458    table: &str,
3459    attachments: &[section::Attachment<'_>],
3460) -> Result<Table> {
3461    let path = path.as_ref();
3462    let (_, size, slot, bytes, _) = slot_bytes(path)?;
3463    let (mut entries, views) = decode_catalog(&bytes, size)?;
3464    let at = entries
3465        .iter()
3466        .position(|entry| entry.name == table)
3467        .ok_or_else(|| invalid(&format!("the file holds no table called {table}")))?;
3468    let file = OpenOptions::new().write(true).read(true).open(path).map_err(io)?;
3469    let mut version = [0; 4];
3470    read_at(&file, 8, &mut version)?;
3471    let version = u32::from_le_bytes(version);
3472    // Readable is not the same as writable. A format 22 file has no section table, and giving its
3473    // directory one without moving the number in its header would leave a file that claims to be
3474    // format 22 and is not, which is worse than refusing. Rewriting it with this build is the
3475    // answer, and the format is at 0.3.x, so nobody has one of these that this project did not
3476    // just make.
3477    if version != FORMAT {
3478        return Err(invalid(&format!(
3479            "the file is format {version} and a graph section needs format {FORMAT}, so it has \
3480             to be written again"
3481        )));
3482    }
3483    let mut directory = vec![0; entries[at].directory.length as usize];
3484    read_at(&file, entries[at].directory.offset, &mut directory)?;
3485    if checksum(&directory) != entries[at].directory.hash {
3486        return Err(invalid(&format!("the directory of table {table} does not checksum")));
3487    }
3488    let mut held = decode_directory(&directory, size)?;
3489    let mut cursor = size;
3490    for one in attachments {
3491        let written = write_section(&file, &mut cursor, one, held.generation)?;
3492        held.sections.retain(|old| !(old.kind == one.kind && old.id == one.id));
3493        held.sections.push(written);
3494    }
3495    if held.sections.len() > MAX_SECTIONS {
3496        return Err(invalid("the table would name more sections than the bound allows"));
3497    }
3498    let encoded = encode_directory(&held)?;
3499    if encoded.len() > MAX_DIRECTORY {
3500        return Err(invalid("directory exceeds the configured bound"));
3501    }
3502    let offset = append(&file, &mut cursor, &encoded)?;
3503    entries[at].directory = Page {
3504        offset,
3505        length: u32::try_from(encoded.len()).map_err(|_| invalid("directory length overflow"))?,
3506        hash: checksum(&encoded),
3507    };
3508    // The views the file already had, written back unchanged. Attaching a section to a table says
3509    // nothing about a view and must not drop one.
3510    let catalog = encode_catalog(&entries, &views)?;
3511    if catalog.len() > MAX_DIRECTORY {
3512        return Err(invalid("catalog exceeds the configured bound"));
3513    }
3514    let offset = append(&file, &mut cursor, &catalog)?;
3515    file.sync_all().map_err(io)?;
3516    let generation =
3517        slot.generation.checked_add(1).ok_or_else(|| invalid("native file generation overflow"))?;
3518    let committed = Slot {
3519        offset,
3520        length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
3521        generation,
3522        hash: checksum(&catalog),
3523    };
3524    write_at(&file, slot_offset(generation), &committed.bytes())?;
3525    file.sync_all().map_err(io)?;
3526    Ok(held)
3527}
3528
3529/// One column's frequency synopsis as values with their row counts, shared by every clone of a
3530/// reader.
3531type Synopsis = Arc<Vec<(Value, u64)>>;
3532
3533/// Reads committed native column pages without holding the table in memory.
3534#[derive(Debug, Clone)]
3535pub struct Reader {
3536    file: Arc<File>,
3537    table: Arc<Table>,
3538    dictionaries: Arc<Vec<OnceLock<Arc<Vector>>>>,
3539    /// Held while a global dictionary is being opened, one per column.
3540    ///
3541    /// The [`OnceLock`] above says whether one has been opened, which is the question a reader that
3542    /// already has it needs answered and is free. It does not say whether one is being opened, and
3543    /// the difference matters because every worker of a scan wants the same dictionary at the same
3544    /// moment. Without this they all miss, all read the page, all verify it and all decode it, and
3545    /// all but one throw the answer away. ClickBench 38 reads the URL dictionary, which is 515,958
3546    /// entries, and was paying for it twice.
3547    loading: Arc<Vec<Mutex<()>>>,
3548    /// Each column's frequency synopsis as values, the first time anything asks for it. See
3549    /// [`Reader::decode_frequencies`].
3550    frequency_values: Arc<Vec<OnceLock<Synopsis>>>,
3551    /// Stored frequency sections are decoded once per open table. A small directory can hold the
3552    /// summary inline, but a larger one otherwise rereads and decodes the same section on every
3553    /// plan and every summary-backed aggregate.
3554    frequency_summaries: Arc<Vec<OnceLock<Arc<FrequencySummary>>>>,
3555    /// How many global dictionaries have been opened. A scan of a dictionary column should open its
3556    /// dictionary once however many workers it has, and the test that says so is the only thing
3557    /// keeping it that way.
3558    opened: Arc<AtomicUsize>,
3559    /// The membership sieves of one stripe of one column, by column and then by stripe, read the
3560    /// first time a probe asks about them. A query filters on one or two columns and never looks at
3561    /// the rest, so reading these at open would be the whole index for the sake of a fraction of it.
3562    sieves: Arc<Vec<Vec<SieveSlot>>>,
3563    /// The per part ranges of one stripe of one column, by column and then by stripe, read the
3564    /// first time something compares that column and kept after that.
3565    part_ranges: Arc<Vec<Vec<RangeSlot>>>,
3566    /// Which stripe and which part of it every part of the table is, by table wide part number.
3567    places: Arc<Vec<Place>>,
3568    cache: Arc<Shelf>,
3569    /// Where the pages above are counted against the database's budget. See [`PagePool`].
3570    pool: PagePool,
3571    /// How many whole stripe pages have been read, which is what the sharing above is judged on. A
3572    /// scan of a column should read each of its stripes once however many workers it has.
3573    pages: Arc<AtomicUsize>,
3574    /// How many index sections have been read. A scan of a column should read each of its stripes
3575    /// once here too, and the test that says so is the only thing keeping it that way.
3576    indexes: Arc<AtomicUsize>,
3577    /// The file's size when it was opened, for [`Reader::layout`].
3578    size: u64,
3579    /// The committed directory's size, for [`Reader::layout`].
3580    directory: u64,
3581    /// What opening the file cost, which is a number rather than a claim.
3582    opening: Opening,
3583}
3584
3585/// What [`Reader::open`] read before it returned.
3586///
3587/// `spec/stats/04-in-memory.md` section 4.2 says opening a table reads the header and the directory
3588/// and nothing else, and once that document's statistics are in the file the tempting change is to
3589/// load a column summary or two on the way past, because they are small and the next query will
3590/// want them. A hundred milliseconds of that is a hundred milliseconds nobody asked for, and an
3591/// embedded database is opened by processes that are about to run one trivial query.
3592///
3593/// So the claim gets a number. Both of these are fixed by the schema and the stripe count and are
3594/// independent of how many rows the file holds, and the test that says so is what stops the
3595/// tempting change from landing quietly.
3596#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
3597pub struct Opening {
3598    /// How many times the file was read. The header, then each directory slot that looked valid
3599    /// enough to check, so three at the most.
3600    pub reads: u32,
3601    /// How many bytes those reads asked for.
3602    pub bytes: u64,
3603}
3604
3605/// What a reader has read, while it was being opened and since.
3606#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
3607pub struct Reads {
3608    /// What opening cost, before any query had been planned.
3609    pub opening: Opening,
3610    /// Whole stripe pages read since.
3611    pub pages: usize,
3612    /// Index sections read since.
3613    pub indexes: usize,
3614    /// Global dictionaries opened since. One per dictionary column that a query touched, however
3615    /// many workers touched it, which is a claim only a test can keep true.
3616    pub dictionaries: usize,
3617}
3618
3619/// Where one table wide part number lands.
3620#[derive(Debug, Clone, Copy)]
3621struct Place {
3622    stripe: u32,
3623    part: u32,
3624    rows: u32,
3625}
3626
3627/// One part's bytes inside one column page.
3628#[derive(Debug, Clone, Copy)]
3629struct PartSpan {
3630    start: usize,
3631    length: usize,
3632    hash: u64,
3633}
3634
3635/// What a reader holds for one stripe of one column.
3636///
3637/// The index is small and is loaded whether the caller wants the whole page or one part of it. The
3638/// page is loaded only by a scan, because a sparse fetch that wants a thousand rows out of sixty
3639/// four thousand would be reading sixty four times what it uses.
3640#[derive(Debug, Clone)]
3641struct CachedColumn {
3642    stripe: usize,
3643    index: Arc<Vec<PartSpan>>,
3644    page: Option<Arc<Vec<u8>>>,
3645}
3646
3647/// One column's stripes a reader holds, and which of them somebody is reading right now.
3648///
3649/// The pages are one slot per stripe of the table rather than a list of the ones being kept, so
3650/// finding a page is an index and not a walk. That matters because the walk happened under the
3651/// lock, once per part per column, and a scan that gives a whole stripe to each of thirty two
3652/// workers keeps enough pages that walking them was the longest thing the lock was held for. The
3653/// slots cost a pointer per stripe per column, which on the ClickBench file is eight kilobytes
3654/// against the forty megabytes of pages they point at. `order` is which of them are filled, oldest
3655/// first, because that is the one thing the slots cannot say by themselves.
3656///
3657/// `loading` is what keeps a scan from reading the same page once per worker. It is a list and not
3658/// a set because it holds at most one stripe per worker on the column and is walked far less often
3659/// than a hash of it would be built.
3660///
3661/// `index` is every index this reader has ever read for the column, one slot per stripe, and it is
3662/// never evicted. An index is a few hundred bytes and a page is a quarter of a megabyte, so the two
3663/// do not belong under the same budget. Riding in the page cache meant a worker that came back to a
3664/// stripe after its page had been evicted read the index again with it, which on the full
3665/// ClickBench file was about thirteen hundred reads out of a hundred and fourteen thousand.
3666#[derive(Debug, Default)]
3667struct Cached {
3668    pages: Vec<Option<Resident>>,
3669    loading: Vec<usize>,
3670    index: Vec<Option<Arc<Vec<PartSpan>>>>,
3671}
3672
3673/// One page a reader holds, and whether anyone has read it since the pool last looked.
3674#[derive(Debug, Clone)]
3675struct Resident {
3676    page: Arc<Vec<u8>>,
3677    used: Arc<AtomicBool>,
3678}
3679
3680/// Every column's pages of one reader, with how many each column holds and the floor under that.
3681#[derive(Debug)]
3682struct Shelf {
3683    columns: Vec<Mutex<Cached>>,
3684    /// How many pages each column holds right now. Counted outside the column locks so that the
3685    /// pool can tell whether a column is at its floor without taking a lock it might be under.
3686    held: Vec<AtomicUsize>,
3687    /// How many stripes of one column are kept whatever the budget says. See
3688    /// [`CACHED_STRIPES_PER_COLUMN`] for what sets it and [`Reader::keep_stripes`] for who raises it.
3689    kept: AtomicUsize,
3690}
3691
3692/// The pages every reader of one database keeps, under one budget in bytes.
3693///
3694/// A reader lives as long as the database does, so the pages it holds are what the next query finds
3695/// already in memory. They used to be four stripes a column, oldest out first, which on TPC-H SF1
3696/// meant every query read every page of lineitem off the file again and paid the system call for
3697/// it. Keeping every page there costs 38 MB and took a third of the system time off the suite.
3698///
3699/// So the question is no longer how many stripes a column keeps but how many bytes the database
3700/// does, and one budget answers it for every reader at once. A table nobody queries gives its pages
3701/// up to one that is being queried, which a count per column cannot do.
3702///
3703/// Pages leave by the clock. Each has a bit a read sets, and when the pool is over budget it walks
3704/// from the oldest: a page with the bit set loses the bit and goes round again, and a page without
3705/// it goes. That keeps what is read over and over and lets a page one scan read once go first.
3706///
3707/// The old count is still a floor. A column never gives up a page while it holds four or fewer,
3708/// because a scan whose workers evict each other's pages reads a quarter of a megabyte for every
3709/// part it takes, and a budget of zero is the cache as it was before the pool existed.
3710#[derive(Debug, Clone, Default)]
3711pub struct PagePool {
3712    ring: Arc<Mutex<Ring>>,
3713    budget: Arc<AtomicUsize>,
3714}
3715
3716#[derive(Debug, Default)]
3717struct Ring {
3718    held: VecDeque<Held>,
3719    bytes: usize,
3720}
3721
3722/// One page in the pool, pointing back at the reader that holds it.
3723///
3724/// Weak, because a reader that has gone, which every reader does at a checkpoint, should take its
3725/// pages with it and not have them kept alive by the pool.
3726#[derive(Debug)]
3727struct Held {
3728    shelf: Weak<Shelf>,
3729    column: usize,
3730    stripe: usize,
3731    bytes: usize,
3732    used: Arc<AtomicBool>,
3733}
3734
3735impl PagePool {
3736    /// A pool that keeps up to `budget` bytes of pages beyond each column's floor.
3737    #[must_use]
3738    pub fn new(budget: usize) -> Self {
3739        let pool = Self::default();
3740        pool.budget.store(budget, Atomic::Relaxed);
3741        pool
3742    }
3743
3744    /// The bytes of pages the pool is counting now.
3745    ///
3746    /// # Panics
3747    ///
3748    /// If the pool's lock is poisoned, which takes a panic while it was held.
3749    #[must_use]
3750    pub fn bytes(&self) -> usize {
3751        self.ring.lock().map_or(0, |ring| ring.bytes)
3752    }
3753
3754    /// Counts a page a reader has just taken in, and lets pages go until the pool is back under its
3755    /// budget or it has looked at every page once.
3756    ///
3757    /// Called with no column lock held. The pages that go are chosen under the pool's lock and
3758    /// dropped under their column's lock afterwards, so no thread ever holds both.
3759    fn admit(&self, held: Held) {
3760        let budget = self.budget.load(Atomic::Relaxed);
3761        let mut gone = Vec::new();
3762        {
3763            let Ok(mut ring) = self.ring.lock() else { return };
3764            ring.bytes += held.bytes;
3765            ring.held.push_back(held);
3766            // One lap and no more. A page read since the last pass loses its bit on this one and
3767            // can only go on a later one, which is the second chance the clock is named for.
3768            let mut looked = 0;
3769            let limit = ring.held.len();
3770            while ring.bytes > budget && looked < limit {
3771                looked += 1;
3772                let Some(entry) = ring.held.pop_front() else { break };
3773                let Some(shelf) = entry.shelf.upgrade() else {
3774                    ring.bytes -= entry.bytes;
3775                    continue;
3776                };
3777                if entry.used.swap(false, Atomic::Relaxed) {
3778                    ring.held.push_back(entry);
3779                    continue;
3780                }
3781                let count = &shelf.held[entry.column];
3782                if count.load(Atomic::Relaxed) <= shelf.kept.load(Atomic::Relaxed).max(1) {
3783                    ring.held.push_back(entry);
3784                    continue;
3785                }
3786                count.fetch_sub(1, Atomic::Relaxed);
3787                ring.bytes -= entry.bytes;
3788                gone.push((shelf, entry));
3789            }
3790            // A reader that has gone leaves its entries behind, and with a budget nobody reaches
3791            // they would pile up one checkpoint after another. The front is where the oldest are.
3792            while ring.held.front().is_some_and(|entry| entry.shelf.strong_count() == 0) {
3793                if let Some(entry) = ring.held.pop_front() {
3794                    ring.bytes -= entry.bytes;
3795                }
3796            }
3797        }
3798        for (shelf, entry) in gone {
3799            let Ok(mut cached) = shelf.columns[entry.column].lock() else { continue };
3800            if let Some(slot) = cached.pages.get_mut(entry.stripe) {
3801                if slot.as_ref().is_some_and(|slot| Arc::ptr_eq(&slot.used, &entry.used)) {
3802                    *slot = None;
3803                }
3804            }
3805        }
3806    }
3807}
3808
3809/// Stripes of one column a reader keeps the bytes of, when nobody has asked for more.
3810///
3811/// This has to hold at least as many stripes as a column has workers in it at once, or the workers
3812/// evict each other's pages and read them again. Four is what a scan that hands parts out in order
3813/// needs, because then every worker is within a few parts of every other and at most a couple of
3814/// stripes are open at a time. A scan that hands a whole stripe to each worker has one stripe open
3815/// per worker for the length of that stripe, and it says so with [`Reader::keep_stripes`] rather
3816/// than paying for sixteen slots on every table that is read one part at a time.
3817///
3818/// It multiplies by the page size, which is a quarter of a megabyte for a four byte column, and by
3819/// the number of columns a query touches.
3820const CACHED_STRIPES_PER_COLUMN: usize = 4;
3821
3822/// The sieves of one stripe of one column, once somebody has asked for them.
3823type SieveSlot = OnceLock<Arc<Vec<Option<Sieve>>>>;
3824
3825type RangeSlot = OnceLock<Arc<Vec<Range>>>;
3826
3827#[derive(Debug)]
3828struct NativeText {
3829    file: Arc<File>,
3830    /// How many values the dictionary holds.
3831    values: usize,
3832    /// Where each value ends inside its payload block, packed at `offset_bits` in runs of
3833    /// [`TEXT_OFFSET_RUN`].
3834    ///
3835    /// Ends rather than starts, because then a block of 1,024 values is 1,024 numbers rather than
3836    /// 1,025: the start of a value is the end of the one before it, and the first value of a block
3837    /// starts at zero by construction. Relative to the block rather than to the payload, because a
3838    /// reader decodes a whole block and slices it, so an offset into the payload is a number it
3839    /// would have to subtract a base from anyway.
3840    offsets: Vec<u8>,
3841    /// Bits one offset is packed at, which is what the largest block of this column spans and is the
3842    /// same for every block of it.
3843    offset_bits: usize,
3844    /// The same ends unpacked, built once enough readers have asked for one at a time.
3845    ///
3846    /// Reading one offset out of the packed form costs about fifty instructions: a division to find
3847    /// the run, a bounds check to slice it, a shift to reach the bit the value starts at and a
3848    /// narrowing on the way out. That is the right price for a reader that wants a handful. It is
3849    /// the wrong price for `STRLEN` over a column, which asks for one per row and nothing else, and
3850    /// where a million of them was a third of ClickBench 28.
3851    ///
3852    /// With the ends unpacked every read is a load, and a vector of lengths is one loop over them.
3853    /// The table is built only once the reads say it will be used, which is what
3854    /// [`Self::ends_worth_unpacking`] decides and [`Self::ends_asked`] counts towards, because a
3855    /// table built for a reader that wanted three values is four bytes a value spent on nothing.
3856    value_ends: OnceLock<Option<Vec<u32>>>,
3857    /// The length of every value, worked out of [`Self::value_ends`] the first time a vector of
3858    /// lengths is asked for.
3859    ///
3860    /// A length out of the ends is two loads, a test for whether the value opens its block and a
3861    /// check that it does not end before it starts, which came to thirteen instructions a row on
3862    /// ClickBench 28. Out of this it is one load. The order is checked once for the whole table
3863    /// while it is built, and a column that fails it gets no table and goes on reading the ends,
3864    /// which is where the error is reported. Four bytes a value, and only for a column something
3865    /// has asked the length of a vector at a time.
3866    value_lens: OnceLock<Option<Vec<u32>>>,
3867    /// How many single offset reads have come in while the table is not built.
3868    ///
3869    /// Relaxed, and read only against a threshold, so two threads racing here means the table is
3870    /// built one read early or one read late. Counting stops the moment the table exists, because
3871    /// [`OnceLock::get`] settles it before this is touched.
3872    ends_asked: AtomicUsize,
3873    /// How many entries the sorted order has, which is the value count.
3874    ranks: usize,
3875    /// Where the sorted order starts in the file. It is read a block at a time and only when
3876    /// something searches it, so a query that never compares this column against a literal never
3877    /// touches it at all.
3878    rank_at: u64,
3879    /// Where each block of the sorted order ends, as a byte offset from `rank_at`. A block is packed
3880    /// at whatever width its own heads need, so unlike the entries it replaced its length is not
3881    /// arithmetic on the block number.
3882    rank_ends: Vec<u64>,
3883    rank_hashes: Vec<u64>,
3884    rank_blocks: Vec<OnceLock<Result<Vec<u8>>>>,
3885    /// Bits one code is packed at, which is what the value count needs and is the same for every
3886    /// block of the column.
3887    code_bits: usize,
3888    /// The sorted order turned round, built the first time a reader asks for it.
3889    ///
3890    /// Four bytes per value against the four the offsets already hold, so a column that has this is
3891    /// carrying half again what it carried before rather than something of a new order. It is built
3892    /// only when something asks, which is a grouped min or max over this column and nothing else,
3893    /// and that reader was going to read the payload of this column once per row otherwise.
3894    code_ranks: OnceLock<Option<Vec<u32>>>,
3895    /// Where each block of the payload starts in the file, and how many stored bytes it is.
3896    ///
3897    /// Absolute rather than an offset from a base the blocks share, because a block is written the
3898    /// moment it fills and what comes after it in the file is whatever the load wrote next. A file
3899    /// old enough to have them back to back is read into these same two lists by adding the base to
3900    /// the ends it carries, so nothing below here knows which kind of file it came from.
3901    starts: Vec<u64>,
3902    lengths: Vec<u64>,
3903    hashes: Vec<u64>,
3904    /// Conservative four-byte substring signatures, read only by a compatible LIKE filter.
3905    grams: Option<NativeGrams>,
3906    /// The payload, read and decoded a block at a time and kept after that.
3907    blocks: Vec<OnceLock<Result<Vec<u8>>>>,
3908    /// How many decoded payload bytes this column keeps before a sweep stops keeping what it reads.
3909    /// [`TEXT_KEEP_BUDGET`] everywhere but in the test of the ceiling.
3910    keep_budget: usize,
3911    /// Roughly how many decoded payload bytes are being kept, which is what [`TEXT_KEEP_BUDGET`]
3912    /// is measured against.
3913    ///
3914    /// Roughly, because two threads that keep the same block at the same time both add its length
3915    /// while [`OnceLock`] keeps one of the two. That makes the count read high and the budget bind
3916    /// a little early, which is the harmless direction, and it costs one relaxed add a block rather
3917    /// than a lock on the path every scan of a string column goes through.
3918    payload_kept: AtomicUsize,
3919    /// Which payload blocks a sweep has decoded before, one flag a block.
3920    ///
3921    /// A sweep keeps a block the second time it decodes it and not the first. A process that runs
3922    /// one statement, which is how a benchmark or a script uses the engine, sweeps each block once
3923    /// and so keeps nothing: on ten million rows a `URL LIKE` held 396 MB with every block kept and
3924    /// 97 MB with none, for the same processor time. A session that asks again pays the decode one
3925    /// more time and reads kept blocks from then on, under the same [`TEXT_KEEP_BUDGET`].
3926    swept: Vec<AtomicBool>,
3927    /// The boundaries this dictionary has already been searched for, by the value searched for.
3928    ///
3929    /// A search is the expensive thing this type does. It settles a probe on the stored head where
3930    /// it can and reads a value where it cannot, and reading a value decodes the payload block it
3931    /// sits in, so one search can cost several blocks. The thing that makes remembering worth it is
3932    /// that the same search comes back: a top N asks once a chunk whether anything left can beat its
3933    /// worst candidate, and the worst candidate settles long before the chunks run out.
3934    ///
3935    /// Shared across the instances of a scan rather than kept per instance, because each of them has
3936    /// its own worst candidate and all of them are searching the same dictionary. One lock per chunk
3937    /// is nothing next to a probe of a file.
3938    ///
3939    /// Bounded by [`TEXT_SEARCH_MEMO`] and emptied rather than evicted when it is full. What fills
3940    /// it is a top N improving its bound, which happens a few dozen times and then stops, so the
3941    /// bound is there for the filter that searches for a different literal every chunk rather than
3942    /// for anything this is meant to help.
3943    searched: Mutex<HashMap<Vec<u8>, (usize, bool)>>,
3944}
3945
3946#[derive(Debug)]
3947struct NativeGrams {
3948    start: u64,
3949    length: usize,
3950    /// How long one block's signature is.
3951    width: usize,
3952    hash: u64,
3953    /// For each literal asked about lately, whether each block might hold it.
3954    ///
3955    /// The answer for every block at once, worked out by one pass over the signatures a window at a
3956    /// time, rather than the signatures read in and kept. On ClickBench `URL` they are 21 MB for
3957    /// ten million rows and a verdict is 2,650 flags, and a filter asks the same question of every
3958    /// block, so the pass is paid once and what stays resident is the flags.
3959    verdicts: Mutex<Vec<Verdict>>,
3960}
3961
3962/// A literal and whether each block might hold it.
3963type Verdict = (Vec<u8>, Arc<[bool]>);
3964
3965/// How many literals a column remembers the verdicts of.
3966const GRAM_VERDICTS: usize = 8;
3967
3968impl NativeGrams {
3969    /// Whether each block might hold `literal`, remembered or worked out now.
3970    ///
3971    /// The lock is held over the pass so that the threads of one scan, which all ask about the
3972    /// same literal at the start, read the signatures once between them.
3973    fn verdicts(&self, file: &File, literal: &[u8]) -> Result<Arc<[bool]>> {
3974        let mut held = self.verdicts.lock().map_err(|_| invalid("a poisoned signature verdict"))?;
3975        if let Some((_, verdict)) = held.iter().find(|(asked, _)| asked == literal) {
3976            return Ok(Arc::clone(verdict));
3977        }
3978        let wanted = literal.windows(4).map(|gram| gram_bits(gram, self.width)).collect::<Vec<_>>();
3979        let mut verdict = Vec::with_capacity(self.length / self.width);
3980        let window = GRAM_WINDOW / self.width * self.width;
3981        let hash = walk_checksummed(file, self.start, self.length, window, |bytes| {
3982            verdict.extend(bytes.chunks(self.width).map(|bits| {
3983                wanted
3984                    .iter()
3985                    .flatten()
3986                    .all(|&bit| bits.get(bit / 8).is_some_and(|byte| byte & (1 << (bit % 8)) != 0))
3987            }));
3988            Ok(())
3989        })?;
3990        if hash != self.hash {
3991            return Err(invalid("global dictionary substring signatures checksum differs"));
3992        }
3993        let verdict: Arc<[bool]> = verdict.into();
3994        if held.len() >= GRAM_VERDICTS {
3995            held.remove(0);
3996        }
3997        held.push((literal.to_vec(), Arc::clone(&verdict)));
3998        Ok(verdict)
3999    }
4000
4001    fn footprint(&self) -> usize {
4002        self.verdicts.lock().map_or(0, |held| {
4003            held.iter().map(|(asked, verdict)| asked.capacity() + verdict.len()).sum()
4004        })
4005    }
4006}
4007
4008/// How many searched for values a column's dictionary remembers the boundary of.
4009///
4010/// See [`NativeText::searched`]. Small because the case it is for repeats one value, not because a
4011/// larger one would be wrong.
4012const TEXT_SEARCH_MEMO: usize = 64;
4013
4014/// How many values of a dictionary go in one block of the payload.
4015///
4016/// The block is the unit the string cascade encodes, the unit a checksum covers, and the unit a
4017/// reader has to decode to get at a single value, so it is the one number the payload format turns
4018/// on. Blocking by values rather than by bytes is what keeps a value out of two blocks at once: the
4019/// block holding a code is `code / TEXT_PAYLOAD_VALUES` and nothing has to be stitched.
4020///
4021/// A probe on the five ClickBench columns that have a dictionary worth the name, written up on
4022/// #347, measured the ratio and the decode speed at 128, 256, 512, 1,024 and 4,096 values. Both get
4023/// better all the way up, because front coding and the LZ matcher have more to look back at and
4024/// because the per chunk setup is spread over more values. What stops it is the point read: a query
4025/// that wants ten values has to decode ten blocks, so the block is what a lookup costs. At 1,024
4026/// values a block is between 67 KB and 394 KB decoded across those five columns, and the ratios are
4027/// 2.3 to 4.5. Going up to 4,096 buys two to six percent more and makes a block as much as 1.5 MB.
4028/// Going down to 512 gives up five to nine percent.
4029const TEXT_PAYLOAD_VALUES: usize = 1024;
4030
4031/// Eight KiB per payload block, which is what makes a four-byte substring a useful negative test on
4032/// a column of URLs.
4033///
4034/// Two KiB was the first answer and on ClickBench `URL` it proved almost nothing. A block of 1,024
4035/// sorted URLs holds about seventeen thousand distinct four-byte grams, and at two bits each that
4036/// set nine in ten of the sixteen thousand bits there were, so `LIKE '%google%'` passed most blocks
4037/// it had no match in and decoded them. At eight KiB four bits in ten are set, and of the 2,650
4038/// blocks of `URL` in ten million rows a needle that is in none of them passes 36. The signatures
4039/// are not read into memory, see [`NativeGrams::verdicts`], so the width costs file and not
4040/// resident memory.
4041const TEXT_GRAM_BYTES: usize = 8192;
4042
4043/// The signature width of a format 28 file, which is still read.
4044const NARROW_GRAM_BYTES: usize = 2048;
4045
4046/// How much of a column's signatures a verdict reads at a time.
4047const GRAM_WINDOW: usize = 256 << 10;
4048
4049/// A fast mixing step for exactly four bytes, shared by load and query, into a signature of
4050/// `width` bytes.
4051fn gram_bits(bytes: &[u8], width: usize) -> [usize; 2] {
4052    let original = u32::from_le_bytes(bytes.try_into().expect("a four-byte gram"));
4053    let mut first = original ^ (original >> 16);
4054    first = first.wrapping_mul(0x7feb_352d);
4055    first ^= first >> 15;
4056    let mut second = original ^ (original >> 17);
4057    second = second.wrapping_mul(0x846c_a68b);
4058    second ^= second >> 16;
4059    let mask = width * 8 - 1;
4060    [(first as usize) & mask, (second as usize) & mask]
4061}
4062
4063/// How many decoded payload bytes one dictionary keeps before a sweep stops keeping what it reads.
4064///
4065/// A sweep of the whole dictionary decodes every block whatever it does, and the only question is
4066/// whether it hangs on to them. Keeping all of them is 4.2 GB on ClickBench `URL` at a hundred
4067/// million rows, which is what #997 was right to stop. Keeping none of them means the next query
4068/// asking the same thing decodes all of it again, and on the same column at a million rows that
4069/// took a `LIKE` from 2.7 ms to 16.2 ms, because the decode used to be paid once by a session and
4070/// is now paid by every statement in it. Neither end is the answer. A bound is.
4071///
4072/// So a sweep keeps what it decodes until the column is holding this much and decodes without
4073/// keeping after that. At a million rows the five ClickBench string columns decode to between 8 MB
4074/// and 85 MB, so they sit inside it and a repeated `LIKE` reads a decoded block rather than a
4075/// stored one. At a hundred million rows `URL` fills it and the rest of that column is read and
4076/// dropped, which is the old cost on the part that does not fit and none of the old footprint.
4077///
4078/// Two hundred and fifty six megabytes a column is a number and not a policy, and the policy is
4079/// what should replace it: this wants to be a buffer pool over the whole database, sized against
4080/// the memory limit the session was given, with the blocks of every column competing for it and the
4081/// least useful one evicted. That is F2 work. What is here is the part of it that can be written
4082/// without an eviction order, which is a ceiling.
4083const TEXT_KEEP_BUDGET: usize = 256 * 1024 * 1024;
4084
4085/// The length of every value out of where each one ends inside its payload block, or `None` for
4086/// ends that go backwards somewhere inside a block.
4087///
4088/// A value that opens a block starts at zero and every other one starts where the value before it
4089/// ends, so a block is a run of differences.
4090fn lengths_of(ends: &[u32]) -> Option<Vec<u32>> {
4091    let mut lens = Vec::with_capacity(ends.len());
4092    for block in ends.chunks(TEXT_PAYLOAD_VALUES) {
4093        let mut start = 0;
4094        for &end in block {
4095            lens.push(end.checked_sub(start)?);
4096            start = end;
4097        }
4098    }
4099    Some(lens)
4100}
4101
4102/// How many offsets go in one packed run.
4103///
4104/// A payload block holds 1,024 values and `bitpack::pack_tail` takes fewer than 1,024 at a time,
4105/// since a whole unit of that many belongs in the transposed layout instead. So the offsets of a
4106/// block go in two runs. Five hundred and twelve values at any width is a whole number of bytes, so
4107/// a run starts where a multiply says it does and nothing is padded.
4108const TEXT_OFFSET_RUN: usize = 512;
4109
4110/// Bytes at the front of a global dictionary index: the value count, the values a payload block
4111/// holds, the block count and the bits an offset is packed at.
4112const DICTIONARY_HEADER: usize = 16;
4113
4114/// Set beside the offset width in the fourth word of a global dictionary index, meaning each
4115/// payload block says where in the file it starts and how long it is, rather than sitting directly
4116/// behind the block before it.
4117///
4118/// In that word rather than in a word of its own because the width is at most 32 and lives in a
4119/// `u32`, so the top of it has never been anything. A build old enough not to know the flag reads
4120/// the file's format before it reads any of this and refuses it there, and if it somehow did get
4121/// here it would find an offset width of two billion and say so.
4122///
4123/// The point of the flag is that a block written the moment it fills does not know what will be
4124/// written after it, so the payload of a column cannot be one run of bytes unless the whole column
4125/// is held until the file is closed. That is the memory the load cannot afford. What it costs is
4126/// eight bytes a block, against the block being a thousand values.
4127const DICTIONARY_SCATTERED: u32 = 1 << 31;
4128/// The dictionary index carries one four-byte substring signature per payload block.
4129const DICTIONARY_GRAMS: u32 = 1 << 30;
4130/// Each signature is [`TEXT_GRAM_BYTES`] long rather than the [`NARROW_GRAM_BYTES`] a format 28
4131/// file wrote.
4132const DICTIONARY_WIDE_GRAMS: u32 = 1 << 29;
4133/// Every flag the width word of a dictionary can carry above the offset width.
4134const DICTIONARY_FLAGS: u32 = DICTIONARY_SCATTERED | DICTIONARY_GRAMS | DICTIONARY_WIDE_GRAMS;
4135
4136/// How many entries of a dictionary's sorted order sit in one block that is read and checked as a
4137/// unit.
4138///
4139/// Five hundred and twelve entries is between two and three kilobytes on the ClickBench string
4140/// columns, which is well under a page. A binary search over half a million entries makes nineteen
4141/// probes, and the first ten land in ten different blocks while the last nine land in the one block
4142/// that holds the answer, so the whole search reads about thirty kilobytes of a megabyte of order. A
4143/// smaller block would save a little on the early probes, cost a checksum and an end list four times
4144/// as long, and give the heads less to share a base with. A larger one would read more than it uses
4145/// on every probe.
4146const TEXT_RANK_BLOCK: usize = 512;
4147
4148/// Bytes at the front of a rank block, which is the base of its heads and the width they are packed
4149/// at.
4150///
4151/// An entry used to be twelve bytes flat, eight for the head and four for the code, and on the five
4152/// ClickBench columns that have a dictionary worth the name that was 744 MB of a 12.2 GB file. Both
4153/// halves of it are nearly empty. The heads are the first eight bytes of the values in sorted order,
4154/// so a block of five hundred and twelve of them spans a tiny slice of the column, and on a column of
4155/// URLs they are all `http://w` and the block holds one distinct head. The codes are positions in a
4156/// dictionary of eighteen million, which is twenty five bits and not thirty two.
4157///
4158/// So a block now writes the smallest head in it, the bits the largest is above that, and the heads
4159/// and the codes packed at the width each needs. A block where every head agrees costs nine bytes
4160/// and the codes.
4161const RANK_BLOCK_HEADER: usize = size_of::<u64>() + 1;
4162
4163impl NativeText {
4164    /// One block of the payload, read and decoded the first time anything asks for a value in it.
4165    ///
4166    /// The bytes handed back are the values of the block laid end to end, which is what the offsets
4167    /// describe, so a caller slices it with the offsets it already has. Where the block sits in the
4168    /// file is the only thing the caller cannot work out for itself, because the stored form is
4169    /// shorter than the decoded one and by a different amount in every block.
4170    fn payload_block(&self, block: usize) -> Result<Option<&[u8]>> {
4171        let Some(slot) = self.blocks.get(block) else { return Ok(None) };
4172        let bytes = slot.get_or_init(|| self.decode_block(block)).as_ref().map_err(Clone::clone)?;
4173        Ok(Some(bytes.as_slice()))
4174    }
4175
4176    /// Reads and decodes one block of the payload, without deciding who keeps it.
4177    ///
4178    /// [`Self::payload_block`] keeps it forever, which is what a point read wants and what a walk
4179    /// of the whole dictionary must not do. Both call this and they differ in nothing else.
4180    fn decode_block(&self, block: usize) -> Result<Vec<u8>> {
4181        let len = self.lengths[block];
4182        let mut stored = vec![
4183            0;
4184            usize::try_from(len).map_err(|_| invalid(
4185                "global dictionary block does not fit in memory"
4186            ))?
4187        ];
4188        read_at(&self.file, self.starts[block], &mut stored)?;
4189        if checksum(&stored) != self.hashes[block] {
4190            return Err(invalid("global dictionary payload checksum differs"));
4191        }
4192        let first = block * TEXT_PAYLOAD_VALUES;
4193        let last = (first + TEXT_PAYLOAD_VALUES).min(self.values);
4194        let want = self.end_within(last - 1)? as usize;
4195        let values = string::decode_flat(&stored)?;
4196        if values.len() != last - first {
4197            return Err(invalid("global dictionary block holds the wrong value count"));
4198        }
4199        let bytes = values.into_bytes();
4200        if bytes.len() != want {
4201            return Err(invalid("global dictionary block decodes to the wrong length"));
4202        }
4203        Ok(bytes)
4204    }
4205
4206    /// How many single offset reads make [`Self::value_ends`] worth building.
4207    ///
4208    /// As many reads as the dictionary has values. Building the table costs about thirty
4209    /// instructions a value once the fresh pages it lands in are counted, and a read out of it saves
4210    /// about thirty five, so it repays itself after roughly one read per value. The reads so far are
4211    /// the only guess there is at the reads to come, and waiting until they match the size of the
4212    /// dictionary is betting that a column read that much will be read that much again.
4213    ///
4214    /// A sixteenth was the first answer, from counting the unpacking alone at three instructions a
4215    /// value. ClickBench 38 showed what that missed: it reads about twenty thousand titles a
4216    /// statement out of a dictionary of three hundred and fifty thousand, crossed a sixteenth in its
4217    /// second statement and was two percent slower for a table it did not read enough to repay. A
4218    /// scan asking for the length of every row crosses it part way through its first statement on
4219    /// ClickBench, where a string column has about two rows for every value, and a filter that keeps
4220    /// a few thousand rows never does. The floor is there
4221    /// because a short dictionary would otherwise build a table for a handful of reads.
4222    fn ends_worth_unpacking(&self) -> usize {
4223        self.values.max(TEXT_PAYLOAD_VALUES)
4224    }
4225
4226    /// The unpacked ends, if they are built or if this read is the one that makes them worth it.
4227    fn value_ends(&self) -> Option<&[u32]> {
4228        if let Some(built) = self.value_ends.get() {
4229            return built.as_deref();
4230        }
4231        if self.ends_asked.fetch_add(1, Atomic::Relaxed) < self.ends_worth_unpacking() {
4232            return None;
4233        }
4234        self.value_ends.get_or_init(|| self.unpack_ends()).as_deref()
4235    }
4236
4237    /// Every end of the column, a run at a time.
4238    ///
4239    /// `None` rather than an error on anything wrong, because this is a cache in front of a reader
4240    /// that answers the same question. A column whose offsets are short or whose ends do not fit in
4241    /// four bytes gets no table and the same error it would have got, from the read that wanted it.
4242    fn unpack_ends(&self) -> Option<Vec<u32>> {
4243        let mut ends = vec![0u32; self.values];
4244        for (run, into) in ends.chunks_mut(TEXT_OFFSET_RUN).enumerate() {
4245            let bytes = self.offsets.get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)?;
4246            bitpack::unpack_tail_into(bytes, self.offset_bits, into, |bits| {
4247                u32::try_from(bits).unwrap_or(u32::MAX)
4248            })
4249            .ok()?;
4250        }
4251        // An end that did not fit was stored as the sentinel, and a real one cannot reach it because
4252        // a payload block is far smaller than four gigabytes. So the column keeps the packed reader.
4253        if ends.contains(&u32::MAX) { None } else { Some(ends) }
4254    }
4255
4256    /// Where the value at `index` ends inside its payload block.
4257    fn end_within(&self, index: usize) -> Result<u32> {
4258        if let Some(ends) = self.value_ends() {
4259            return ends
4260                .get(index)
4261                .copied()
4262                .ok_or_else(|| invalid("global dictionary offsets are short"));
4263        }
4264        let run = index / TEXT_OFFSET_RUN;
4265        let bytes = self
4266            .offsets
4267            .get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)
4268            .ok_or_else(|| invalid("global dictionary offsets are short"))?;
4269        let end = bitpack::tail_at(bytes, self.offset_bits, index % TEXT_OFFSET_RUN)
4270            .map_err(|_| invalid("global dictionary offsets are short"))?;
4271        u32::try_from(end).map_err(|_| invalid("global dictionary offset is past the payload"))
4272    }
4273
4274    /// Where every value in `first..last` ends inside its payload block, in one pass over the runs.
4275    ///
4276    /// [`Self::end_within`] answers for one value and pays for it twice over: it shifts a window to
4277    /// the bit the value starts at, and the copy that fills that window is a length the compiler does
4278    /// not know, so it is a call to `memcpy` rather than a load. A sweep asked for two of those per
4279    /// value, one for the end and one for the start that is the end before it, and on the ClickBench
4280    /// `URL` dictionary of eighteen million that was most of the half second a `LIKE` over it took.
4281    ///
4282    /// [`bitpack::unpack_tail_into`] walks the run instead, which makes the window a fixed width and
4283    /// so an unaligned load, and reads the bit position off a counter. A run is five hundred and
4284    /// twelve values and a block is two of them, so a block of a thousand and twenty four values
4285    /// costs two calls here and nothing per value.
4286    ///
4287    /// The answer is written straight into the result. A run that is wanted from its first value,
4288    /// which is every run but the one the sweep starts in, unpacks into its own window of the result
4289    /// and is never copied. Only a run joined part way through needs the scratch buffer, and there is
4290    /// at most one of those per sweep, so the buffer is allocated the first time one turns up.
4291    fn ends_within(&self, first: usize, last: usize) -> Result<Vec<u64>> {
4292        let mut ends = vec![0u64; last.saturating_sub(first)];
4293        let mut scratch = Vec::new();
4294        let mut at = first;
4295        while at < last {
4296            let run = at / TEXT_OFFSET_RUN;
4297            let stop = ((run + 1) * TEXT_OFFSET_RUN).min(last);
4298            let held = self.values.saturating_sub(run * TEXT_OFFSET_RUN).min(TEXT_OFFSET_RUN);
4299            let bytes = self
4300                .offsets
4301                .get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)
4302                .ok_or_else(|| invalid("global dictionary offsets are short"))?;
4303            let from = at % TEXT_OFFSET_RUN;
4304            let upto = stop - run * TEXT_OFFSET_RUN;
4305            if upto > held || bytes.len() < bitpack::tail_len(held, self.offset_bits) {
4306                return Err(invalid("global dictionary offsets are short"));
4307            }
4308            let into = &mut ends[at - first..stop - first];
4309            if from == 0 {
4310                bitpack::unpack_tail_into(bytes, self.offset_bits, into, |bits| bits)
4311                    .map_err(|_| invalid("global dictionary offsets are short"))?;
4312            } else {
4313                scratch.resize(held, 0);
4314                bitpack::unpack_tail_into(bytes, self.offset_bits, &mut scratch, |bits| bits)
4315                    .map_err(|_| invalid("global dictionary offsets are short"))?;
4316                into.copy_from_slice(&scratch[from..upto]);
4317            }
4318            at = stop;
4319        }
4320        Ok(ends)
4321    }
4322
4323    /// Where the value at `index` starts inside its payload block, which is where the value before
4324    /// it ended unless it is the first of the block.
4325    fn start_within(&self, index: usize) -> Result<u32> {
4326        if index % TEXT_PAYLOAD_VALUES == 0 { Ok(0) } else { self.end_within(index - 1) }
4327    }
4328
4329    /// Where the value at `index` starts and ends inside its payload block.
4330    ///
4331    /// The two offsets sit next to each other in the same run unless the value opens one, and a run
4332    /// of seventeen bit offsets, which is what a block of a thousand strings needs, puts a pair of
4333    /// them inside one eight byte load. So the common case reads the packed bytes once rather than
4334    /// twice and does the bounds arithmetic once. This is asked once per string a text column hands
4335    /// out, and on ClickBench 27 the two reads together were a quarter of the query.
4336    fn span_within(&self, index: usize) -> Result<(u32, u32)> {
4337        if let Some(ends) = self.value_ends() {
4338            let end =
4339                *ends.get(index).ok_or_else(|| invalid("global dictionary offsets are short"))?;
4340            // The value before it in the same block, and zero where there is no value before it.
4341            // `index` is inside the table, so the one under it is too.
4342            let start = if index % TEXT_PAYLOAD_VALUES == 0 { 0 } else { ends[index - 1] };
4343            if start > end {
4344                return Err(invalid("global dictionary value ends before it starts"));
4345            }
4346            return Ok((start, end));
4347        }
4348        let within = index % TEXT_OFFSET_RUN;
4349        let (start, end) = if within == 0 {
4350            (self.start_within(index)?, self.end_within(index)?)
4351        } else {
4352            let run = index / TEXT_OFFSET_RUN;
4353            let bytes = self
4354                .offsets
4355                .get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)
4356                .ok_or_else(|| invalid("global dictionary offsets are short"))?;
4357            let (start, end) = bitpack::tail_pair(bytes, self.offset_bits, within)
4358                .map_err(|_| invalid("global dictionary offsets are short"))?;
4359            let ends = u32::try_from(end)
4360                .map_err(|_| invalid("global dictionary offset is past the payload"))?;
4361            let starts = u32::try_from(start)
4362                .map_err(|_| invalid("global dictionary offset is past the payload"))?;
4363            (starts, ends)
4364        };
4365        if start > end {
4366            return Err(invalid("global dictionary value ends before it starts"));
4367        }
4368        Ok((start, end))
4369    }
4370
4371    /// The block of the sorted order that holds `rank`, and where in it that rank sits.
4372    ///
4373    /// The block is read from the file and checked against the hash the index carries for it the
4374    /// first time anything asks, and kept after that, the same way a payload block is. A search
4375    /// makes about as many probes as the order has bits, so the whole search reads a handful of
4376    /// these and never the rest.
4377    fn rank_parts(&self, rank: usize) -> Result<(&[u8], usize)> {
4378        let slot = self
4379            .rank_blocks
4380            .get(rank / TEXT_RANK_BLOCK)
4381            .ok_or_else(|| invalid("global dictionary rank is past the order"))?;
4382        let block = slot
4383            .get_or_init(|| {
4384                let which = rank / TEXT_RANK_BLOCK;
4385                let start = if which == 0 { 0 } else { self.rank_ends[which - 1] };
4386                let end = self.rank_ends[which];
4387                let mut bytes = vec![0; (end - start) as usize];
4388                read_at(&self.file, self.rank_at + start, &mut bytes)?;
4389                if checksum(&bytes)
4390                    != *self
4391                        .rank_hashes
4392                        .get(rank / TEXT_RANK_BLOCK)
4393                        .ok_or_else(|| invalid("global dictionary rank block has no checksum"))?
4394                {
4395                    return Err(invalid("global dictionary rank checksum differs"));
4396                }
4397                Ok(bytes)
4398            })
4399            .as_ref()
4400            .map_err(Clone::clone)?;
4401        Ok((block.as_slice(), rank % TEXT_RANK_BLOCK))
4402    }
4403
4404    /// The first eight bytes of the value at `rank`, as the integer a comparison reads.
4405    fn head_at(&self, rank: usize) -> Result<u64> {
4406        let (block, within) = self.rank_parts(rank)?;
4407        let (base, width, packed) = rank_heads(block)?;
4408        let above = bitpack::tail_at(packed, width, within)
4409            .map_err(|_| invalid("global dictionary rank block is short of heads"))?;
4410        Ok(base.wrapping_add(above))
4411    }
4412
4413    /// The packed codes of one rank block, which follow the heads on the next byte boundary.
4414    fn rank_codes<'block>(&self, block: &'block [u8], count: usize) -> Result<&'block [u8]> {
4415        let (_, width, packed) = rank_heads(block)?;
4416        packed
4417            .get(bitpack::tail_len(count, width)..)
4418            .ok_or_else(|| invalid("global dictionary rank block is short of codes"))
4419    }
4420
4421    /// How many entries the block holding `rank` has, which is a full block except at the end.
4422    fn rank_block_len(&self, rank: usize) -> usize {
4423        let first = rank / TEXT_RANK_BLOCK * TEXT_RANK_BLOCK;
4424        TEXT_RANK_BLOCK.min(self.ranks - first)
4425    }
4426}
4427
4428/// The base, the width and the packed bytes of one rank block's heads.
4429fn rank_heads(block: &[u8]) -> Result<(u64, usize, &[u8])> {
4430    let header = block
4431        .get(..RANK_BLOCK_HEADER)
4432        .ok_or_else(|| invalid("global dictionary rank block is short"))?;
4433    let base = u64::from_le_bytes(header[..8].try_into().expect("eight bytes"));
4434    let width = header[8] as usize;
4435    if width > 64 {
4436        return Err(invalid("global dictionary rank block packs heads past a word"));
4437    }
4438    Ok((base, width, &block[RANK_BLOCK_HEADER..]))
4439}
4440
4441/// Bits one offset of a dictionary takes, which is what its widest payload block spans.
4442///
4443/// One width for the whole column rather than one a block. A block is 1,024 values of the same
4444/// column, so the blocks of a column are within a factor of two of each other on every ClickBench
4445/// string column, and a width a block would save a fraction of a bit and cost a byte a block plus
4446/// the arithmetic that finds where a block starts.
4447fn offset_width(ends: &[u32]) -> usize {
4448    // The ends are already relative to the block the value is in, so the last end of a block is that
4449    // block's total and the largest end anywhere is the widest block. There is no subtraction left
4450    // to do and no need to walk the blocks to find where one starts.
4451    let span = ends.iter().copied().max().unwrap_or(0);
4452    (u32::BITS - span.leading_zeros()) as usize
4453}
4454
4455/// How many bytes `values` offsets take at `bits`, which is what the reader has to know before it
4456/// has read any of them.
4457fn offset_bytes(values: usize, bits: usize) -> usize {
4458    let full = values / TEXT_OFFSET_RUN;
4459    let rest = values % TEXT_OFFSET_RUN;
4460    full * TEXT_OFFSET_RUN / 8 * bits + bitpack::tail_len(rest, bits)
4461}
4462
4463/// The end of every value within its payload block, packed a run at a time.
4464/// A run never straddles a block, because [`TEXT_OFFSET_RUN`] divides [`TEXT_PAYLOAD_VALUES`], which
4465/// is what lets this be a walk of the ends rather than arithmetic against a per block base.
4466fn encode_offsets(ends: &[u32], bits: usize, out: &mut Vec<u8>) -> Result<()> {
4467    let mut run = Vec::with_capacity(TEXT_OFFSET_RUN);
4468    for chunk in ends.chunks(TEXT_OFFSET_RUN) {
4469        run.clear();
4470        run.extend(chunk.iter().map(|&end| u64::from(end)));
4471        bitpack::pack_tail(&run, bits, out)
4472            .map_err(|_| invalid("global dictionary offsets do not pack"))?;
4473    }
4474    Ok(())
4475}
4476
4477/// How many bits a code of a dictionary of `values` entries takes.
4478fn code_width(values: usize) -> usize {
4479    match u64::try_from(values).unwrap_or(u64::MAX) {
4480        0 | 1 => 0,
4481        last => (u64::BITS - (last - 1).leading_zeros()) as usize,
4482    }
4483}
4484
4485impl TextSource for NativeText {
4486    fn len(&self) -> usize {
4487        self.values
4488    }
4489
4490    fn might_contain(&self, first: usize, literal: &[u8]) -> Result<bool> {
4491        let Some(grams) = &self.grams else { return Ok(true) };
4492        if literal.len() < 4 || first >= self.values {
4493            return Ok(true);
4494        }
4495        let verdict = grams.verdicts(&self.file, literal)?;
4496        Ok(verdict.get(first / TEXT_PAYLOAD_VALUES).copied().unwrap_or(true))
4497    }
4498
4499    fn bytes_at(&self, index: usize) -> Result<Option<&[u8]>> {
4500        if index >= self.values {
4501            return Ok(None);
4502        }
4503        let (start, end) = self.span_within(index)?;
4504        if start == end {
4505            return Ok(Some(&[]));
4506        }
4507        // A block holds a fixed number of values rather than a fixed number of bytes, so the value
4508        // is in one block and the offsets already say where in it.
4509        let block = index / TEXT_PAYLOAD_VALUES;
4510        let Some(bytes) = self.payload_block(block)? else { return Ok(None) };
4511        Ok(bytes.get(start as usize..end as usize))
4512    }
4513
4514    fn bytes_len_at(&self, index: usize) -> Result<Option<usize>> {
4515        if index >= self.values {
4516            return Ok(None);
4517        }
4518        let (start, end) = self.span_within(index)?;
4519        Ok(Some((end - start) as usize))
4520    }
4521
4522    /// Every length out of the unpacked ends in one loop, which is the point of having them.
4523    ///
4524    /// The whole run of positions counts towards [`Self::ends_worth_unpacking`] at once, because a
4525    /// caller asking for a vector of lengths has said how many it wants, and a vector of them is
4526    /// usually enough on its own. Until the table is worth building this is the row at a time read,
4527    /// the same as the default.
4528    fn bytes_lens_at(&self, indices: &[u32], into: &mut [i64]) -> Result<()> {
4529        self.ends_asked.fetch_add(indices.len(), Atomic::Relaxed);
4530        let Some(ends) = self.value_ends() else {
4531            for (slot, &index) in into.iter_mut().zip(indices) {
4532                *slot = self
4533                    .bytes_len_at(index as usize)?
4534                    .map_or(0, |len| i64::try_from(len).unwrap_or(i64::MAX));
4535            }
4536            return Ok(());
4537        };
4538        if let Some(lens) = self.value_lens.get_or_init(|| lengths_of(ends)) {
4539            for (slot, &index) in into.iter_mut().zip(indices) {
4540                // Past the end is no value and so no length, which is what a row at a time read
4541                // says.
4542                *slot = lens.get(index as usize).map_or(0, |&len| i64::from(len));
4543            }
4544            return Ok(());
4545        }
4546        for (slot, &index) in into.iter_mut().zip(indices) {
4547            let index = index as usize;
4548            // Past the end is no value and so no length, which is what a row at a time read says.
4549            let Some(&end) = ends.get(index) else {
4550                *slot = 0;
4551                continue;
4552            };
4553            let start = if index % TEXT_PAYLOAD_VALUES == 0 { 0 } else { ends[index - 1] };
4554            if start > end {
4555                return Err(invalid("global dictionary value ends before it starts"));
4556            }
4557            *slot = i64::from(end - start);
4558        }
4559        Ok(())
4560    }
4561
4562    /// The rest of the block holding `first`, decoded into a buffer that may die with the call.
4563    ///
4564    /// A block is the unit this format decodes, so a walk that wants every value is going to decode
4565    /// every block whatever it does. The question is whether it keeps them, and both answers are
4566    /// wrong on their own. [`Self::payload_block`] keeps every block it is asked for, so a reader
4567    /// that walked the whole dictionary through `bytes_at` ended up holding the whole dictionary
4568    /// decoded, 4.2 GB on ClickBench `URL`. Keeping none of them makes the next statement asking
4569    /// the same question decode all of it again, which on the same column at a million rows is a
4570    /// `LIKE` going from 2.7 ms to 16.2 ms.
4571    ///
4572    /// So a sweep keeps what it decodes for the second time while the column is under
4573    /// [`TEXT_KEEP_BUDGET`] and drops it after that. A block already in hand is used where it is there and costs nothing either way.
4574    fn sweep(
4575        &self,
4576        first: usize,
4577        limit: usize,
4578        body: &mut dyn FnMut(usize, &[u8]) -> Result<()>,
4579    ) -> Result<usize> {
4580        let limit = limit.min(self.values);
4581        if first >= limit {
4582            return Ok(first);
4583        }
4584        let block = first / TEXT_PAYLOAD_VALUES;
4585        let last = ((block + 1) * TEXT_PAYLOAD_VALUES).min(limit);
4586        let decoded;
4587        let kept = self.blocks.get(block).and_then(OnceLock::get);
4588        let again = kept.is_none()
4589            && self.swept.get(block).is_some_and(|swept| swept.swap(true, Atomic::Relaxed));
4590        let bytes: &[u8] = match kept {
4591            Some(Ok(kept)) => kept,
4592            _ if again && self.payload_kept.load(Atomic::Relaxed) < self.keep_budget => {
4593                let kept = self
4594                    .payload_block(block)?
4595                    .ok_or_else(|| invalid("global dictionary block is past the payload"))?;
4596                self.payload_kept.fetch_add(kept.len(), Atomic::Relaxed);
4597                kept
4598            }
4599            _ => {
4600                decoded = self.decode_block(block)?;
4601                &decoded
4602            }
4603        };
4604        let ends = self.ends_within(first, last)?;
4605        if ends.len() != last - first {
4606            return Err(invalid("global dictionary offsets are short"));
4607        }
4608        let mut start = u64::from(self.start_within(first)?);
4609        // row at a time: the caller is handed one value after another, and what it does with one is
4610        // its own business, so there is no shape here for anything but a walk.
4611        for (index, &end) in (first..last).zip(&ends) {
4612            let value = usize::try_from(start)
4613                .ok()
4614                .zip(usize::try_from(end).ok())
4615                .and_then(|(from, to)| bytes.get(from..to))
4616                .ok_or_else(|| invalid("global dictionary value is past its block"))?;
4617            body(index, value)?;
4618            start = end;
4619        }
4620        Ok(last)
4621    }
4622
4623    /// Each block the indices land in, decoded once and dropped, or read where it is already kept.
4624    ///
4625    /// Never kept, unlike [`Self::sweep`] under its budget, because a scattered read is a one off:
4626    /// a synopsis turned into values is turned once and remembered by the reader as values, a few
4627    /// kilobytes, where the blocks it went through are megabytes nobody asks for again.
4628    fn visit(
4629        &self,
4630        indices: &[usize],
4631        body: &mut dyn FnMut(usize, &[u8]) -> Result<()>,
4632    ) -> Result<()> {
4633        let mut at = 0;
4634        while at < indices.len() {
4635            let block = indices[at] / TEXT_PAYLOAD_VALUES;
4636            let upto =
4637                at + indices[at..].partition_point(|&index| index / TEXT_PAYLOAD_VALUES == block);
4638            let wanted = &indices[at..upto];
4639            if wanted.iter().any(|&index| index >= self.values) {
4640                return Err(invalid("a visited value is past the global dictionary"));
4641            }
4642            let decoded;
4643            let bytes: &[u8] = match self.blocks.get(block).and_then(OnceLock::get) {
4644                Some(Ok(kept)) => kept,
4645                _ => {
4646                    decoded = self.decode_block(block)?;
4647                    &decoded
4648                }
4649            };
4650            for (offset, &index) in wanted.iter().enumerate() {
4651                let (start, end) = self.span_within(index)?;
4652                let value = bytes
4653                    .get(start as usize..end as usize)
4654                    .ok_or_else(|| invalid("global dictionary value is past its block"))?;
4655                body(at + offset, value)?;
4656            }
4657            at = upto;
4658        }
4659        Ok(())
4660    }
4661
4662    fn ranks(&self) -> Option<usize> {
4663        (self.ranks > 0).then_some(self.ranks)
4664    }
4665
4666    /// The boundary for `wanted`, out of [`Self::searched`] where it is there and put there where
4667    /// it is not.
4668    ///
4669    /// The lock is held over the search rather than dropped and taken again, so that two threads
4670    /// asking for the same value at the same time do the work once between them. That is the shape
4671    /// the scan actually arrives in: sixteen instances of a top N, all reading the same column, all
4672    /// improving their bound over the same early chunks.
4673    fn below(&self, ranks: usize, wanted: &[u8]) -> Result<(usize, bool)> {
4674        let mut memo = self.searched.lock().map_err(|_| invalid("a poisoned dictionary search"))?;
4675        if let Some(&answer) = memo.get(wanted) {
4676            return Ok(answer);
4677        }
4678        let answer = search_below(self, ranks, wanted)?;
4679        if memo.len() >= TEXT_SEARCH_MEMO {
4680            memo.clear();
4681        }
4682        memo.insert(wanted.to_vec(), answer);
4683        Ok(answer)
4684    }
4685
4686    fn compare_rank(&self, rank: usize, wanted: &[u8]) -> Result<Ordering> {
4687        // The head settles the probe unless the two values start with the same eight bytes, and
4688        // only then is a value read. On a column of URLs that is the difference between a search
4689        // that touches one block of the payload and a search that touches nineteen of them.
4690        let settled = self.head_at(rank)?.cmp(&head(wanted));
4691        if settled != Ordering::Equal {
4692            return Ok(settled);
4693        }
4694        let code = self.code_at_rank(rank)?;
4695        let bytes = self
4696            .bytes_at(code as usize)?
4697            .ok_or_else(|| invalid("global dictionary order names a code it does not have"))?;
4698        Ok(bytes.cmp(wanted))
4699    }
4700
4701    fn code_at_rank(&self, rank: usize) -> Result<u32> {
4702        let (block, within) = self.rank_parts(rank)?;
4703        let codes = self.rank_codes(block, self.rank_block_len(rank))?;
4704        let code = bitpack::tail_at(codes, self.code_bits, within)
4705            .map_err(|_| invalid("global dictionary rank block is short of codes"))?;
4706        let code = u32::try_from(code)
4707            .map_err(|_| invalid("global dictionary order names a code it does not have"))?;
4708        if code as usize >= self.len() {
4709            return Err(invalid("global dictionary order names a code it does not have"));
4710        }
4711        Ok(code)
4712    }
4713
4714    fn code_ranks(&self) -> Option<&[u32]> {
4715        // The order is a permutation of the positions, so inverting it needs every position to be
4716        // named exactly once. Anything else and the slice would have holes, and a caller indexing
4717        // it by a code would read a rank that belongs to nothing.
4718        if self.ranks == 0 || self.ranks != self.len() {
4719            return None;
4720        }
4721        self.code_ranks
4722            .get_or_init(|| {
4723                let mut ranks = vec![u32::MAX; self.ranks];
4724                // A block at a time rather than a rank at a time, because reading it per rank pays
4725                // for the bounds check, the division and the lock on every one of them.
4726                for first in (0..self.ranks).step_by(TEXT_RANK_BLOCK) {
4727                    let (block, _) = self.rank_parts(first).ok()?;
4728                    let count = self.rank_block_len(first);
4729                    let codes = self.rank_codes(block, count).ok()?;
4730                    for (within, code) in bitpack::unpack_tail(codes, self.code_bits, count)
4731                        .ok()?
4732                        .into_iter()
4733                        .enumerate()
4734                    {
4735                        let code = usize::try_from(code).ok()?;
4736                        *ranks.get_mut(code)? = u32::try_from(first + within).ok()?;
4737                    }
4738                }
4739                if ranks.contains(&u32::MAX) {
4740                    return None;
4741                }
4742                Some(ranks)
4743            })
4744            .as_deref()
4745    }
4746
4747    fn footprint(&self) -> usize {
4748        self.offsets.capacity()
4749            + self
4750                .value_ends
4751                .get()
4752                .and_then(Option::as_ref)
4753                .map_or(0, |ends| ends.capacity() * size_of::<u32>())
4754            + self
4755                .value_lens
4756                .get()
4757                .and_then(Option::as_ref)
4758                .map_or(0, |lens| lens.capacity() * size_of::<u32>())
4759            + self
4760                .code_ranks
4761                .get()
4762                .and_then(Option::as_ref)
4763                .map_or(0, |ranks| ranks.capacity() * size_of::<u32>())
4764            + self.rank_hashes.capacity() * size_of::<u64>()
4765            + self.rank_ends.capacity() * size_of::<u64>()
4766            + self.rank_blocks.capacity() * size_of::<OnceLock<Result<Vec<u8>>>>()
4767            + self
4768                .rank_blocks
4769                .iter()
4770                .filter_map(OnceLock::get)
4771                .filter_map(|result| result.as_ref().ok())
4772                .map(Vec::capacity)
4773                .sum::<usize>()
4774            + self.blocks.capacity() * size_of::<OnceLock<Result<Vec<u8>>>>()
4775            + self.hashes.capacity() * size_of::<u64>()
4776            + self.starts.capacity() * size_of::<u64>()
4777            + self.lengths.capacity() * size_of::<u64>()
4778            + self.grams.as_ref().map_or(0, NativeGrams::footprint)
4779            + self
4780                .blocks
4781                .iter()
4782                .filter_map(OnceLock::get)
4783                .filter_map(|result| result.as_ref().ok())
4784                .map(Vec::capacity)
4785                .sum::<usize>()
4786    }
4787}
4788
4789/// Every table wide part number in order, with the stripe it belongs to.
4790fn places(table: &Table) -> Result<Vec<Place>> {
4791    let mut places = Vec::with_capacity(table.stripes.len().saturating_mul(STRIPE_PARTS));
4792    for (at, stripe) in table.stripes.iter().enumerate() {
4793        let index = u32::try_from(at).map_err(|_| invalid("too many stripes"))?;
4794        for (part, &rows) in stripe.parts.iter().enumerate() {
4795            places.push(Place {
4796                stripe: index,
4797                part: u32::try_from(part).map_err(|_| invalid("too many parts in a stripe"))?,
4798                rows,
4799            });
4800        }
4801    }
4802    Ok(places)
4803}
4804
4805/// Reads one column's section of a stripe's index page.
4806///
4807/// The section carries its own checksum, so a reader that wants one column out of a hundred and
4808/// five preads a few hundred bytes and still knows that what it got is what was written.
4809fn read_index(file: &File, stripe: &Stripe, column: usize) -> Result<Vec<PartSpan>> {
4810    let parts = stripe.parts.len();
4811    let section = index_section(parts)?;
4812    let at = column.checked_mul(section).ok_or_else(|| invalid("index page offset overflow"))?;
4813    let end = at.checked_add(section).ok_or_else(|| invalid("index page offset overflow"))?;
4814    if end > stripe.index.length as usize {
4815        return Err(invalid("index page is shorter than its columns"));
4816    }
4817    let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
4818    let mut bytes = vec![0; section];
4819    let offset = stripe
4820        .index
4821        .offset
4822        .checked_add(at as u64)
4823        .ok_or_else(|| invalid("index page offset overflow"))?;
4824    read_at(file, offset, &mut bytes)?;
4825    let entries = section - size_of::<u64>();
4826    let stored = u64::from_le_bytes(bytes[entries..].try_into().expect("eight bytes"));
4827    if checksum(&bytes[..entries]) != stored {
4828        // With where it was read from, because the two ways this fires look identical from the
4829        // message alone: a file somebody damaged, and a file we wrote to the wrong offset.
4830        return Err(invalid(&format!(
4831            "index page section checksum differs, column {column} of {parts} parts at {offset}, \
4832             wanted {stored:016x} and got {:016x}",
4833            checksum(&bytes[..entries]),
4834        )));
4835    }
4836    let mut spans = Vec::with_capacity(parts);
4837    let mut start = 0_usize;
4838    for part in 0..parts {
4839        let at = part * INDEX_ENTRY;
4840        let length = u32::from_le_bytes(bytes[at..at + 4].try_into().expect("four bytes")) as usize;
4841        let hash = u64::from_le_bytes(bytes[at + 4..at + 12].try_into().expect("eight bytes"));
4842        spans.push(PartSpan { start, length, hash });
4843        start = start.checked_add(length).ok_or_else(|| invalid("column page length overflow"))?;
4844    }
4845    if start != page.length as usize {
4846        return Err(invalid("column page length differs from its index"));
4847    }
4848    Ok(spans)
4849}
4850
4851/// One part's bytes out of a whole column page.
4852fn part_bytes(page: &[u8], span: PartSpan) -> Result<&[u8]> {
4853    let end = span.start.checked_add(span.length).ok_or_else(|| invalid("part range overflow"))?;
4854    page.get(span.start..end).ok_or_else(|| invalid("part exceeds its column page"))
4855}
4856
4857/// Puts one stripe of one column in the cache, and hands back the page for the pool to count when
4858/// it is a page the column did not already hold.
4859///
4860/// The index goes in its own slot and stays. Only the page is under the budget, and the pool is
4861/// what enforces it, once the caller has let go of the column's lock.
4862fn remember(cached: &mut Cached, held: &CachedColumn) -> Option<(usize, Arc<AtomicBool>)> {
4863    if let Some(slot) = cached.index.get_mut(held.stripe) {
4864        if slot.is_none() {
4865            *slot = Some(Arc::clone(&held.index));
4866        }
4867    }
4868    let page = held.page.clone()?;
4869    let slot = cached.pages.get_mut(held.stripe)?;
4870    if slot.is_some() {
4871        return None;
4872    }
4873    let bytes = page.len();
4874    // Set, so that the page a worker has just paid to read is not the one the pass it pays for
4875    // lets go of before the worker has read a part out of it.
4876    let used = Arc::new(AtomicBool::new(true));
4877    *slot = Some(Resident { page, used: Arc::clone(&used) });
4878    Some((bytes, used))
4879}
4880
4881/// Every table a native file holds, without the directory of any of them.
4882///
4883/// This is what opening a database reads. It is the small level of the directory, so the cost is
4884/// proportional to how many tables there are rather than to how much data they hold, and a session
4885/// that touches two tables of eight decodes two table directories.
4886///
4887/// The file handle is shared with every reader this hands out. Eight tables in one file is one open
4888/// file descriptor, not eight, which is the other thing one file buys over a file per table.
4889#[derive(Debug, Clone)]
4890pub struct Catalog {
4891    file: Arc<File>,
4892    size: u64,
4893    entries: Arc<Vec<Entry>>,
4894    /// The views the file holds, whole, since a view has no second level to read later.
4895    views: Arc<Vec<ViewEntry>>,
4896    opening: Opening,
4897    /// Where every reader this hands out counts its pages.
4898    pool: PagePool,
4899}
4900
4901/// Signed integer sums and non-null counts for selected columns, plus total table rows.
4902#[derive(Debug, Clone, PartialEq, Eq)]
4903pub struct CertifiedSums {
4904    pub columns: Vec<(i128, u64)>,
4905    pub rows: u64,
4906}
4907
4908/// Exact ends of an integer or date column, including a certified all-null column.
4909#[derive(Debug, Clone, Copy, PartialEq, Eq)]
4910pub enum IntegerExtremes {
4911    Null,
4912    Values { low: i128, high: i128 },
4913}
4914
4915/// A complete numeric value-to-row-count synopsis; `None` represents SQL NULL.
4916pub type NumericFrequencies = Vec<(Option<i128>, u64)>;
4917
4918impl Catalog {
4919    /// Reads the highest valid catalog slot and nothing under it.
4920    ///
4921    /// The readers it hands out keep pages in a pool of their own with no budget, so each column
4922    /// holds its floor of four stripes and no more. A database opens with [`Catalog::open_in`].
4923    ///
4924    /// # Errors
4925    ///
4926    /// If the file has no valid committed catalog or a catalog pointer is out of bounds.
4927    pub fn open(path: impl AsRef<Path>) -> Result<Self> {
4928        Self::open_in(path, &PagePool::default())
4929    }
4930
4931    /// The same, with every reader it hands out keeping its pages in `pool`.
4932    ///
4933    /// # Errors
4934    ///
4935    /// If the file has no valid committed catalog or a catalog pointer is out of bounds.
4936    pub fn open_in(path: impl AsRef<Path>, pool: &PagePool) -> Result<Self> {
4937        let (file, size, _, bytes, opening) = slot_bytes(path)?;
4938        let (entries, views) = decode_catalog(&bytes, size)?;
4939        Ok(Self {
4940            file: Arc::new(file),
4941            size,
4942            entries: Arc::new(entries),
4943            views: Arc::new(views),
4944            opening,
4945            pool: pool.clone(),
4946        })
4947    }
4948
4949    /// The tables in the file, in the order they were written.
4950    pub fn names(&self) -> impl ExactSizeIterator<Item = &str> {
4951        self.entries.iter().map(|entry| entry.name.as_str())
4952    }
4953
4954    /// The same tables with how many rows each of them holds.
4955    ///
4956    /// The names alone answer which tables the file has, which is what a checkpoint needs to know.
4957    /// A load asks a second question: whether a table already in the file is really in the way of
4958    /// the one it wants to write. A table with no rows is not, because it has no pages the next
4959    /// generation would have to carry, so the count has to come out of the catalog beside the name.
4960    pub fn rows(&self) -> impl ExactSizeIterator<Item = (&str, usize)> {
4961        self.entries.iter().map(|entry| (entry.name.as_str(), entry.rows))
4962    }
4963
4964    /// The views in the file, in the order they were written.
4965    ///
4966    /// Whole, unlike [`Catalog::names`], which hands back names and makes the caller ask for a table
4967    /// by one. A view is a few strings and a column list and it was all read at open, so there is
4968    /// nothing left to go and fetch and no reason to make the caller ask twice.
4969    pub fn views(&self) -> impl ExactSizeIterator<Item = &ViewEntry> {
4970        self.views.iter()
4971    }
4972
4973    /// How many tables the file holds.
4974    #[must_use]
4975    pub fn len(&self) -> usize {
4976        self.entries.len()
4977    }
4978
4979    /// Whether the file holds no table at all, which is what [`Writer::empty`] writes and what a
4980    /// database somebody dropped the last table out of comes back as.
4981    #[must_use]
4982    pub fn is_empty(&self) -> bool {
4983        self.entries.is_empty()
4984    }
4985
4986    /// Opens one table by name, decoding its directory now.
4987    ///
4988    /// # Errors
4989    ///
4990    /// If there is no table by that name, or its directory is torn or points outside the file.
4991    pub fn table(&self, name: &str) -> Result<Reader> {
4992        let entry = self
4993            .entries
4994            .iter()
4995            .find(|entry| entry.name == name)
4996            .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
4997        // Checked and then decoded a window at a time, so that the directory's own bytes are never
4998        // all in memory beside the table they decode into. It is read twice, and the second read
4999        // comes out of the page cache the first one filled.
5000        let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
5001        if file_checksum(&self.file, offset, length)? != entry.directory.hash {
5002            return Err(invalid(&format!("the directory of table {name} does not checksum")));
5003        }
5004        let mut opening = self.opening;
5005        opening.reads += 1;
5006        opening.bytes += u64::from(entry.directory.length);
5007        Reader::build(
5008            Arc::clone(&self.file),
5009            self.size,
5010            read_directory(Cursor::over(&self.file, offset, length), self.size, Some(offset))?,
5011            u64::from(entry.directory.length),
5012            opening,
5013            self.pool.clone(),
5014        )
5015    }
5016
5017    /// Counts non-null, nonzero values from a validated native directory without building a
5018    /// reader for every stripe. Returns `None` when the bounded frequency synopsis cannot prove
5019    /// the count, so callers can use the ordinary query path.
5020    pub fn nonzero_count(&self, name: &str, column: usize) -> Result<Option<u64>> {
5021        let entry = self
5022            .entries
5023            .iter()
5024            .find(|entry| entry.name == name)
5025            .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
5026        let Some(field) = entry.fields.get(column) else {
5027            return Err(invalid("frequency column index out of range"));
5028        };
5029        if !matches!(
5030            field.ty,
5031            LogicalType::TinyInt
5032                | LogicalType::SmallInt
5033                | LogicalType::Integer
5034                | LogicalType::BigInt
5035                | LogicalType::UTinyInt
5036                | LogicalType::USmallInt
5037                | LogicalType::UInteger
5038                | LogicalType::UBigInt
5039        ) {
5040            return Ok(None);
5041        }
5042        let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
5043        if file_checksum(&self.file, offset, length)? != entry.directory.hash {
5044            return Err(invalid(&format!("the directory of table {name} does not checksum")));
5045        }
5046        if let Some(count) = entry.nonzero.get(column).copied().flatten() {
5047            return Ok(Some(count));
5048        }
5049        quick_nonzero(
5050            Cursor::over(&self.file, offset, length),
5051            &entry.name,
5052            &entry.fields,
5053            entry.rows,
5054            column,
5055        )
5056    }
5057
5058    /// Exact signed-integer sums and non-null counts from the small catalog. The table directory
5059    /// checksum is still checked once before any certificate can answer a query.
5060    pub fn aggregate_sums(&self, name: &str, columns: &[usize]) -> Result<Option<CertifiedSums>> {
5061        let entry = self
5062            .entries
5063            .iter()
5064            .find(|entry| entry.name == name)
5065            .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
5066        let mut sums = Vec::with_capacity(columns.len());
5067        for &column in columns {
5068            let Some(field) = entry.fields.get(column) else {
5069                return Err(invalid("aggregate column index out of range"));
5070            };
5071            if !signed_integer(&field.ty) {
5072                return Ok(None);
5073            }
5074            let Some(sum) = entry.aggregates[column] else {
5075                return Ok(None);
5076            };
5077            sums.push(sum);
5078        }
5079        let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
5080        if file_checksum(&self.file, offset, length)? != entry.directory.hash {
5081            return Err(invalid(&format!("the directory of table {name} does not checksum")));
5082        }
5083        Ok(Some(CertifiedSums { columns: sums, rows: entry.rows as u64 }))
5084    }
5085
5086    /// Exact non-null distinct count from the small catalog, after checking the table directory.
5087    pub fn distinct_count(&self, name: &str, column: usize) -> Result<Option<u64>> {
5088        let entry = self
5089            .entries
5090            .iter()
5091            .find(|entry| entry.name == name)
5092            .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
5093        let Some(count) = entry.distincts.get(column).copied() else {
5094            return Err(invalid("distinct column index out of range"));
5095        };
5096        let Some(count) = count else { return Ok(None) };
5097        let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
5098        if file_checksum(&self.file, offset, length)? != entry.directory.hash {
5099            return Err(invalid(&format!("the directory of table {name} does not checksum")));
5100        }
5101        Ok(Some(count))
5102    }
5103
5104    /// Exact integer or date ends from the small catalog after checking the table directory.
5105    pub fn integer_extremes(&self, name: &str, column: usize) -> Result<Option<IntegerExtremes>> {
5106        let entry = self
5107            .entries
5108            .iter()
5109            .find(|entry| entry.name == name)
5110            .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
5111        let Some(extremes) = entry.extremes.get(column).copied() else {
5112            return Err(invalid("extremes column index out of range"));
5113        };
5114        let Some(extremes) = extremes else { return Ok(None) };
5115        let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
5116        if file_checksum(&self.file, offset, length)? != entry.directory.hash {
5117            return Err(invalid(&format!("the directory of table {name} does not checksum")));
5118        }
5119        Ok(Some(match extremes {
5120            None => IntegerExtremes::Null,
5121            Some((low, high)) => IntegerExtremes::Values { low, high },
5122        }))
5123    }
5124
5125    /// Complete numeric frequencies from the small catalog, after checking the table directory.
5126    pub fn exact_numeric_frequencies(
5127        &self,
5128        name: &str,
5129        column: usize,
5130    ) -> Result<Option<NumericFrequencies>> {
5131        let entry = self
5132            .entries
5133            .iter()
5134            .find(|entry| entry.name == name)
5135            .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
5136        let Some(frequencies) = entry.frequencies.get(column).cloned() else {
5137            return Err(invalid("numeric frequency column index out of range"));
5138        };
5139        let Some(frequencies) = frequencies else { return Ok(None) };
5140        let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
5141        if file_checksum(&self.file, offset, length)? != entry.directory.hash {
5142            return Err(invalid(&format!("the directory of table {name} does not checksum")));
5143        }
5144        Ok(Some(frequencies))
5145    }
5146
5147    /// The schema copied into the small file catalog, available without opening the table directory.
5148    pub fn table_fields(&self, name: &str) -> Option<&[Field]> {
5149        self.entries.iter().find(|entry| entry.name == name).map(|entry| entry.fields.as_slice())
5150    }
5151}
5152
5153/// Where the slot naming `generation` goes, which is the one the generation before it did not use.
5154///
5155/// Generation 1 takes the slot at 16, so a file written once is byte for byte the file this wrote
5156/// before there was a second generation to write.
5157fn slot_offset(generation: u64) -> u64 {
5158    16 + (generation - 1) % 2 * SLOT_BYTES as u64
5159}
5160
5161/// The header and the bytes the highest valid slot points at.
5162///
5163/// Both levels of the directory are reached this way, so the magic check, the version check and the
5164/// choice between the two slots live here rather than being written out twice.
5165fn slot_bytes(path: impl AsRef<Path>) -> Result<(File, u64, Slot, Vec<u8>, Opening)> {
5166    let mut file = File::open(path).map_err(io)?;
5167    let size = file.metadata().map_err(io)?.len();
5168    if size < HEADER {
5169        return Err(invalid("file is shorter than its header"));
5170    }
5171    let mut header = [0; HEADER as usize];
5172    file.read_exact(&mut header).map_err(io)?;
5173    let mut opening = Opening { reads: 1, bytes: HEADER };
5174    let version = u32::from_le_bytes([header[8], header[9], header[10], header[11]]);
5175    // The two halves are worth telling apart. A wrong magic is a file that was never ours and
5176    // the answer is to look at the path. A wrong version is our own file from another build,
5177    // and the number this build wants is the only thing that tells the reader whether to
5178    // rebuild the file or to go back to the binary that wrote it.
5179    if &header[..8] != MAGIC {
5180        return Err(invalid("the header does not begin with a rudb native magic"));
5181    }
5182    if !READABLE.contains(&version) {
5183        return Err(invalid(&format!(
5184            "the file is format {version} and this build reads format {FORMAT}, so it has to \
5185                 be written again"
5186        )));
5187    }
5188    let mut selected = None;
5189    for start in [16, 16 + SLOT_BYTES] {
5190        let slot = Slot::read(&header[start..start + SLOT_BYTES]);
5191        if slot.generation == 0 || slot.length == 0 || slot.length as usize > MAX_DIRECTORY {
5192            continue;
5193        }
5194        let Some(end) = slot.offset.checked_add(u64::from(slot.length)) else { continue };
5195        if slot.offset < HEADER || end > size {
5196            continue;
5197        }
5198        let mut bytes = vec![0; slot.length as usize];
5199        file.seek(SeekFrom::Start(slot.offset)).map_err(io)?;
5200        file.read_exact(&mut bytes).map_err(io)?;
5201        opening.reads += 1;
5202        opening.bytes += u64::from(slot.length);
5203        if checksum(&bytes) == slot.hash
5204            && selected
5205                .as_ref()
5206                .is_none_or(|(old, _): &(Slot, Vec<u8>)| old.generation < slot.generation)
5207        {
5208            selected = Some((slot, bytes));
5209        }
5210    }
5211    let (slot, bytes) = selected.ok_or_else(|| invalid("no committed directory slot is valid"))?;
5212    Ok((file, size, slot, bytes, opening))
5213}
5214
5215impl Reader {
5216    /// Opens a file that holds exactly one table.
5217    ///
5218    /// # Errors
5219    ///
5220    /// If the file has no valid committed directory, a directory pointer is out of bounds, or the
5221    /// file holds more than one table, which is a file that has to be opened by name.
5222    pub fn open(path: impl AsRef<Path>) -> Result<Self> {
5223        let catalog = Catalog::open(path)?;
5224        let mut names = catalog.names();
5225        let name = names.next().ok_or_else(|| invalid("the file holds no table"))?.to_string();
5226        if names.next().is_some() {
5227            return Err(invalid(
5228                "the file holds more than one table, so it has to be opened by name",
5229            ));
5230        }
5231        catalog.table(&name)
5232    }
5233
5234    /// Builds a reader over one decoded table directory.
5235    fn build(
5236        file: Arc<File>,
5237        size: u64,
5238        table: Table,
5239        directory: u64,
5240        opening: Opening,
5241        pool: PagePool,
5242    ) -> Result<Self> {
5243        let places = places(&table)?;
5244        let dictionaries = (0..table.fields.len()).map(|_| OnceLock::new()).collect();
5245        let table_fields = table.fields.len();
5246        let stripes = table.stripes.len();
5247        let columns = (0..table.fields.len())
5248            .map(|_| {
5249                Mutex::new(Cached {
5250                    pages: (0..stripes).map(|_| None).collect(),
5251                    index: (0..stripes).map(|_| None).collect(),
5252                    ..Cached::default()
5253                })
5254            })
5255            .collect::<Vec<_>>();
5256        let cache = Shelf {
5257            columns,
5258            held: (0..table_fields).map(|_| AtomicUsize::new(0)).collect(),
5259            kept: AtomicUsize::new(CACHED_STRIPES_PER_COLUMN),
5260        };
5261        let sieves: Vec<Vec<SieveSlot>> = (0..table.fields.len())
5262            .map(|_| table.stripes.iter().map(|_| OnceLock::new()).collect())
5263            .collect();
5264        let part_ranges: Vec<Vec<RangeSlot>> = (0..table.fields.len())
5265            .map(|_| table.stripes.iter().map(|_| OnceLock::new()).collect())
5266            .collect();
5267        Ok(Self {
5268            file,
5269            table: Arc::new(table),
5270            dictionaries: Arc::new(dictionaries),
5271            loading: Arc::new((0..table_fields).map(|_| Mutex::new(())).collect()),
5272            frequency_values: Arc::new((0..table_fields).map(|_| OnceLock::new()).collect()),
5273            frequency_summaries: Arc::new((0..table_fields).map(|_| OnceLock::new()).collect()),
5274            opened: Arc::new(AtomicUsize::new(0)),
5275            sieves: Arc::new(sieves),
5276            part_ranges: Arc::new(part_ranges),
5277            places: Arc::new(places),
5278            cache: Arc::new(cache),
5279            pool,
5280            pages: Arc::new(AtomicUsize::new(0)),
5281            indexes: Arc::new(AtomicUsize::new(0)),
5282            size,
5283            directory,
5284            opening,
5285        })
5286    }
5287
5288    /// What this reader has read so far, and what opening it cost.
5289    ///
5290    /// Public because the claim of `spec/stats/04-in-memory.md` section 4.2 is about this number
5291    /// and a claim nobody can check is a comment. A caller that wants to know whether opening a
5292    /// file touched the data asks here, and gets an answer that does not depend on what the page
5293    /// cache happened to hold.
5294    #[must_use]
5295    pub fn reads(&self) -> Reads {
5296        Reads {
5297            opening: self.opening,
5298            pages: self.pages.load(Atomic::Relaxed),
5299            indexes: self.indexes.load(Atomic::Relaxed),
5300            dictionaries: self.opened.load(Atomic::Relaxed),
5301        }
5302    }
5303
5304    /// Where the file's bytes went, from the directory alone.
5305    ///
5306    /// No page is read, so this costs the same on a 45 GB table as on an empty one. See [`Layout`]
5307    /// for what is charged where and for why the three things that are not columns stay separate.
5308    #[must_use]
5309    pub fn layout(&self) -> Layout {
5310        let table = &self.table;
5311        let stripes = table.stripes.as_slice();
5312        let columns = table
5313            .fields
5314            .iter()
5315            .enumerate()
5316            .map(|(at, field)| ColumnLayout {
5317                name: field.name.clone(),
5318                kind: field.ty.to_string(),
5319                pages: sum(stripes.iter().map(|stripe| span_bytes(&stripe.pages, at))),
5320                memberships: sum(stripes.iter().map(|stripe| stripe.memberships.bytes(at))),
5321                sieves: sum(stripes.iter().map(|stripe| stripe.sieves.bytes(at))),
5322                part_ranges: sum(stripes.iter().map(|stripe| stripe.part_ranges.bytes(at))),
5323                dictionary: dictionary_bytes(table, at),
5324            })
5325            .collect();
5326        Layout {
5327            file: self.size,
5328            rows: table.rows,
5329            stripes: stripes.len(),
5330            parts: self.places.len(),
5331            columns,
5332            indexes: sum(stripes.iter().map(|stripe| u64::from(stripe.index.length))),
5333            directory: self.directory,
5334            header: HEADER,
5335        }
5336    }
5337
5338    /// What every part of one column is stored as, which is what `pragma_storage_info` reports.
5339    ///
5340    /// Unlike [`Self::layout`] this reads the data, because the encoder's choice is in the page and
5341    /// nowhere else. The directory says how many bytes a column took and says nothing about what
5342    /// shape they are in, and the shape is the question worth asking: the same rows in a different
5343    /// order come back bit packed on one file and plain on another, and that is the difference a
5344    /// clustered load makes to a scan.
5345    ///
5346    /// One read per stripe rather than one per part. A part is a few kilobytes out of a page that
5347    /// is a quarter of a megabyte, so asking part by part would read the same page sixty four
5348    /// times. Nothing is put in the page cache, because a caller asking what a file looks like is
5349    /// not about to scan it and evicting the pages a real query wants would be a poor trade.
5350    ///
5351    /// # Errors
5352    ///
5353    /// If the column is outside the schema, or a page, index section or checksum is invalid.
5354    pub fn stored(&self, column: usize) -> Result<Vec<StoredPart>> {
5355        let field = self
5356            .table
5357            .fields
5358            .get(column)
5359            .ok_or_else(|| invalid("stored column index out of range"))?;
5360        let mut stored = Vec::with_capacity(self.places.len());
5361        let mut row = 0;
5362        for (at, stripe) in self.table.stripes.iter().enumerate() {
5363            let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
5364            let index = read_index(&self.file, stripe, column)?;
5365            let mut bytes = vec![0; page.length as usize];
5366            read_at(&self.file, page.offset, &mut bytes)?;
5367            let ranges = self.stripe_part_ranges(at, column);
5368            for (part, &rows) in stripe.parts.iter().enumerate() {
5369                let span = *index.get(part).ok_or_else(|| invalid("part index out of range"))?;
5370                let held = part_bytes(&bytes, span)?;
5371                let range = ranges.and_then(|held| held.get(part));
5372                stored.push(StoredPart {
5373                    stripe: at,
5374                    part,
5375                    row,
5376                    rows: rows as usize,
5377                    encoding: page_encoding(&field.ty, rows as usize, held),
5378                    bytes: span.length as u64,
5379                    page: page.offset,
5380                    offset: span.start as u64,
5381                    low: range
5382                        .and_then(|range| range.low.clone())
5383                        .and_then(|bound| bound.into_value(&field.ty)),
5384                    high: range
5385                        .and_then(|range| range.high.clone())
5386                        .and_then(|bound| bound.into_value(&field.ty)),
5387                    nulls: range.map(|range| range.nulls),
5388                });
5389                row += rows as usize;
5390            }
5391        }
5392        Ok(stored)
5393    }
5394
5395    /// How many parts the table has, which is how many chunks a scan of it reads.
5396    #[must_use]
5397    pub fn parts(&self) -> usize {
5398        self.places.len()
5399    }
5400
5401    /// The parts of each stripe, in table wide part numbers.
5402    ///
5403    /// A scan that wants one worker to own the page it reads hands work out in these runs. The
5404    /// stripes are contiguous in part numbering and all but the last hold sixty four parts, but a
5405    /// stripe can be flushed early when rows arrive out of order, so the runs are read off the
5406    /// directory rather than worked out from a constant.
5407    #[must_use]
5408    pub fn stripe_parts(&self) -> Vec<std::ops::Range<usize>> {
5409        let mut runs = Vec::with_capacity(self.table.stripes.len());
5410        let mut start = 0;
5411        for stripe in &self.table.stripes {
5412            let end = start + stripe.parts.len();
5413            runs.push(start..end);
5414            start = end;
5415        }
5416        runs
5417    }
5418
5419    /// How many rows one stripe holds, in the numbering [`Self::stripe_parts`] hands back.
5420    ///
5421    /// Off the directory, which is already in memory, rather than by the caller asking for each
5422    /// part in turn through the catalog. Nothing past the end holds any rows.
5423    #[must_use]
5424    pub fn stripe_rows(&self, stripe: usize) -> usize {
5425        self.table.stripes.get(stripe).map_or(0, |held| held.rows)
5426    }
5427
5428    /// Asks the page cache to keep `stripes` stripes of every column instead of the default.
5429    ///
5430    /// This only ever raises the number. A scan that gives each worker a whole stripe has one page
5431    /// per column per worker open at once, and a cache smaller than that is worse than no cache at
5432    /// all: every worker's page is evicted by the others before it has finished its stripe, so it
5433    /// reads a quarter of a megabyte for every part it takes out of it.
5434    pub fn keep_stripes(&self, stripes: usize) {
5435        self.cache.kept.fetch_max(stripes, Atomic::Relaxed);
5436    }
5437
5438    /// Rows in one part, or zero when the part number is past the table.
5439    #[must_use]
5440    pub fn part_rows(&self, at: usize) -> usize {
5441        self.places.get(at).map_or(0, |place| place.rows as usize)
5442    }
5443
5444    /// The committed table directory.
5445    #[must_use]
5446    pub fn table(&self) -> &Table {
5447        &self.table
5448    }
5449
5450    /// Exact leading frequencies when the stored synopsis proves a count-descending prefix.
5451    ///
5452    /// The returned list can be longer than `top`. Keeping the stored tail lets a later TopN apply
5453    /// additional ordering keys without losing a value tied with the requested boundary.
5454    ///
5455    /// # Errors
5456    ///
5457    /// If the column is outside the schema or a stored value does not fit its declared type.
5458    pub fn top_frequencies(&self, column: usize, top: usize) -> Result<Option<Vec<(Value, u64)>>> {
5459        let field = self
5460            .table
5461            .fields
5462            .get(column)
5463            .ok_or_else(|| invalid("frequency column index out of range"))?;
5464        let Some(summary) = self.frequency_summary(column)? else {
5465            return Ok(None);
5466        };
5467        if top == 0 || summary.entries.len() < top {
5468            return Ok(None);
5469        }
5470        let boundary = summary.entries[top - 1].count;
5471        if boundary <= summary.omitted_max {
5472            return Ok(None);
5473        }
5474        self.decode_frequencies(column, &field.ty, &summary.entries).map(Some)
5475    }
5476
5477    /// Exact leading counts for a numeric key paired with a stable-dictionary string key.
5478    ///
5479    /// The stored prefix is returned only when its requested boundary strictly beats the bound on
5480    /// every pair omitted at load time. The returned tail may be longer than `top`, as with
5481    /// [`Self::top_frequencies`], so downstream ordering can settle ties without reading rows.
5482    ///
5483    /// # Errors
5484    ///
5485    /// If either column is outside the schema or persisted pair metadata is inconsistent with the
5486    /// frequency synopsis or dictionary it names.
5487    pub fn top_pair_frequencies(
5488        &self,
5489        first: usize,
5490        second: usize,
5491        top: usize,
5492    ) -> Result<Option<PairFrequencyCounts>> {
5493        if first >= self.table.fields.len() || second >= self.table.fields.len() {
5494            return Err(invalid("pair frequency column index out of range"));
5495        }
5496        let Some(summary) =
5497            self.table.pair_frequencies.iter().find(|summary| {
5498                summary.first as usize == first && summary.second as usize == second
5499            })
5500        else {
5501            return Ok(None);
5502        };
5503        if top == 0 || summary.entries.len() < top {
5504            return Ok(None);
5505        }
5506        let boundary = summary.entries[top - 1].count;
5507        if boundary <= summary.omitted_max {
5508            return Ok(None);
5509        }
5510        let first_summary = self
5511            .frequency_summary(first)?
5512            .ok_or_else(|| invalid("pair frequency first column has no synopsis"))?;
5513        let anchors = self
5514            .decode_frequencies(first, &self.table.fields[first].ty, &first_summary.entries)?
5515            .into_iter()
5516            .map(|(value, _)| value)
5517            .collect::<Vec<_>>();
5518        let dictionary = self
5519            .dictionary(second)?
5520            .ok_or_else(|| invalid("pair frequency second column has no dictionary"))?;
5521        let mut codes = summary.entries.iter().filter_map(|entry| entry.second).collect::<Vec<_>>();
5522        codes.sort_unstable();
5523        codes.dedup();
5524        let texts = dictionary
5525            .try_values_visited(&codes.iter().map(|&code| code as usize).collect::<Vec<_>>())?;
5526        let mut out = Vec::with_capacity(summary.entries.len());
5527        for entry in &summary.entries {
5528            if entry.count < boundary {
5529                break;
5530            }
5531            let first = anchors
5532                .get(entry.first_entry as usize)
5533                .cloned()
5534                .ok_or_else(|| invalid("pair frequency anchor is outside its values"))?;
5535            let second = match entry.second {
5536                None => Value::Null,
5537                Some(code) => {
5538                    let at = codes
5539                        .binary_search(&code)
5540                        .map_err(|_| invalid("pair frequency code was not among the codes read"))?;
5541                    texts[at].clone()
5542                }
5543            };
5544            out.push((vec![first, second], entry.count));
5545        }
5546        Ok(Some(out))
5547    }
5548
5549    /// Every value of one column with the number of rows holding it, when the synopsis is complete.
5550    ///
5551    /// The heavy hitter pass keeps a bounded set of candidates and decrements them all when it runs
5552    /// out of room, so what it usually ends with is the leading values and a bound on everything it
5553    /// dropped. `omitted_max` of zero says that never happened: no candidate was ever decremented and
5554    /// the entries did not overflow the stored budget, so the list is every distinct value of the
5555    /// column with an exact count, and a null counts as a value of its own rather than being skipped.
5556    ///
5557    /// That makes a whole class of question answerable without reading a row. How many rows hold a
5558    /// value, how many do not, and what a `GROUP BY` of that column with a count over it produces are
5559    /// all in here. It is only ever true of a column with few enough distinct values, which is the
5560    /// case worth having, because that is exactly the column a grouping or an equality filter would
5561    /// otherwise walk every row to answer.
5562    ///
5563    /// `None` when the column has no synopsis, or has one that dropped anything.
5564    ///
5565    /// # Errors
5566    ///
5567    /// If the column is outside the schema or a stored value does not fit its declared type.
5568    pub fn exact_frequencies(&self, column: usize) -> Result<Option<Vec<(Value, u64)>>> {
5569        let Some(prefix) = self.frequency_prefix(column)? else {
5570            return Ok(None);
5571        };
5572        Ok((prefix.omitted_max == 0).then_some(prefix.entries))
5573    }
5574
5575    /// Every value the synopsis lists with the number of rows holding it, and a bound on the rest.
5576    ///
5577    /// The counts are exact whether or not the list is complete. The heavy hitter pass keeps a
5578    /// bounded candidate set and then recounts only the candidates that survived it, so a value that
5579    /// made it into the list carries the number of rows that really hold it rather than whatever the
5580    /// pass had left over. What the pass loses is values, not counts.
5581    ///
5582    /// `omitted_max` is how many rows the most common value left out can hold, and zero says nothing
5583    /// was left out at all, which is what [`exact_frequencies`] asks for. Above zero the list is the
5584    /// leading values of the column and everything else is somewhere between no rows and that bound.
5585    ///
5586    /// That prefix is worth reading on its own. A column with a value in half its rows and a long
5587    /// tail behind it has no complete synopsis and never will, and it is the column where dividing
5588    /// the rows by the distinct count is furthest from the truth.
5589    ///
5590    /// `None` when the column has no synopsis.
5591    ///
5592    /// # Errors
5593    ///
5594    /// If the column is outside the schema or a stored value does not fit its declared type.
5595    ///
5596    /// [`exact_frequencies`]: Self::exact_frequencies
5597    pub fn frequency_prefix(&self, column: usize) -> Result<Option<FrequencyPrefix>> {
5598        let field = self
5599            .table
5600            .fields
5601            .get(column)
5602            .ok_or_else(|| invalid("frequency column index out of range"))?;
5603        let Some(summary) = self.frequency_summary(column)? else {
5604            return Ok(None);
5605        };
5606        let entries = self.decode_frequencies(column, &field.ty, &summary.entries)?;
5607        Ok(Some(FrequencyPrefix { entries, omitted_max: summary.omitted_max }))
5608    }
5609
5610    /// One column's synopsis, read back from the file when the directory left it there.
5611    fn frequency_summary(&self, column: usize) -> Result<Option<Cow<'_, FrequencySummary>>> {
5612        Ok(match self.table.frequencies.get(column) {
5613            None | Some(None) => None,
5614            Some(Some(Frequencies::Held(summary))) => Some(Cow::Borrowed(summary)),
5615            Some(Some(Frequencies::Stored { span, values })) => {
5616                let slot = self
5617                    .frequency_summaries
5618                    .get(column)
5619                    .ok_or_else(|| invalid("frequency column index out of range"))?;
5620                if let Some(summary) = slot.get() {
5621                    return Ok(Some(Cow::Borrowed(summary.as_ref())));
5622                }
5623                let field = self
5624                    .table
5625                    .fields
5626                    .get(column)
5627                    .ok_or_else(|| invalid("frequency column index out of range"))?;
5628                let mut bytes = vec![0; span.length as usize];
5629                read_at(&self.file, span.offset, &mut bytes)?;
5630                let summary =
5631                    decode_summary(&mut Cursor::new(&bytes), field, self.table.rows, *values)?;
5632                let summary = summary.ok_or_else(|| invalid("a stored synopsis is missing"))?;
5633                let _ = slot.set(Arc::new(summary));
5634                Some(Cow::Borrowed(slot.get().expect("the decoded summary was stored").as_ref()))
5635            }
5636        })
5637    }
5638
5639    /// Turns stored frequency entries into values of the column's own type.
5640    ///
5641    /// Remembered per column, because the planner asks once for every estimate that touches the
5642    /// column and the executor asks again, and the answer is a few hundred values. The codes of a
5643    /// string column are read through [`Vector::try_values_visited`], which does not keep the blocks
5644    /// it decodes, so what a query answered out of the synopsis holds is those values and not the
5645    /// hundred or so dictionary blocks they are scattered over.
5646    fn decode_frequencies(
5647        &self,
5648        column: usize,
5649        ty: &LogicalType,
5650        entries: &[FrequencyEntry],
5651    ) -> Result<Vec<(Value, u64)>> {
5652        if let Some(values) = self.frequency_values.get(column).and_then(OnceLock::get) {
5653            return Ok(values.as_ref().clone());
5654        }
5655        let values = self.decode_frequencies_once(column, ty, entries)?;
5656        if let Some(slot) = self.frequency_values.get(column) {
5657            let _ = slot.set(Arc::new(values.clone()));
5658        }
5659        Ok(values)
5660    }
5661
5662    fn decode_frequencies_once(
5663        &self,
5664        column: usize,
5665        ty: &LogicalType,
5666        entries: &[FrequencyEntry],
5667    ) -> Result<Vec<(Value, u64)>> {
5668        let stored_texts = self.table.frequency_texts.get(column).filter(|texts| !texts.is_empty());
5669        if stored_texts.is_some_and(|texts| texts.len() != entries.len()) {
5670            return Err(invalid("frequency text count differs from its synopsis"));
5671        }
5672        let dictionary = if *ty == LogicalType::Varchar && stored_texts.is_none() {
5673            self.dictionary(column)?
5674        } else {
5675            None
5676        };
5677        let mut codes = entries
5678            .iter()
5679            .filter_map(|entry| match entry.value {
5680                FrequencyValue::Code(code) => Some(code as usize),
5681                _ => None,
5682            })
5683            .collect::<Vec<_>>();
5684        codes.sort_unstable();
5685        codes.dedup();
5686        let texts = match &dictionary {
5687            Some(dictionary) if !codes.is_empty() => dictionary.try_values_visited(&codes)?,
5688            _ => Vec::new(),
5689        };
5690        let mut out = Vec::with_capacity(entries.len());
5691        for (entry_at, entry) in entries.iter().enumerate() {
5692            let value = match entry.value {
5693                FrequencyValue::Null => {
5694                    if stored_texts.and_then(|texts| texts[entry_at].as_ref()).is_some() {
5695                        return Err(invalid("a null frequency entry has text"));
5696                    }
5697                    Value::Null
5698                }
5699                FrequencyValue::Integer(value) => match *ty {
5700                    LogicalType::TinyInt => Value::TinyInt(
5701                        i8::try_from(value)
5702                            .map_err(|_| invalid("frequency TINYINT is out of range"))?,
5703                    ),
5704                    LogicalType::UTinyInt => Value::UTinyInt(
5705                        u8::try_from(value)
5706                            .map_err(|_| invalid("frequency UTINYINT is out of range"))?,
5707                    ),
5708                    LogicalType::USmallInt => Value::USmallInt(
5709                        u16::try_from(value)
5710                            .map_err(|_| invalid("frequency USMALLINT is out of range"))?,
5711                    ),
5712                    LogicalType::UInteger => Value::UInteger(
5713                        u32::try_from(value)
5714                            .map_err(|_| invalid("frequency UINTEGER is out of range"))?,
5715                    ),
5716                    LogicalType::UBigInt => Value::UBigInt(
5717                        u64::try_from(value)
5718                            .map_err(|_| invalid("frequency UBIGINT is out of range"))?,
5719                    ),
5720                    LogicalType::SmallInt => Value::SmallInt(
5721                        i16::try_from(value)
5722                            .map_err(|_| invalid("frequency SMALLINT is out of range"))?,
5723                    ),
5724                    LogicalType::Integer => Value::Integer(
5725                        i32::try_from(value)
5726                            .map_err(|_| invalid("frequency INTEGER is out of range"))?,
5727                    ),
5728                    LogicalType::BigInt => Value::BigInt(
5729                        i64::try_from(value)
5730                            .map_err(|_| invalid("frequency BIGINT is out of range"))?,
5731                    ),
5732                    LogicalType::Date => Value::Date(
5733                        i32::try_from(value)
5734                            .map_err(|_| invalid("frequency DATE is out of range"))?,
5735                    ),
5736                    LogicalType::Timestamp => Value::Timestamp(
5737                        i64::try_from(value)
5738                            .map_err(|_| invalid("frequency TIMESTAMP is out of range"))?,
5739                    ),
5740                    _ => return Err(invalid("integer frequency belongs to another type")),
5741                },
5742                FrequencyValue::Code(code) => {
5743                    if let Some(text) = stored_texts.and_then(|texts| texts[entry_at].as_ref()) {
5744                        Value::Varchar(
5745                            String::from_utf8(text.clone())
5746                                .map_err(|_| invalid("frequency text is not UTF-8"))?,
5747                        )
5748                    } else {
5749                        if dictionary.is_none() {
5750                            return Err(invalid("frequency code has no dictionary or stored text"));
5751                        }
5752                        let at = codes
5753                            .binary_search(&(code as usize))
5754                            .map_err(|_| invalid("frequency code was not among the codes read"))?;
5755                        texts[at].clone()
5756                    }
5757                }
5758            };
5759            out.push((value, entry.count));
5760        }
5761        Ok(out)
5762    }
5763
5764    /// Sparse rows belonging to the bounded numeric frequency candidate set.
5765    ///
5766    /// The list is omitted when collecting it would exceed the fixed storage budget. A composite
5767    /// aggregate may accept a result over these rows only when its requested boundary is strictly
5768    /// greater than `omitted_max`.
5769    ///
5770    /// # Errors
5771    ///
5772    /// If the column is outside the schema.
5773    pub fn frequency_occurrences(&self, column: usize) -> Result<Option<FrequencyOccurrences>> {
5774        let field = self
5775            .table
5776            .fields
5777            .get(column)
5778            .ok_or_else(|| invalid("frequency column index out of range"))?;
5779        let Some(summary) = self.frequency_summary(column)? else {
5780            return Ok(None);
5781        };
5782        if summary.ordinals.is_empty() {
5783            return Ok(None);
5784        }
5785        let (anchors, anchor_indices) = if summary.ordinal_entries.len() == summary.ordinals.len() {
5786            let entries = self.decode_frequencies(column, &field.ty, &summary.entries)?;
5787            (entries.into_iter().map(|(value, _)| value).collect(), summary.ordinal_entries.clone())
5788        } else {
5789            (Vec::new(), Vec::new())
5790        };
5791        Ok(Some(FrequencyOccurrences {
5792            omitted_max: summary.omitted_max,
5793            ordinals: summary.ordinals.clone(),
5794            anchors,
5795            anchor_indices,
5796        }))
5797    }
5798
5799    /// How many distinct values one column holds, counting a null as no value.
5800    ///
5801    /// A string column of this format is written against one dictionary that covers the whole table.
5802    /// A code is handed out the first time a value is seen and nothing ever removes one, so the
5803    /// number of codes is the number of distinct values exactly rather than an estimate. That makes
5804    /// `COUNT(DISTINCT column)` over a whole table a question the directory already knows the answer
5805    /// to, and the alternative is a hash table with a row per distinct value built from a pass over
5806    /// every row.
5807    ///
5808    /// A null in the column used to make this `None` and no longer does. A null row is written as
5809    /// the code for the empty string, so a nullable column's dictionary can hold an empty string
5810    /// that no row of it actually has, and the dictionary on its own does not say which case it is.
5811    /// The writer does know, because it counts the non-null rows that use each code on its way to
5812    /// the frequency summary, so it records how many codes any row holds and the directory carries
5813    /// that number. This reads it rather than the size of the dictionary, which also means the
5814    /// dictionary page is not opened to answer.
5815    ///
5816    /// An integer column has no dictionary, and its count comes from the set the writer keeps on its
5817    /// numeric frequency pass instead, which is exact up to a cap. `None` for a column past that cap
5818    /// and for every column that is neither, where a sketch would answer approximately and SQL asked
5819    /// for the exact number.
5820    ///
5821    /// # Errors
5822    ///
5823    /// If the column is outside the schema.
5824    pub fn distinct_values(&self, column: usize) -> Result<Option<u64>> {
5825        self.table
5826            .distincts
5827            .get(column)
5828            .copied()
5829            .ok_or_else(|| invalid("distinct column index out of range"))
5830    }
5831
5832    /// How many rows of one column are null, added up over the stripes.
5833    ///
5834    /// Every stripe records this exactly when it is written, because a null count is not a bound
5835    /// that is allowed to be wide the way a minimum and a maximum are: a filter that reads one too
5836    /// many is slow and a `COUNT` that reads one too many is wrong. Adding up a few hundred numbers
5837    /// already in memory is what makes `COUNT(column)` over a whole table free.
5838    ///
5839    /// # Errors
5840    ///
5841    /// If the column is outside the schema.
5842    pub fn null_count(&self, column: usize) -> Result<u64> {
5843        if column >= self.table.fields.len() {
5844            return Err(invalid("null count column index out of range"));
5845        }
5846        let mut nulls = 0_u64;
5847        for stripe in &self.table.stripes {
5848            let range = stripe
5849                .zone
5850                .column(column)
5851                .ok_or_else(|| invalid("stripe zone is narrower than the schema"))?;
5852            nulls = nulls
5853                .checked_add(range.nulls as u64)
5854                .ok_or_else(|| invalid("null count overflow"))?;
5855        }
5856        Ok(nulls)
5857    }
5858
5859    /// The smallest and the largest value of one string column, from the order beside its values.
5860    ///
5861    /// The dictionary holds exactly the values the column holds, so the first and the last of them
5862    /// in sorted order are the column's minimum and maximum. Two reads of a rank block settle what
5863    /// otherwise walks a million rows.
5864    ///
5865    /// `None` when the column is not a string, when the file was written before version 9 and so has
5866    /// no order, when the column has no values at all, or when it has a null in it, which is the
5867    /// placeholder again: the empty string a null is written as would sort ahead of every real
5868    /// value and be reported as the minimum.
5869    ///
5870    /// # Errors
5871    ///
5872    /// If the column is outside the schema, or a rank names a code the dictionary does not have.
5873    pub fn text_extremes(&self, column: usize) -> Result<Option<(Value, Value)>> {
5874        if self.null_count(column)? > 0 {
5875            return Ok(None);
5876        }
5877        let Some(dictionary) = self.dictionary(column)? else { return Ok(None) };
5878        let Some(ranks) = dictionary.ranks() else { return Ok(None) };
5879        if ranks == 0 {
5880            return Ok(None);
5881        }
5882        let low = text_at_rank(&dictionary, 0)?;
5883        let high = text_at_rank(&dictionary, ranks - 1)?;
5884        Ok(Some((low, high)))
5885    }
5886
5887    /// The smallest and the largest value of one column, when every stripe wrote exact ends.
5888    ///
5889    /// A stripe's ends are allowed to be wider than the truth, because a bound that rules out a
5890    /// chunk that could not match is still correct when it rules out nothing. That is what makes
5891    /// them cheap to write for a bit packed or a dictionary column, and it is also what stops them
5892    /// answering a `MIN`. So each stripe says which of the two it wrote, and this answers only when
5893    /// all of them walked their rows.
5894    ///
5895    /// `None` for a column with no ends, for an empty table, and for a column any stripe of which
5896    /// guessed. Nulls need no special case, because the ends skip them the same way `MIN` does.
5897    ///
5898    /// One case is given up on that did not have to be. A stripe merges the ends of its sixty four
5899    /// parts, and a part with no ends at all erases the merged ones, because a part whose rows are
5900    /// not covered by the stripe's ends is a stripe that would skip rows it should keep. A part of
5901    /// nothing but nulls has no rows to cover and so did not need to erase anything, but the merge
5902    /// cannot tell that part from a part whose layout it could not read. So a column with a chunk
5903    /// of nothing but nulls in the middle of it goes and reads the rows. That is slow and right,
5904    /// and the fix is a row count per part rather than anything here.
5905    ///
5906    /// # Errors
5907    ///
5908    /// If the column is outside the schema.
5909    pub fn exact_extremes(&self, column: usize) -> Result<Option<(Bound, Bound)>> {
5910        if column >= self.table.fields.len() {
5911            return Err(invalid("extremes column index out of range"));
5912        }
5913        let mut low: Option<Bound> = None;
5914        let mut high: Option<Bound> = None;
5915        for stripe in &self.table.stripes {
5916            let range = stripe
5917                .zone
5918                .column(column)
5919                .ok_or_else(|| invalid("stripe zone is narrower than the schema"))?;
5920            if !range.exact {
5921                return Ok(None);
5922            }
5923            // A stripe of nothing but nulls has no ends and says nothing about the column's, which
5924            // is why this skips it rather than giving up on the whole column. A stripe that has
5925            // rows and still has no end is a layout whose values this cannot see, and skipping that
5926            // one would answer with an end taken from the other stripes, so it gives up instead.
5927            let (Some(small), Some(large)) = (range.low.as_ref(), range.high.as_ref()) else {
5928                if stripe.rows > range.nulls {
5929                    return Ok(None);
5930                }
5931                continue;
5932            };
5933            low = Some(low.map_or_else(|| small.clone(), |held| held.smaller(small.clone())));
5934            high = Some(high.map_or_else(|| large.clone(), |held| held.larger(large.clone())));
5935        }
5936        Ok(low.zip(high))
5937    }
5938
5939    /// The sum of one integer column and how many rows went into it, when every stripe wrote one.
5940    ///
5941    /// The count beside the sum is the non-null rows, because that is what a `SUM` adds up and what
5942    /// an `AVG` divides by, and a caller that had to work it out from the row count and the null
5943    /// count would be doing the same walk twice.
5944    ///
5945    /// `None` for anything that is not an integer column, for a file written by something that did
5946    /// not record it, and when adding the stripes together would overflow.
5947    ///
5948    /// # Errors
5949    ///
5950    /// If the column is outside the schema.
5951    pub fn exact_sum(&self, column: usize) -> Result<Option<(i128, u64)>> {
5952        if column >= self.table.fields.len() {
5953            return Err(invalid("sum column index out of range"));
5954        }
5955        let mut total = 0_i128;
5956        let mut rows = 0_u64;
5957        for stripe in &self.table.stripes {
5958            let range = stripe
5959                .zone
5960                .column(column)
5961                .ok_or_else(|| invalid("stripe zone is narrower than the schema"))?;
5962            let Some(part) = range.sum else { return Ok(None) };
5963            let Some(sum) = total.checked_add(part) else { return Ok(None) };
5964            total = sum;
5965            rows = rows.saturating_add(stripe.rows as u64 - range.nulls as u64);
5966        }
5967        Ok(Some((total, rows)))
5968    }
5969
5970    /// Certified host groups over a string column, when the caller's inclusive row-count bound
5971    /// excludes every host the synopsis omitted.
5972    pub fn host_groups(
5973        &self,
5974        column: usize,
5975        minimum_count: u64,
5976    ) -> Result<Option<Vec<host::HostEntry>>> {
5977        if column >= self.table.fields.len() {
5978            return Err(invalid("host group column index out of range"));
5979        }
5980        let Some(summary) = &self.table.host_groups else { return Ok(None) };
5981        if summary.column != column || minimum_count <= summary.omitted_max {
5982            return Ok(None);
5983        }
5984        Ok(Some(summary.entries.clone()))
5985    }
5986
5987    /// The global dictionary of a column, opened once however many workers ask for it at once.
5988    ///
5989    /// The unlocked look is first because it is the answer every time after the first and it costs a
5990    /// load. Everybody who misses it queues on [`Self::loading`] and looks again on the way in, so
5991    /// the one who arrived first does the reading and the rest take what it left. Waiting is the
5992    /// cheaper thing to do: the work behind the lock is a page read, a checksum and the decode of a
5993    /// dictionary that can hold half a million entries, and the alternative is every worker of the
5994    /// scan doing all of it and all but one dropping the result on the floor.
5995    fn dictionary(&self, column: usize) -> Result<Option<Arc<Vector>>> {
5996        let Some(page) = self.table.dictionaries[column] else { return Ok(None) };
5997        if let Some(dictionary) = self.dictionaries[column].get() {
5998            return Ok(Some(Arc::clone(dictionary)));
5999        }
6000        let _queued = self.loading[column].lock().map_err(|_| invalid("a poisoned dictionary"))?;
6001        if let Some(dictionary) = self.dictionaries[column].get() {
6002            return Ok(Some(Arc::clone(dictionary)));
6003        }
6004        self.opened.fetch_add(1, Atomic::Relaxed);
6005        let dictionary = Arc::new(open_global_dictionary(
6006            Arc::clone(&self.file),
6007            page,
6008            &self.table.fields[column].ty,
6009            TEXT_KEEP_BUDGET,
6010        )?);
6011        let _ = self.dictionaries[column].set(Arc::clone(&dictionary));
6012        Ok(Some(dictionary))
6013    }
6014
6015    /// Reads one section's extent table and checks it against the entry that names it.
6016    ///
6017    /// # Errors
6018    ///
6019    /// If the entry points outside the file, the table does not checksum, or it does not decode as
6020    /// a run of extents in element order.
6021    pub fn extents(&self, of: &Section) -> Result<Vec<section::Extent>> {
6022        if of.extent_bytes == 0 {
6023            return Ok(Vec::new());
6024        }
6025        let mut bytes = vec![0; of.extent_bytes as usize];
6026        read_at(&self.file, of.extent_page, &mut bytes)?;
6027        if checksum(&bytes) != of.hash {
6028            return Err(invalid("a section's extent table does not checksum"));
6029        }
6030        let extents = section::decode_extents(&bytes)?;
6031        if extents.len() != of.extents as usize {
6032            return Err(invalid("a section's extent table is not the length the entry says"));
6033        }
6034        Ok(extents)
6035    }
6036
6037    /// Reads and verifies one extent of a section.
6038    ///
6039    /// This is what section 3.2's second rule is for. A reduction that needs one extent of a two
6040    /// gigabyte forward link reads and checksums that extent and nothing else, which is the whole
6041    /// difference between a structure that works at SF100 and issue #745.
6042    ///
6043    /// # Errors
6044    ///
6045    /// If the extent points outside the file, or its bytes do not checksum.
6046    pub fn extent(&self, of: &section::Extent) -> Result<Vec<u8>> {
6047        let end = of
6048            .offset
6049            .checked_add(u64::from(of.length))
6050            .ok_or_else(|| invalid("an extent overflows the file"))?;
6051        if of.offset < HEADER || end > self.size {
6052            return Err(invalid("an extent is outside the file"));
6053        }
6054        let mut bytes = vec![0; of.length as usize];
6055        read_at(&self.file, of.offset, &mut bytes)?;
6056        if checksum(&bytes) != of.hash {
6057            return Err(invalid("an extent does not checksum"));
6058        }
6059        Ok(bytes)
6060    }
6061
6062    /// Reads a whole section's payload, every extent of it, in order.
6063    ///
6064    /// For a structure that is resident anyway, which a key map is. Anything large enough that the
6065    /// split matters should be walking [`Reader::extents`] and taking the one it needs.
6066    ///
6067    /// # Errors
6068    ///
6069    /// If the extent table or any extent fails its check.
6070    pub fn payload(&self, of: &Section) -> Result<Vec<u8>> {
6071        let extents = self.extents(of)?;
6072        let mut bytes =
6073            Vec::with_capacity(sum(extents.iter().map(|one| u64::from(one.length))) as usize);
6074        for one in &extents {
6075            if one.first != bytes.len() as u64 {
6076                return Err(invalid("a section's extents do not join up"));
6077            }
6078            bytes.extend_from_slice(&self.extent(one)?);
6079        }
6080        // The same exception `write_section` makes: a budget record has no bytes, so its
6081        // `header_bytes` is a size rather than a header and there is nothing for it to run past.
6082        if !bytes.is_empty() && of.header_bytes as usize > bytes.len() {
6083            return Err(invalid("a section's header is longer than its payload"));
6084        }
6085        Ok(bytes)
6086    }
6087
6088    /// Reads only the named columns from one part.
6089    ///
6090    /// The whole stripe page each column lives in is read and kept, because a scan asks for the
6091    /// parts of a stripe one after another and this is what turns sixty four reads into one.
6092    ///
6093    /// # Errors
6094    ///
6095    /// If a part, column, page, or checksum is invalid.
6096    pub fn read(&self, part: usize, columns: &[usize]) -> Result<Chunk> {
6097        self.read_impl(part, columns, true, None)
6098    }
6099
6100    /// Reads named columns from one part without keeping the stripe page it came out of.
6101    ///
6102    /// This is for sparse row fetches after a selective TopN or filter, which reach a few parts of
6103    /// a stripe rather than all of them. A caller that will read most of a stripe should use
6104    /// [`Self::read`] instead, because this reads and discards the page index every time.
6105    ///
6106    /// # Errors
6107    ///
6108    /// If a part, column, page, or checksum is invalid.
6109    pub fn read_sparse(&self, part: usize, columns: &[usize]) -> Result<Chunk> {
6110        self.read_impl(part, columns, false, None)
6111    }
6112
6113    /// Reads named columns from one part, only at the rows `positions` names.
6114    ///
6115    /// For a scan that already knows which rows of the part it keeps, from the columns it read
6116    /// first. A compressed string page decompresses only those rows, and every other page is
6117    /// decoded whole and gathered, which is what reading it and narrowing it costs anyway. With
6118    /// `whole` the stripe's pages are kept the way [`Self::read`] keeps them, and without it they
6119    /// are not, the way [`Self::read_sparse`] does.
6120    ///
6121    /// # Errors
6122    ///
6123    /// If a part, column, page, or checksum is invalid, or the positions do not rise or run past
6124    /// the end of the part.
6125    pub fn read_rows(
6126        &self,
6127        part: usize,
6128        columns: &[usize],
6129        positions: &[u32],
6130        whole: bool,
6131    ) -> Result<Chunk> {
6132        self.read_impl(part, columns, whole, Some(positions))
6133    }
6134
6135    /// Whether an exact global-code membership index proves that the stripe holding a part cannot
6136    /// contain any of the sorted candidate codes.
6137    ///
6138    /// # Errors
6139    ///
6140    /// If the part, column, index page, checksum, or delta stream is invalid.
6141    pub fn skips_codes(&self, part: usize, column: usize, candidates: &[u32]) -> Result<bool> {
6142        if candidates.is_empty() {
6143            return Ok(true);
6144        }
6145        if candidates.windows(2).any(|pair| pair[0] >= pair[1]) {
6146            return Err(Error::internal("native code candidates are not sorted and unique"));
6147        }
6148        let stripe = self.stripe_of(part)?;
6149        let Some(page) = stripe.memberships.get(column) else {
6150            return Ok(false);
6151        };
6152        let mut bytes = vec![0; page.length as usize];
6153        read_at(&self.file, page.offset, &mut bytes)?;
6154        if checksum(&bytes) != page.hash {
6155            return Err(invalid("membership page checksum differs"));
6156        }
6157        let codes = decode_membership(&bytes)?;
6158        let mut left = 0;
6159        let mut right = 0;
6160        while left < codes.len() && right < candidates.len() {
6161            match codes[left].cmp(&candidates[right]) {
6162                Ordering::Less => left += 1,
6163                Ordering::Greater => right += 1,
6164                Ordering::Equal => return Ok(false),
6165            }
6166        }
6167        Ok(true)
6168    }
6169
6170    fn stripe_of(&self, part: usize) -> Result<&Stripe> {
6171        let place = self.places.get(part).ok_or_else(|| invalid("part index out of range"))?;
6172        self.table
6173            .stripes
6174            .get(place.stripe as usize)
6175            .ok_or_else(|| invalid("stripe index out of range"))
6176    }
6177
6178    /// The page index of one column of one stripe, and its page when the caller wants all of it.
6179    ///
6180    /// A scan hands parts out in order, so every worker on a column crosses into a new stripe within
6181    /// a few parts of the others and they all want the same page at the same moment. This used to
6182    /// let all of them read it, which cost the scan as many copies of every page as it had workers.
6183    /// On the full ClickBench file a `MIN(EventDate), MAX(EventDate)` moved 3.2 GB off the disk to
6184    /// look at 400 MB of column.
6185    ///
6186    /// A worker that finds the page it wants already being read neither waits for it nor reads it
6187    /// again. It comes back with the index alone, which sends [`Reader::read_impl`] down the path
6188    /// that reads the one part it came for, a few kilobytes against a quarter of a megabyte, and it
6189    /// picks the page up from the cache on its next part. Waiting would be the other way to avoid
6190    /// the duplicate read and it is worse: the pages that matter are the wide string ones, they take
6191    /// milliseconds to copy even warm, and every other worker would be stopped for all of it.
6192    ///
6193    /// The file is never read under the lock.
6194    fn held(&self, at: usize, stripe: &Stripe, column: usize, whole: bool) -> Result<CachedColumn> {
6195        let cache =
6196            self.cache.columns.get(column).ok_or_else(|| invalid("column index out of range"))?;
6197        let mut cached = cache.lock().map_err(|_| invalid("column page cache is poisoned"))?;
6198        let known = cached.index.get(at).and_then(Clone::clone);
6199        let page = cached.pages.get(at).and_then(Option::as_ref).map(|slot| {
6200            slot.used.store(true, Atomic::Relaxed);
6201            Arc::clone(&slot.page)
6202        });
6203        if let Some(index) = known.clone() {
6204            if !whole || page.is_some() {
6205                return Ok(CachedColumn { stripe: at, index, page });
6206            }
6207        }
6208        if cached.loading.contains(&at) {
6209            drop(cached);
6210            // The index is almost always already here, because somebody read this stripe to get
6211            // into the loading list in the first place, so this branch usually costs no read at
6212            // all and the one part read in `read_impl` is all the losing worker pays for.
6213            if let Some(index) = known {
6214                return Ok(CachedColumn { stripe: at, index, page: None });
6215            }
6216            let held = self.page_of(stripe, column, at, false, None)?;
6217            let mut cached = cache.lock().map_err(|_| invalid("column page cache is poisoned"))?;
6218            remember(&mut cached, &held);
6219            return Ok(held);
6220        }
6221        cached.loading.push(at);
6222        drop(cached);
6223
6224        let read = self.page_of(stripe, column, at, whole, known);
6225
6226        // The stripe leaves the loading list and its page enters the cache under one lock. Doing
6227        // them separately would leave a moment where another worker sees neither and reads the
6228        // page a second time, which is the whole thing this is here to stop.
6229        let mut cached = cache.lock().map_err(|_| invalid("column page cache is poisoned"))?;
6230        if let Some(position) = cached.loading.iter().position(|loading| *loading == at) {
6231            cached.loading.remove(position);
6232        }
6233        let held = read?;
6234        let taken = remember(&mut cached, &held);
6235        drop(cached);
6236        if let Some((bytes, used)) = taken {
6237            self.cache.held[column].fetch_add(1, Atomic::Relaxed);
6238            self.pool.admit(Held {
6239                shelf: Arc::downgrade(&self.cache),
6240                column,
6241                stripe: at,
6242                bytes,
6243                used,
6244            });
6245        }
6246        Ok(held)
6247    }
6248
6249    /// Reads one stripe's index for a column, and its page when the caller wants all of it.
6250    ///
6251    /// `known` is the index when the reader has already read it, which after the first worker
6252    /// through a stripe it always has, because [`remember`] keeps every index for the life of the
6253    /// reader. Without that a scan reads the index again on every part that misses the page cache.
6254    fn page_of(
6255        &self,
6256        stripe: &Stripe,
6257        column: usize,
6258        at: usize,
6259        whole: bool,
6260        known: Option<Arc<Vec<PartSpan>>>,
6261    ) -> Result<CachedColumn> {
6262        let index = match known {
6263            Some(index) => index,
6264            None => {
6265                self.indexes.fetch_add(1, Atomic::Relaxed);
6266                Arc::new(read_index(&self.file, stripe, column)?)
6267            }
6268        };
6269        let page = if whole {
6270            self.pages.fetch_add(1, Atomic::Relaxed);
6271            let span = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
6272            let mut bytes = vec![0; span.length as usize];
6273            read_at(&self.file, span.offset, &mut bytes)?;
6274            Some(Arc::new(bytes))
6275        } else {
6276            None
6277        };
6278        Ok(CachedColumn { stripe: at, index, page })
6279    }
6280
6281    fn read_impl(
6282        &self,
6283        at: usize,
6284        columns: &[usize],
6285        whole: bool,
6286        positions: Option<&[u32]>,
6287    ) -> Result<Chunk> {
6288        let place = *self.places.get(at).ok_or_else(|| invalid("part index out of range"))?;
6289        let index = place.stripe as usize;
6290        let stripe =
6291            self.table.stripes.get(index).ok_or_else(|| invalid("stripe index out of range"))?;
6292        let rows = place.rows as usize;
6293        let mut picked = Vec::with_capacity(columns.len());
6294        for &column in columns {
6295            let field = self
6296                .table
6297                .fields
6298                .get(column)
6299                .ok_or_else(|| invalid("column index out of range"))?;
6300            let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
6301            let held = self.held(index, stripe, column, whole)?;
6302            let span = *held
6303                .index
6304                .get(place.part as usize)
6305                .ok_or_else(|| invalid("part index out of range"))?;
6306            let owned;
6307            let bytes = match &held.page {
6308                Some(held) => part_bytes(held, span)?,
6309                None => {
6310                    let offset = page
6311                        .offset
6312                        .checked_add(span.start as u64)
6313                        .ok_or_else(|| invalid("part range overflow"))?;
6314                    let mut bytes = vec![0; span.length];
6315                    read_at(&self.file, offset, &mut bytes)?;
6316                    owned = bytes;
6317                    &owned
6318                }
6319            };
6320            if checksum(bytes) != span.hash {
6321                return Err(invalid(&format!(
6322                    "column page checksum differs, column {column} part {} at {}+{} of {} bytes, \
6323                     wanted {:016x} and got {:016x}",
6324                    place.part,
6325                    page.offset,
6326                    span.start,
6327                    span.length,
6328                    span.hash,
6329                    checksum(bytes),
6330                )));
6331            }
6332            let dictionary = self.dictionary(column)?;
6333            // Held as a page, because a column that came out of a file is handed out more than
6334            // once. A group by clones its key columns out of the chunk so the keys outlive it, a
6335            // projection of a bare column name does the same, and a cut of a flat run copies unless
6336            // the run is a page. One `Arc` per column per part buys all of those, and it moves the
6337            // run into the `Arc` without touching a value.
6338            let vector = match positions {
6339                None => decode(&field.ty, rows, bytes, dictionary)?,
6340                Some(positions) => decode_at(&field.ty, rows, bytes, dictionary, positions)?,
6341            };
6342            picked.push(vector.into_pages());
6343        }
6344        Chunk::with_rows(picked, positions.map_or(rows, <[u32]>::len))
6345    }
6346
6347    /// Whether persisted statistics prove that a part cannot match the predicates.
6348    ///
6349    /// Three of them, asked cheapest first.
6350    ///
6351    /// The stripe's bounds are in memory already, so they are free, and they are also the coarsest:
6352    /// every part of a stripe gets the same answer and a scan that skips one part that way skips all
6353    /// sixty four. Then the part's own bounds, which are a read of one page per column per stripe
6354    /// and are sixty four times finer. Then the sieves, which are per part and answer equality, the
6355    /// test bounds are worst at: a column of identifiers has every stripe and nearly every part
6356    /// covering the whole of its type, so bounds keep them all and the sieve keeps the ones that
6357    /// really hold the value.
6358    ///
6359    /// The middle one is what an ordered comparison on a column the rows are not sorted by needs. On
6360    /// ClickBench 24 the stripe bounds leave eight stripes of sixteen alive, which is half the file,
6361    /// and the part bounds leave thirty parts of nine hundred and seventy four.
6362    #[must_use]
6363    pub fn skips(&self, part: usize, probes: &[Probe]) -> bool {
6364        let Some(place) = self.places.get(part).copied() else { return false };
6365        let Some(stripe) = self.table.stripes.get(place.stripe as usize) else { return false };
6366        if stripe.zone.skips(probes) {
6367            return true;
6368        }
6369        probes.iter().any(|probe| self.outside(place, probe) || self.sifted(place, probe))
6370    }
6371
6372    /// Whether the bounds of one part rule out one probe.
6373    ///
6374    /// The part's own two ends, which are narrower than the stripe's and cost a page read the first
6375    /// time this is asked about a column. A column with no page here answers `false`, which is the
6376    /// answer a caller got before there were any.
6377    fn outside(&self, place: Place, probe: &Probe) -> bool {
6378        match self.stripe_part_ranges(place.stripe as usize, probe.column) {
6379            Some(ranges) => ranges
6380                .get(place.part as usize)
6381                .is_some_and(|range| range.excludes(probe.op, &probe.value)),
6382            None => false,
6383        }
6384    }
6385
6386    /// The per part ranges of one stripe of one column, read once and kept.
6387    ///
6388    /// `None` when the column has no page in that stripe and when the page is damaged, on the same
6389    /// reasoning as the sieves: this is an index over data that is still there, so a caller that
6390    /// cannot read one reads the rows and gets the right answer slowly.
6391    fn stripe_part_ranges(&self, stripe: usize, column: usize) -> Option<&[Range]> {
6392        let slot = self.part_ranges.get(column)?.get(stripe)?;
6393        if let Some(held) = slot.get() {
6394            return Some(held);
6395        }
6396        let page = self.table.stripes.get(stripe)?.part_ranges.get(column)?;
6397        let mut bytes = vec![0; page.length as usize];
6398        read_at(&self.file, page.offset, &mut bytes).ok()?;
6399        if checksum(&bytes) != page.hash {
6400            return None;
6401        }
6402        let ranges = Arc::new(decode_part_ranges(&bytes).ok()?);
6403        let _ = slot.set(ranges);
6404        slot.get().map(|held| held.as_slice())
6405    }
6406
6407    /// Whether persisted statistics prove that every row of a part matches the predicates.
6408    ///
6409    /// Only the bounds. The sieves say nothing here, because a sieve that holds a value is a sieve
6410    /// that may be holding somebody else's hash, so it can rule a part out and can never wave one
6411    /// through.
6412    ///
6413    /// The stripe first and the part after it, the same two steps and in the same order as
6414    /// [`Self::skips`]. The stripe's bounds are in memory already and its null count covers sixty
6415    /// four parts rather than one, so a stripe that answers is an answer for nothing, and the part's
6416    /// own bounds are only read for the probes it could not settle. Both directions are safe: a
6417    /// stretch where everything passes contains no narrower stretch where something fails, and a
6418    /// stripe with no nulls has no nulls in any of its parts.
6419    ///
6420    /// A string end a part recorded is cut down to its first few bytes, so a part's stretch can be
6421    /// wider than its rows really are as well. That is the same safe direction for the same reason,
6422    /// and it is why this asks the two ends rather than anything `exact` says.
6423    #[must_use]
6424    pub fn certain(&self, part: usize, probes: &[Probe]) -> bool {
6425        let Some(place) = self.places.get(part).copied() else { return false };
6426        let Some(stripe) = self.table.stripes.get(place.stripe as usize) else { return false };
6427        if stripe.zone.certain(probes) {
6428            return true;
6429        }
6430        probes
6431            .iter()
6432            .all(|probe| stripe.zone.certain(slice::from_ref(probe)) || self.inside(place, probe))
6433    }
6434
6435    /// Whether one part's own two ends prove that every row of it passes `probe`.
6436    ///
6437    /// The mirror of [`Self::outside`], reading the same page. `false` for a part whose stripe wrote
6438    /// no range page, which is a stripe of one part, because there the stripe's own bounds are the
6439    /// part's and the caller has already asked them.
6440    fn inside(&self, place: Place, probe: &Probe) -> bool {
6441        match self.stripe_part_ranges(place.stripe as usize, probe.column) {
6442            Some(ranges) => ranges
6443                .get(place.part as usize)
6444                .is_some_and(|range| range.certain(probe.op, &probe.value)),
6445            None => false,
6446        }
6447    }
6448
6449    /// Whether the bounds of one stripe prove that none of its parts can match the predicates.
6450    ///
6451    /// The cheap half of [`Self::skips`], asked about a whole stripe at once. The bounds live in the
6452    /// directory and are already in memory, so this answers without touching the file, and that is
6453    /// the reason it is worth having on its own: a caller that wants to know roughly where the work
6454    /// is before it starts any workers can ask this about sixteen stripes for nothing, where asking
6455    /// [`Self::skips`] about nine hundred parts would read and decode a sieve page per stripe first.
6456    ///
6457    /// It keeps stripes that [`Self::skips`] would rule out part by part, which is the right way for
6458    /// it to be wrong: the parts are still checked when they are read.
6459    #[must_use]
6460    pub fn stripe_skips(&self, stripe: usize, probes: &[Probe]) -> bool {
6461        self.table.stripes.get(stripe).is_some_and(|held| held.zone.skips(probes))
6462    }
6463
6464    /// Whether the sieve of one part rules out one probe.
6465    ///
6466    /// Only equality. An ordered comparison is what the bounds are for and a sieve says nothing
6467    /// about it, and a read that cannot answer keeps the part, which is the answer a caller with no
6468    /// sieve gets anyway.
6469    fn sifted(&self, place: Place, probe: &Probe) -> bool {
6470        if probe.op != Op::Equal {
6471            return false;
6472        }
6473        match self.stripe_sieves(place.stripe as usize, probe.column) {
6474            Some(sieves) => sieves
6475                .get(place.part as usize)
6476                .and_then(Option::as_ref)
6477                .is_some_and(|sieve| sieve.excludes(&probe.value)),
6478            None => false,
6479        }
6480    }
6481
6482    /// The sieves of one stripe of one column, read once and kept.
6483    ///
6484    /// `None` when the column has no sieves in that stripe, when the page is damaged, and when the
6485    /// bytes are not a page this version can read. A sieve is an index over data that is still there
6486    /// and a caller that cannot read one reads the rows, so this is the one place in the file where
6487    /// a bad checksum is a slow query rather than an error.
6488    fn stripe_sieves(&self, stripe: usize, column: usize) -> Option<&[Option<Sieve>]> {
6489        let slot = self.sieves.get(column)?.get(stripe)?;
6490        if let Some(held) = slot.get() {
6491            return Some(held);
6492        }
6493        let page = self.table.stripes.get(stripe)?.sieves.get(column)?;
6494        let mut bytes = vec![0; page.length as usize];
6495        read_at(&self.file, page.offset, &mut bytes).ok()?;
6496        if checksum(&bytes) != page.hash {
6497            return None;
6498        }
6499        let sieves = Arc::new(decode_sieves(&bytes).ok()?);
6500        let _ = slot.set(sieves);
6501        slot.get().map(|held| held.as_slice())
6502    }
6503}
6504
6505/// The value sitting at one position of a dictionary's sorted order.
6506fn text_at_rank(dictionary: &Vector, rank: usize) -> Result<Value> {
6507    let code = dictionary.code_at_rank(rank)? as usize;
6508    let text = dictionary
6509        .try_text_at(code)?
6510        .ok_or_else(|| invalid("global dictionary order names a code it does not have"))?;
6511    Ok(Value::Varchar(text.into()))
6512}
6513
6514/// Writes one span of a file at an offset, without depending on where the cursor is.
6515///
6516/// The writer owns an offset of its own and passes it in here, so that nothing it writes depends on
6517/// a cursor that a read is entitled to move. Both of these can come back short and both loop.
6518#[cfg(unix)]
6519fn write_at(file: &File, mut offset: u64, mut bytes: &[u8]) -> Result<()> {
6520    use std::os::unix::fs::FileExt;
6521    while !bytes.is_empty() {
6522        let written = file.write_at(bytes, offset).map_err(io)?;
6523        if written == 0 {
6524            return Err(invalid("a write to the native file wrote nothing"));
6525        }
6526        offset += written as u64;
6527        bytes = &bytes[written..];
6528    }
6529    Ok(())
6530}
6531
6532/// The same write, on the call Windows spells differently.
6533#[cfg(windows)]
6534fn write_at(file: &File, mut offset: u64, mut bytes: &[u8]) -> Result<()> {
6535    use std::os::windows::fs::FileExt;
6536    while !bytes.is_empty() {
6537        let written = file.seek_write(bytes, offset).map_err(io)?;
6538        if written == 0 {
6539            return Err(invalid("a write to the native file wrote nothing"));
6540        }
6541        offset += written as u64;
6542        bytes = &bytes[written..];
6543    }
6544    Ok(())
6545}
6546
6547/// Somewhere that is neither, where the cursor is all there is.
6548#[cfg(not(any(unix, windows)))]
6549fn write_at(file: &File, offset: u64, bytes: &[u8]) -> Result<()> {
6550    use std::io::Write;
6551    let mut file = file.try_clone().map_err(io)?;
6552    file.seek(SeekFrom::Start(offset)).map_err(io)?;
6553    file.write_all(bytes).map_err(io)
6554}
6555
6556/// Reads one span of a file at an offset, without moving a cursor anybody else can see.
6557///
6558/// Every reader of a table shares one [`File`] behind an [`Arc`], and a grouped aggregate reads its
6559/// pages from several threads at once, so this has to be positional. Seeking and then reading is
6560/// two calls with a gap in the middle, and in that gap another thread's seek lands and the read
6561/// comes back with somebody else's bytes.
6562///
6563/// Both of these can come back short, so both loop. A read of zero bytes before the span is filled
6564/// means the file stops earlier than the directory said it does.
6565#[cfg(unix)]
6566fn read_at(file: &File, mut offset: u64, mut bytes: &mut [u8]) -> Result<()> {
6567    use std::os::unix::fs::FileExt;
6568    while !bytes.is_empty() {
6569        let read = file.read_at(bytes, offset).map_err(io)?;
6570        if read == 0 {
6571            return Err(invalid("column page ends before its declared length"));
6572        }
6573        offset += read as u64;
6574        bytes = &mut bytes[read..];
6575    }
6576    Ok(())
6577}
6578
6579/// The same read, on the call Windows spells differently.
6580///
6581/// `seek_read` is one `ReadFile` carrying the offset with it, so two of them cannot interleave the
6582/// way a seek and a read can. It does leave the shared cursor somewhere afterwards, which is why
6583/// nothing in this file may read that cursor.
6584#[cfg(windows)]
6585fn read_at(file: &File, mut offset: u64, mut bytes: &mut [u8]) -> Result<()> {
6586    use std::os::windows::fs::FileExt;
6587    while !bytes.is_empty() {
6588        let read = file.seek_read(bytes, offset).map_err(io)?;
6589        if read == 0 {
6590            return Err(invalid("column page ends before its declared length"));
6591        }
6592        offset += read as u64;
6593        bytes = &mut bytes[read..];
6594    }
6595    Ok(())
6596}
6597
6598/// Somewhere that is neither, where the cursor is all there is.
6599///
6600/// This one does race, and there is no way to write it so it does not. Nothing we build for runs
6601/// here, so it exists to keep the crate compiling rather than to be correct under threads.
6602#[cfg(not(any(unix, windows)))]
6603fn read_at(file: &File, offset: u64, bytes: &mut [u8]) -> Result<()> {
6604    let mut file = file.try_clone().map_err(io)?;
6605    file.seek(SeekFrom::Start(offset)).map_err(io)?;
6606    file.read_exact(bytes).map_err(io)
6607}
6608
6609/// What a column type is called in the directory.
6610///
6611/// A tag is a number in a file somebody else wrote, so a tag that has been used is used forever and
6612/// the only thing that may happen to this list is that it grows. 1 to 13 are the tags the format
6613/// had when it could store thirteen types, and 14 to 27 are the rest, in the order they were added
6614/// rather than in an order that means anything.
6615fn type_tag(ty: &LogicalType) -> Result<u8> {
6616    match ty {
6617        LogicalType::SmallInt => Ok(1),
6618        LogicalType::Integer => Ok(2),
6619        LogicalType::BigInt => Ok(3),
6620        LogicalType::Varchar => Ok(4),
6621        LogicalType::Date => Ok(5),
6622        LogicalType::Timestamp => Ok(6),
6623        LogicalType::Boolean => Ok(7),
6624        LogicalType::TinyInt => Ok(8),
6625        LogicalType::UTinyInt => Ok(9),
6626        LogicalType::USmallInt => Ok(10),
6627        LogicalType::UInteger => Ok(11),
6628        LogicalType::UBigInt => Ok(12),
6629        LogicalType::Decimal { .. } => Ok(13),
6630        LogicalType::Float => Ok(14),
6631        LogicalType::Double => Ok(15),
6632        LogicalType::HugeInt => Ok(16),
6633        LogicalType::UHugeInt => Ok(17),
6634        LogicalType::Time => Ok(18),
6635        LogicalType::TimeTz => Ok(19),
6636        LogicalType::TimestampTz => Ok(20),
6637        LogicalType::Interval => Ok(21),
6638        LogicalType::Uuid => Ok(22),
6639        LogicalType::Blob => Ok(23),
6640        LogicalType::Bit => Ok(24),
6641        LogicalType::TimestampS => Ok(25),
6642        LogicalType::TimestampMs => Ok(26),
6643        LogicalType::TimestampNs => Ok(27),
6644        _ => Err(Error::not_implemented(format!("native storage for {ty}"))),
6645    }
6646}
6647
6648/// The tag of a column type, and the parameters of the ones that have any.
6649///
6650/// Only `DECIMAL` has parameters today. Width and scale go after the tag rather than into it
6651/// because they are what says how wide a value is on disk, and a reader that guessed would read the
6652/// wrong number of bytes per row rather than the wrong number of digits.
6653fn put_type(out: &mut Vec<u8>, ty: &LogicalType) -> Result<()> {
6654    out.push(type_tag(ty)?);
6655    if let LogicalType::Decimal { width, scale } = ty {
6656        out.push(*width);
6657        out.push(*scale);
6658    }
6659    Ok(())
6660}
6661
6662/// The other half of [`put_type`], reading the parameters the tag says are there.
6663fn read_type(cur: &mut Cursor<'_>) -> Result<LogicalType> {
6664    let tag = cur.u8()?;
6665    if tag == 13 {
6666        let width = cur.u8()?;
6667        let scale = cur.u8()?;
6668        return LogicalType::decimal(width, scale)
6669            .map_err(|_| invalid("decimal column width and scale are not a decimal"));
6670    }
6671    tag_type(tag)
6672}
6673
6674fn tag_type(tag: u8) -> Result<LogicalType> {
6675    match tag {
6676        1 => Ok(LogicalType::SmallInt),
6677        2 => Ok(LogicalType::Integer),
6678        3 => Ok(LogicalType::BigInt),
6679        4 => Ok(LogicalType::Varchar),
6680        5 => Ok(LogicalType::Date),
6681        6 => Ok(LogicalType::Timestamp),
6682        7 => Ok(LogicalType::Boolean),
6683        8 => Ok(LogicalType::TinyInt),
6684        9 => Ok(LogicalType::UTinyInt),
6685        10 => Ok(LogicalType::USmallInt),
6686        11 => Ok(LogicalType::UInteger),
6687        12 => Ok(LogicalType::UBigInt),
6688        14 => Ok(LogicalType::Float),
6689        15 => Ok(LogicalType::Double),
6690        16 => Ok(LogicalType::HugeInt),
6691        17 => Ok(LogicalType::UHugeInt),
6692        18 => Ok(LogicalType::Time),
6693        19 => Ok(LogicalType::TimeTz),
6694        20 => Ok(LogicalType::TimestampTz),
6695        21 => Ok(LogicalType::Interval),
6696        22 => Ok(LogicalType::Uuid),
6697        23 => Ok(LogicalType::Blob),
6698        24 => Ok(LogicalType::Bit),
6699        25 => Ok(LogicalType::TimestampS),
6700        26 => Ok(LogicalType::TimestampMs),
6701        27 => Ok(LogicalType::TimestampNs),
6702        _ => Err(invalid("column type tag is unknown")),
6703    }
6704}
6705
6706fn put_u16(out: &mut Vec<u8>, value: u16) {
6707    out.extend_from_slice(&value.to_le_bytes());
6708}
6709fn put_u32(out: &mut Vec<u8>, value: u32) {
6710    out.extend_from_slice(&value.to_le_bytes());
6711}
6712fn put_u64(out: &mut Vec<u8>, value: u64) {
6713    out.extend_from_slice(&value.to_le_bytes());
6714}
6715fn put_var_u64(out: &mut Vec<u8>, mut value: u64) {
6716    while value >= 0x80 {
6717        out.push((value as u8 & 0x7f) | 0x80);
6718        value >>= 7;
6719    }
6720    out.push(value as u8);
6721}
6722
6723fn frequency_order(left: FrequencyValue, right: FrequencyValue) -> Ordering {
6724    match (left, right) {
6725        (FrequencyValue::Null, FrequencyValue::Null) => Ordering::Equal,
6726        (FrequencyValue::Null, _) => Ordering::Less,
6727        (_, FrequencyValue::Null) => Ordering::Greater,
6728        (FrequencyValue::Integer(left), FrequencyValue::Integer(right)) => left.cmp(&right),
6729        (FrequencyValue::Code(left), FrequencyValue::Code(right)) => left.cmp(&right),
6730        (FrequencyValue::Integer(_), FrequencyValue::Code(_)) => Ordering::Less,
6731        (FrequencyValue::Code(_), FrequencyValue::Integer(_)) => Ordering::Greater,
6732    }
6733}
6734
6735/// Leaves the [`FREQUENCY_ENTRIES`] commonest entries in order and says what the next one counted.
6736///
6737/// There is one entry a distinct value, so on `URL` this is handed two and a quarter million of
6738/// them and keeps five hundred and twelve. Sorting all of them to throw almost all of them away is
6739/// the whole of what counting a dictionary column used to cost, 2.13 seconds of it on `URL` at eight
6740/// million rows against 11.93 for compressing the same column's values.
6741///
6742/// Partitioning answers both questions instead. It puts the five hundred and thirteenth entry where
6743/// it belongs and everything commoner in front of it, which is the entries to keep and the count to
6744/// report as the largest one omitted, and then only the part that survives is sorted. The order that
6745/// comes out is the order the sort gave, because the tie break makes the comparison total: two
6746/// entries never hold the same value.
6747fn keep_most_frequent(entries: &mut Vec<FrequencyEntry>) -> u64 {
6748    let order = |left: &FrequencyEntry, right: &FrequencyEntry| {
6749        right.count.cmp(&left.count).then_with(|| frequency_order(left.value, right.value))
6750    };
6751    let omitted_max = if entries.len() > FREQUENCY_ENTRIES {
6752        let (_, next, _) = entries.select_nth_unstable_by(FREQUENCY_ENTRIES, order);
6753        let omitted_max = next.count;
6754        entries.truncate(FREQUENCY_ENTRIES);
6755        omitted_max
6756    } else {
6757        0
6758    };
6759    entries.sort_unstable_by(order);
6760    omitted_max
6761}
6762
6763fn code_frequency(
6764    dictionary: &GlobalDictionary,
6765    flat: &[u8],
6766    bases: &[u64],
6767) -> Result<(FrequencySummary, Vec<Option<Vec<u8>>>)> {
6768    let mut entries = dictionary
6769        .counts
6770        .iter()
6771        .enumerate()
6772        .filter(|(_, count)| **count != 0)
6773        .map(|(code, &count)| FrequencyEntry { value: FrequencyValue::Code(code as u32), count })
6774        .collect::<Vec<_>>();
6775    if dictionary.nulls != 0 {
6776        entries.push(FrequencyEntry { value: FrequencyValue::Null, count: dictionary.nulls });
6777    }
6778    let omitted_max = keep_most_frequent(&mut entries);
6779    let mut spans = Vec::with_capacity(entries.len());
6780    let mut text_bytes = 0_usize;
6781    for entry in &entries {
6782        let span = match entry.value {
6783            FrequencyValue::Code(code) => {
6784                let span = GlobalDictionary::value_span(&dictionary.ends, bases, code as usize);
6785                let bytes = flat
6786                    .get(span.0..span.1)
6787                    .ok_or_else(|| invalid("a frequency code is outside its dictionary"))?;
6788                text_bytes = text_bytes.saturating_add(bytes.len());
6789                Some(span)
6790            }
6791            FrequencyValue::Null | FrequencyValue::Integer(_) => None,
6792        };
6793        spans.push(span);
6794    }
6795    let texts = if text_bytes > FREQUENCY_TEXT_BUDGET {
6796        Vec::new()
6797    } else {
6798        spans.into_iter().map(|span| span.map(|(from, to)| flat[from..to].to_vec())).collect()
6799    };
6800    Ok((
6801        FrequencySummary {
6802            entries,
6803            omitted_max,
6804            ordinals: Vec::new(),
6805            ordinal_entries: Vec::new(),
6806        },
6807        texts,
6808    ))
6809}
6810
6811fn encode_directory(table: &Table) -> Result<Vec<u8>> {
6812    let mut out = DIRECTORY.to_vec();
6813    let name = table.name.as_bytes();
6814    put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("table name too long"))?);
6815    out.extend_from_slice(name);
6816    put_u16(&mut out, u16::try_from(table.fields.len()).map_err(|_| invalid("too many columns"))?);
6817    for field in &table.fields {
6818        let name = field.name.as_bytes();
6819        put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("column name too long"))?);
6820        out.extend_from_slice(name);
6821        put_type(&mut out, &field.ty)?;
6822        out.push(u8::from(field.not_null));
6823    }
6824    for dictionary in &table.dictionaries {
6825        match dictionary {
6826            None => out.push(0),
6827            Some(page) => {
6828                out.push(1);
6829                put_u64(&mut out, page.offset);
6830                put_u32(&mut out, page.length);
6831                put_u64(&mut out, page.hash);
6832            }
6833        }
6834    }
6835    for distinct in &table.distincts {
6836        match distinct {
6837            None => out.push(0),
6838            Some(count) => {
6839                out.push(1);
6840                put_u64(&mut out, *count);
6841            }
6842        }
6843    }
6844    put_u64(&mut out, u64::try_from(table.rows).map_err(|_| invalid("row count overflow"))?);
6845    put_u32(&mut out, u32::try_from(table.stripes.len()).map_err(|_| invalid("too many stripes"))?);
6846    for stripe in &table.stripes {
6847        put_u32(
6848            &mut out,
6849            u32::try_from(stripe.parts.len()).map_err(|_| invalid("too many parts in a stripe"))?,
6850        );
6851        for &rows in &stripe.parts {
6852            put_u32(&mut out, rows);
6853        }
6854        put_u64(&mut out, stripe.index.offset);
6855        put_u32(&mut out, stripe.index.length);
6856        for page in &stripe.pages {
6857            put_u64(&mut out, page.offset);
6858            put_u32(&mut out, page.length);
6859        }
6860        // A membership index says which of a dictionary's codes a part holds, so a column the writer
6861        // decided against giving a dictionary has nothing for it to be about and writes none. Every
6862        // file written before that decision existed has a dictionary on every varchar column, so
6863        // this reads those files byte for byte the way it always did.
6864        for ((field, dictionary), membership) in
6865            table.fields.iter().zip(&table.dictionaries).zip(stripe.memberships.slots())
6866        {
6867            if field.ty != LogicalType::Varchar || dictionary.is_none() {
6868                continue;
6869            }
6870            let page =
6871                membership.ok_or_else(|| invalid("string page has no code membership index"))?;
6872            put_u64(&mut out, page.offset);
6873            put_u32(&mut out, page.length);
6874            put_u64(&mut out, page.hash);
6875        }
6876        for sieve in stripe.sieves.slots() {
6877            match sieve {
6878                None => out.push(0),
6879                Some(page) => {
6880                    out.push(1);
6881                    put_u64(&mut out, page.offset);
6882                    put_u32(&mut out, page.length);
6883                    put_u64(&mut out, page.hash);
6884                }
6885            }
6886        }
6887        for held in stripe.part_ranges.slots() {
6888            match held {
6889                None => out.push(0),
6890                Some(page) => {
6891                    out.push(1);
6892                    put_u64(&mut out, page.offset);
6893                    put_u32(&mut out, page.length);
6894                    put_u64(&mut out, page.hash);
6895                }
6896            }
6897        }
6898        for range in stripe.zone.columns() {
6899            put_bound(&mut out, range.low.as_ref())?;
6900            put_bound(&mut out, range.high.as_ref())?;
6901            put_u32(
6902                &mut out,
6903                u32::try_from(range.nulls).map_err(|_| invalid("null count overflow"))?,
6904            );
6905            out.push(u8::from(range.exact));
6906            match range.sum {
6907                None => out.push(0),
6908                Some(total) => {
6909                    out.push(1);
6910                    out.extend_from_slice(&total.to_le_bytes());
6911                }
6912            }
6913        }
6914    }
6915    out.extend_from_slice(FREQUENCIES);
6916    put_u16(
6917        &mut out,
6918        u16::try_from(table.frequencies.len())
6919            .map_err(|_| invalid("too many frequency columns"))?,
6920    );
6921    for summary in &table.frequencies {
6922        let summary = match summary {
6923            None => {
6924                out.push(0);
6925                continue;
6926            }
6927            Some(Frequencies::Held(summary)) => summary,
6928            // Only a reader leaves a synopsis in the file, and nothing writes a reader's table back.
6929            Some(Frequencies::Stored { .. }) => {
6930                return Err(invalid("a synopsis left in the file cannot be written back"));
6931            }
6932        };
6933        out.push(1);
6934        put_u64(&mut out, summary.omitted_max);
6935        put_u32(
6936            &mut out,
6937            u32::try_from(summary.entries.len())
6938                .map_err(|_| invalid("too many frequency entries"))?,
6939        );
6940        for entry in &summary.entries {
6941            match entry.value {
6942                FrequencyValue::Null => out.push(0),
6943                FrequencyValue::Integer(value) => {
6944                    out.push(1);
6945                    out.extend_from_slice(&value.to_le_bytes());
6946                }
6947                FrequencyValue::Code(value) => {
6948                    out.push(2);
6949                    put_u32(&mut out, value);
6950                }
6951            }
6952            put_u64(&mut out, entry.count);
6953        }
6954        put_u32(
6955            &mut out,
6956            u32::try_from(summary.ordinals.len())
6957                .map_err(|_| invalid("too many frequency ordinals"))?,
6958        );
6959        let mut previous = 0_u64;
6960        for (at, &ordinal) in summary.ordinals.iter().enumerate() {
6961            let delta = if at == 0 {
6962                ordinal
6963            } else {
6964                ordinal
6965                    .checked_sub(previous)
6966                    .ok_or_else(|| invalid("frequency ordinals are not ordered"))?
6967            };
6968            if at != 0 && delta == 0 {
6969                return Err(invalid("frequency ordinals are not unique"));
6970            }
6971            put_var_u64(&mut out, delta);
6972            previous = ordinal;
6973        }
6974        if summary.ordinal_entries.len() != summary.ordinals.len() {
6975            return Err(invalid("frequency ordinal values have a different length"));
6976        }
6977        for &entry in &summary.ordinal_entries {
6978            if entry as usize >= summary.entries.len() {
6979                return Err(invalid("frequency ordinal value is outside its entries"));
6980            }
6981            put_u16(&mut out, entry);
6982        }
6983    }
6984    if !table.pair_frequencies.is_empty() {
6985        out.extend_from_slice(PAIR_FREQUENCIES);
6986        put_u16(
6987            &mut out,
6988            u16::try_from(table.pair_frequencies.len())
6989                .map_err(|_| invalid("too many pair frequency summaries"))?,
6990        );
6991        for summary in &table.pair_frequencies {
6992            put_u16(&mut out, summary.first);
6993            put_u16(&mut out, summary.second);
6994            put_u64(&mut out, summary.omitted_max);
6995            put_u16(
6996                &mut out,
6997                u16::try_from(summary.entries.len())
6998                    .map_err(|_| invalid("too many pair frequency entries"))?,
6999            );
7000            for entry in &summary.entries {
7001                put_u16(&mut out, entry.first_entry);
7002                match entry.second {
7003                    None => out.push(0),
7004                    Some(code) => {
7005                        out.push(1);
7006                        put_u32(&mut out, code);
7007                    }
7008                }
7009                put_u64(&mut out, entry.count);
7010            }
7011        }
7012    }
7013    let text_columns = table.frequency_texts.iter().filter(|texts| !texts.is_empty()).count();
7014    if text_columns != 0 {
7015        out.extend_from_slice(FREQUENCY_TEXTS);
7016        put_u16(
7017            &mut out,
7018            u16::try_from(text_columns)
7019                .map_err(|_| invalid("too many string frequency columns"))?,
7020        );
7021        for (column, texts) in table.frequency_texts.iter().enumerate() {
7022            if texts.is_empty() {
7023                continue;
7024            }
7025            put_u16(
7026                &mut out,
7027                u16::try_from(column).map_err(|_| invalid("frequency text column overflows"))?,
7028            );
7029            put_u16(
7030                &mut out,
7031                u16::try_from(texts.len())
7032                    .map_err(|_| invalid("too many frequency text entries"))?,
7033            );
7034            for text in texts {
7035                match text {
7036                    None => out.push(0),
7037                    Some(text) => {
7038                        out.push(1);
7039                        put_u32(
7040                            &mut out,
7041                            u32::try_from(text.len())
7042                                .map_err(|_| invalid("frequency text is too long"))?,
7043                        );
7044                        out.extend_from_slice(text);
7045                    }
7046                }
7047            }
7048        }
7049    }
7050    if let Some(summary) = &table.host_groups {
7051        out.extend_from_slice(HOST_GROUPS);
7052        put_u16(
7053            &mut out,
7054            u16::try_from(summary.column).map_err(|_| invalid("host column overflows"))?,
7055        );
7056        put_u64(&mut out, summary.omitted_max);
7057        put_u16(
7058            &mut out,
7059            u16::try_from(summary.entries.len()).map_err(|_| invalid("too many host groups"))?,
7060        );
7061        for entry in &summary.entries {
7062            put_u32(
7063                &mut out,
7064                u32::try_from(entry.host.len()).map_err(|_| invalid("host name is too long"))?,
7065            );
7066            out.extend_from_slice(entry.host.as_bytes());
7067            put_u64(&mut out, entry.count);
7068            out.extend_from_slice(&entry.bytes_sum.to_le_bytes());
7069            put_u32(
7070                &mut out,
7071                u32::try_from(entry.minimum.len())
7072                    .map_err(|_| invalid("host minimum is too long"))?,
7073            );
7074            out.extend_from_slice(entry.minimum.as_bytes());
7075        }
7076    }
7077    // Written only when there is a declaration, so that the common file is the same bytes it was
7078    // and the section is not a byte of zero on every table in the world that never asked for one.
7079    if let Some(clustering) = &table.clustering {
7080        out.extend_from_slice(CLUSTERING);
7081        out.push(clustering.width().tag());
7082        put_u16(
7083            &mut out,
7084            u16::try_from(clustering.columns().len())
7085                .map_err(|_| invalid("too many clustering columns"))?,
7086        );
7087        for &column in clustering.columns() {
7088            put_u16(
7089                &mut out,
7090                u16::try_from(column).map_err(|_| invalid("clustering column index overflow"))?,
7091            );
7092        }
7093    }
7094    // The section table, last, behind its own magic, for the same reason the frequency block is
7095    // behind its own: a reader that stops before it gets a table with no sections, and a table with
7096    // no sections is a correct table. The one difference from the blocks before it is that this one
7097    // is written even when it is empty, so that a file written by this build always says which
7098    // sections it has rather than leaving a reader to infer it from where the bytes ran out.
7099    out.extend_from_slice(SECTIONS);
7100    put_u64(&mut out, table.generation);
7101    put_u16(
7102        &mut out,
7103        u16::try_from(table.sections.len()).map_err(|_| invalid("too many sections"))?,
7104    );
7105    for held in &table.sections {
7106        held.encode(&mut out)?;
7107    }
7108    if table.dictionary_payloads.iter().any(|&bytes| bytes != 0) {
7109        out.extend_from_slice(DICTIONARY_PAYLOADS);
7110        put_u16(
7111            &mut out,
7112            u16::try_from(table.fields.len()).map_err(|_| invalid("too many columns"))?,
7113        );
7114        for at in 0..table.fields.len() {
7115            put_u64(&mut out, table.dictionary_payloads.get(at).copied().unwrap_or(0));
7116        }
7117    }
7118    Ok(out)
7119}
7120
7121/// The small level of the directory, naming every table in the file.
7122///
7123/// This is what a footer slot points at. Each entry carries its own checksum over its table
7124/// directory, so a table whose directory is torn is found when that table is first touched rather
7125/// than being trusted because the catalog around it checksummed.
7126///
7127/// The views go after the tables and are whole here, since a view is text and a column list and has
7128/// no pages for a second level to point at.
7129fn table_nonzero_counts(table: &Table) -> Vec<Option<u64>> {
7130    table
7131        .fields
7132        .iter()
7133        .enumerate()
7134        .map(|(column, field)| {
7135            if !matches!(
7136                field.ty,
7137                LogicalType::TinyInt
7138                    | LogicalType::SmallInt
7139                    | LogicalType::Integer
7140                    | LogicalType::BigInt
7141                    | LogicalType::UTinyInt
7142                    | LogicalType::USmallInt
7143                    | LogicalType::UInteger
7144                    | LogicalType::UBigInt
7145            ) {
7146                return None;
7147            }
7148            let Some(Frequencies::Held(summary)) = &table.frequencies[column] else {
7149                return None;
7150            };
7151            let zero = summary
7152                .entries
7153                .iter()
7154                .find(|entry| entry.value == FrequencyValue::Integer(0))
7155                .map(|entry| entry.count)
7156                .or_else(|| (summary.omitted_max == 0).then_some(0))?;
7157            let nulls = table.stripes.iter().try_fold(0_u64, |count, stripe| {
7158                count.checked_add(stripe.zone.column(column)?.nulls as u64)
7159            })?;
7160            (table.rows as u64).checked_sub(nulls)?.checked_sub(zero)
7161        })
7162        .collect()
7163}
7164
7165fn signed_integer(ty: &LogicalType) -> bool {
7166    matches!(
7167        ty,
7168        LogicalType::TinyInt | LogicalType::SmallInt | LogicalType::Integer | LogicalType::BigInt
7169    )
7170}
7171
7172fn integer_or_date(ty: &LogicalType) -> bool {
7173    matches!(
7174        ty,
7175        LogicalType::TinyInt
7176            | LogicalType::SmallInt
7177            | LogicalType::Integer
7178            | LogicalType::BigInt
7179            | LogicalType::UTinyInt
7180            | LogicalType::USmallInt
7181            | LogicalType::UInteger
7182            | LogicalType::UBigInt
7183            | LogicalType::Date
7184    )
7185}
7186
7187fn table_integer_extremes(table: &Table) -> Vec<StoredIntegerExtremes> {
7188    table
7189        .fields
7190        .iter()
7191        .enumerate()
7192        .map(|(column, field)| {
7193            if !integer_or_date(&field.ty) {
7194                return None;
7195            }
7196            let mut low: Option<i128> = None;
7197            let mut high: Option<i128> = None;
7198            for stripe in &table.stripes {
7199                let range = stripe.zone.column(column)?;
7200                if !range.exact {
7201                    return None;
7202                }
7203                match (range.low.as_ref(), range.high.as_ref()) {
7204                    (Some(Bound::Int(small)), Some(Bound::Int(large))) => {
7205                        low = Some(low.map_or(*small, |held| held.min(*small)));
7206                        high = Some(high.map_or(*large, |held| held.max(*large)));
7207                    }
7208                    (None, None) if stripe.rows == range.nulls => {}
7209                    _ => return None,
7210                }
7211            }
7212            Some(low.zip(high))
7213        })
7214        .collect()
7215}
7216
7217fn reader_integer_extremes(reader: &Reader) -> Result<Vec<StoredIntegerExtremes>> {
7218    reader
7219        .table
7220        .fields
7221        .iter()
7222        .enumerate()
7223        .map(|(column, field)| {
7224            if !integer_or_date(&field.ty) {
7225                return Ok(None);
7226            }
7227            match reader.exact_extremes(column)? {
7228                Some((Bound::Int(low), Bound::Int(high))) => Ok(Some(Some((low, high)))),
7229                None if reader.null_count(column)? == reader.table.rows as u64 => Ok(Some(None)),
7230                _ => Ok(None),
7231            }
7232        })
7233        .collect()
7234}
7235
7236fn table_complete_numeric_frequencies(table: &Table) -> Vec<StoredNumericFrequencies> {
7237    table
7238        .fields
7239        .iter()
7240        .enumerate()
7241        .map(|(column, field)| {
7242            if !integer_or_date(&field.ty) {
7243                return None;
7244            }
7245            let Some(Frequencies::Held(summary)) = table.frequencies.get(column)?.as_ref() else {
7246                return None;
7247            };
7248            if summary.omitted_max != 0 || summary.entries.len() > MAX_CATALOG_FREQUENCIES {
7249                return None;
7250            }
7251            let entries = summary
7252                .entries
7253                .iter()
7254                .map(|entry| {
7255                    let value = match entry.value {
7256                        FrequencyValue::Null => None,
7257                        FrequencyValue::Integer(value) => Some(value),
7258                        FrequencyValue::Code(_) => return None,
7259                    };
7260                    Some((value, entry.count))
7261                })
7262                .collect::<Option<Vec<_>>>()?;
7263            let rows = entries.iter().try_fold(0_u64, |sum, (_, count)| sum.checked_add(*count))?;
7264            (rows == table.rows as u64).then_some(entries)
7265        })
7266        .collect()
7267}
7268
7269fn numeric_frequency_value(value: &Value) -> Option<Option<i128>> {
7270    Some(match value {
7271        Value::Null => None,
7272        Value::TinyInt(value) => Some(i128::from(*value)),
7273        Value::SmallInt(value) => Some(i128::from(*value)),
7274        Value::Integer(value) | Value::Date(value) => Some(i128::from(*value)),
7275        Value::BigInt(value) => Some(i128::from(*value)),
7276        Value::UTinyInt(value) => Some(i128::from(*value)),
7277        Value::USmallInt(value) => Some(i128::from(*value)),
7278        Value::UInteger(value) => Some(i128::from(*value)),
7279        Value::UBigInt(value) => Some(i128::from(*value)),
7280        _ => return None,
7281    })
7282}
7283
7284fn reader_complete_numeric_frequencies(reader: &Reader) -> Result<Vec<StoredNumericFrequencies>> {
7285    reader
7286        .table
7287        .fields
7288        .iter()
7289        .enumerate()
7290        .map(|(column, field)| {
7291            if !integer_or_date(&field.ty) {
7292                return Ok(None);
7293            }
7294            let Some(summary) = reader.frequency_summary(column)? else { return Ok(None) };
7295            if summary.omitted_max != 0 || summary.entries.len() > MAX_CATALOG_FREQUENCIES {
7296                return Ok(None);
7297            }
7298            let entries = reader.decode_frequencies(column, &field.ty, &summary.entries)?;
7299            let Some(entries) = entries
7300                .iter()
7301                .map(|(value, count)| Some((numeric_frequency_value(value)?, *count)))
7302                .collect::<Option<Vec<_>>>()
7303            else {
7304                return Ok(None);
7305            };
7306            let rows = entries.iter().try_fold(0_u64, |sum, (_, count)| sum.checked_add(*count));
7307            Ok((rows == Some(reader.table.rows as u64)).then_some(entries))
7308        })
7309        .collect()
7310}
7311
7312fn table_exact_sum(table: &Table, column: usize) -> Option<(i128, u64)> {
7313    table.stripes.iter().try_fold((0_i128, 0_u64), |(sum, count), stripe| {
7314        let range = stripe.zone.column(column)?;
7315        let sum = sum.checked_add(range.sum?)?;
7316        let nonnull = (stripe.rows as u64).checked_sub(range.nulls as u64)?;
7317        Some((sum, count.checked_add(nonnull)?))
7318    })
7319}
7320
7321fn table_aggregate_sums(table: &Table) -> Vec<Option<(i128, u64)>> {
7322    table
7323        .fields
7324        .iter()
7325        .enumerate()
7326        .map(|(column, field)| {
7327            signed_integer(&field.ty).then(|| table_exact_sum(table, column)).flatten()
7328        })
7329        .collect()
7330}
7331
7332fn reader_nonzero_counts(reader: &Reader) -> Result<Vec<Option<u64>>> {
7333    reader
7334        .table
7335        .fields
7336        .iter()
7337        .enumerate()
7338        .map(|(column, field)| {
7339            if !matches!(
7340                field.ty,
7341                LogicalType::TinyInt
7342                    | LogicalType::SmallInt
7343                    | LogicalType::Integer
7344                    | LogicalType::BigInt
7345                    | LogicalType::UTinyInt
7346                    | LogicalType::USmallInt
7347                    | LogicalType::UInteger
7348                    | LogicalType::UBigInt
7349            ) {
7350                return Ok(None);
7351            }
7352            let Some(summary) = reader.frequency_summary(column)? else {
7353                return Ok(None);
7354            };
7355            let zero = summary
7356                .entries
7357                .iter()
7358                .find(|entry| entry.value == FrequencyValue::Integer(0))
7359                .map(|entry| entry.count)
7360                .or_else(|| (summary.omitted_max == 0).then_some(0));
7361            let Some(zero) = zero else { return Ok(None) };
7362            let nulls = reader.null_count(column)?;
7363            Ok((reader.table.rows as u64)
7364                .checked_sub(nulls)
7365                .and_then(|count| count.checked_sub(zero)))
7366        })
7367        .collect()
7368}
7369
7370fn reader_aggregate_sums(reader: &Reader) -> Result<Vec<Option<(i128, u64)>>> {
7371    reader
7372        .table
7373        .fields
7374        .iter()
7375        .enumerate()
7376        .map(
7377            |(column, field)| {
7378                if signed_integer(&field.ty) { reader.exact_sum(column) } else { Ok(None) }
7379            },
7380        )
7381        .collect()
7382}
7383
7384fn encode_catalog(entries: &[Entry], views: &[ViewEntry]) -> Result<Vec<u8>> {
7385    let mut out = CATALOG.to_vec();
7386    put_u32(&mut out, u32::try_from(entries.len()).map_err(|_| invalid("too many tables"))?);
7387    for entry in entries {
7388        let name = entry.name.as_bytes();
7389        put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("table name too long"))?);
7390        out.extend_from_slice(name);
7391        put_u64(&mut out, u64::try_from(entry.rows).map_err(|_| invalid("row count overflow"))?);
7392        put_u16(
7393            &mut out,
7394            u16::try_from(entry.fields.len()).map_err(|_| invalid("too many columns"))?,
7395        );
7396        for field in &entry.fields {
7397            let name = field.name.as_bytes();
7398            put_u16(
7399                &mut out,
7400                u16::try_from(name.len()).map_err(|_| invalid("column name too long"))?,
7401            );
7402            out.extend_from_slice(name);
7403            put_type(&mut out, &field.ty)?;
7404            out.push(u8::from(field.not_null));
7405        }
7406        put_u64(&mut out, entry.directory.offset);
7407        put_u32(&mut out, entry.directory.length);
7408        put_u64(&mut out, entry.directory.hash);
7409    }
7410    put_u32(&mut out, u32::try_from(views.len()).map_err(|_| invalid("too many views"))?);
7411    for view in views {
7412        let name = view.name.as_bytes();
7413        put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("view name too long"))?);
7414        out.extend_from_slice(name);
7415        put_long_text(&mut out, &view.sql, "view body")?;
7416        put_long_text(&mut out, &view.statement, "view statement")?;
7417        put_u16(
7418            &mut out,
7419            u16::try_from(view.aliases.len()).map_err(|_| invalid("too many aliases"))?,
7420        );
7421        for alias in &view.aliases {
7422            let alias = alias.as_bytes();
7423            put_u16(
7424                &mut out,
7425                u16::try_from(alias.len()).map_err(|_| invalid("alias name too long"))?,
7426            );
7427            out.extend_from_slice(alias);
7428        }
7429        put_u16(
7430            &mut out,
7431            u16::try_from(view.columns.len()).map_err(|_| invalid("too many columns"))?,
7432        );
7433        for field in &view.columns {
7434            let name = field.name.as_bytes();
7435            put_u16(
7436                &mut out,
7437                u16::try_from(name.len()).map_err(|_| invalid("column name too long"))?,
7438            );
7439            out.extend_from_slice(name);
7440            put_type(&mut out, &field.ty)?;
7441            out.push(u8::from(field.not_null));
7442        }
7443    }
7444    out.extend_from_slice(NONZERO_COUNTS);
7445    for entry in entries {
7446        if entry.nonzero.len() != entry.fields.len() {
7447            return Err(invalid("nonzero count width differs from schema"));
7448        }
7449        for count in &entry.nonzero {
7450            match count {
7451                None => out.push(0),
7452                Some(count) => {
7453                    out.push(1);
7454                    put_u64(&mut out, *count);
7455                }
7456            }
7457        }
7458    }
7459    out.extend_from_slice(AGGREGATE_SUMS);
7460    for entry in entries {
7461        if entry.aggregates.len() != entry.fields.len() {
7462            return Err(invalid("aggregate sum width differs from schema"));
7463        }
7464        for summary in &entry.aggregates {
7465            match summary {
7466                None => out.push(0),
7467                Some((sum, count)) => {
7468                    out.push(1);
7469                    out.extend_from_slice(&sum.to_le_bytes());
7470                    put_u64(&mut out, *count);
7471                }
7472            }
7473        }
7474    }
7475    out.extend_from_slice(DISTINCT_COUNTS);
7476    for entry in entries {
7477        if entry.distincts.len() != entry.fields.len() {
7478            return Err(invalid("distinct count width differs from schema"));
7479        }
7480        for count in &entry.distincts {
7481            match count {
7482                None => out.push(0),
7483                Some(count) => {
7484                    if *count > entry.rows as u64 {
7485                        return Err(invalid("distinct count exceeds table rows"));
7486                    }
7487                    out.push(1);
7488                    put_u64(&mut out, *count);
7489                }
7490            }
7491        }
7492    }
7493    out.extend_from_slice(INTEGER_EXTREMES);
7494    for entry in entries {
7495        if entry.extremes.len() != entry.fields.len() {
7496            return Err(invalid("integer extremes width differs from schema"));
7497        }
7498        for (field, extremes) in entry.fields.iter().zip(&entry.extremes) {
7499            match extremes {
7500                None => out.push(0),
7501                Some(None) if integer_or_date(&field.ty) => out.push(1),
7502                Some(Some((low, high))) if integer_or_date(&field.ty) && low <= high => {
7503                    out.push(2);
7504                    out.extend_from_slice(&low.to_le_bytes());
7505                    out.extend_from_slice(&high.to_le_bytes());
7506                }
7507                _ => return Err(invalid("integer extremes type or range differs")),
7508            }
7509        }
7510    }
7511    out.extend_from_slice(COMPLETE_FREQUENCIES);
7512    for entry in entries {
7513        if entry.frequencies.len() != entry.fields.len() {
7514            return Err(invalid("numeric frequency width differs from schema"));
7515        }
7516        for (field, frequencies) in entry.fields.iter().zip(&entry.frequencies) {
7517            match frequencies {
7518                None => out.push(0),
7519                Some(entries)
7520                    if integer_or_date(&field.ty) && entries.len() <= MAX_CATALOG_FREQUENCIES =>
7521                {
7522                    let mut total = 0_u64;
7523                    for (at, (value, count)) in entries.iter().enumerate() {
7524                        if entries[..at].iter().any(|(held, _)| held == value) {
7525                            return Err(invalid("numeric frequency value repeats"));
7526                        }
7527                        total = total
7528                            .checked_add(*count)
7529                            .ok_or_else(|| invalid("numeric frequency count overflows"))?;
7530                    }
7531                    if total != entry.rows as u64 {
7532                        return Err(invalid("numeric frequencies do not cover table rows"));
7533                    }
7534                    out.push(1);
7535                    out.push(entries.len() as u8);
7536                    for (value, count) in entries {
7537                        match value {
7538                            None => out.push(0),
7539                            Some(value) => {
7540                                out.push(1);
7541                                out.extend_from_slice(&value.to_le_bytes());
7542                            }
7543                        }
7544                        put_u64(&mut out, *count);
7545                    }
7546                }
7547                _ => return Err(invalid("numeric frequency type or width differs")),
7548            }
7549        }
7550    }
7551    Ok(out)
7552}
7553
7554/// A length and that many bytes, for text that is allowed to be longer than a name.
7555fn put_long_text(out: &mut Vec<u8>, text: &str, what: &str) -> Result<()> {
7556    let bytes = text.as_bytes();
7557    put_u32(out, u32::try_from(bytes.len()).map_err(|_| invalid(&format!("{what} too long")))?);
7558    out.extend_from_slice(bytes);
7559    Ok(())
7560}
7561
7562/// Reads the catalog directory back, checking every span against the file before anything is
7563/// allocated for it.
7564fn decode_catalog(bytes: &[u8], size: u64) -> Result<(Vec<Entry>, Vec<ViewEntry>)> {
7565    let mut cur = Cursor::new(bytes);
7566    if cur.take(8)? != CATALOG {
7567        return Err(invalid("catalog magic differs"));
7568    }
7569    let count = cur.u32()? as usize;
7570    let mut entries: Vec<Entry> = Vec::with_capacity(count.min(1024));
7571    for _ in 0..count {
7572        let name = cur.text()?;
7573        let rows = usize::try_from(cur.u64()?).map_err(|_| invalid("row count does not fit"))?;
7574        let width = cur.u16()? as usize;
7575        let mut fields = Vec::with_capacity(width);
7576        for _ in 0..width {
7577            let name = cur.text()?;
7578            let ty = read_type(&mut cur)?;
7579            let not_null = match cur.u8()? {
7580                0 => false,
7581                1 => true,
7582                _ => return Err(invalid("nullability flag differs")),
7583            };
7584            fields.push(Field { name, ty, not_null });
7585        }
7586        let directory = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
7587        let end = directory
7588            .offset
7589            .checked_add(u64::from(directory.length))
7590            .ok_or_else(|| invalid("table directory offset overflow"))?;
7591        if directory.offset < HEADER
7592            || end > size
7593            || directory.length as usize > MAX_DIRECTORY
7594            || directory.length == 0
7595        {
7596            return Err(invalid("table directory range is outside the file"));
7597        }
7598        if entries.iter().any(|held| held.name == name) {
7599            return Err(invalid("two tables in the catalog have the same name"));
7600        }
7601        let nonzero = vec![None; fields.len()];
7602        let aggregates = vec![None; fields.len()];
7603        let distincts = vec![None; fields.len()];
7604        let extremes = vec![None; fields.len()];
7605        let frequencies = vec![None; fields.len()];
7606        entries.push(Entry {
7607            name,
7608            fields,
7609            rows,
7610            directory,
7611            nonzero,
7612            aggregates,
7613            distincts,
7614            extremes,
7615            frequencies,
7616        });
7617    }
7618    // A catalog that ends where the tables end is a catalog with no views in it, which is every
7619    // file written before format 25. That is why the count is allowed to be missing rather than
7620    // read as a zero that has to be there: an older file has nothing after the last table entry at
7621    // all, and [`READABLE`] says those files still open.
7622    let count = if cur.done() { 0 } else { cur.u32()? as usize };
7623    let mut views: Vec<ViewEntry> = Vec::with_capacity(count.min(1024));
7624    for _ in 0..count {
7625        let name = cur.text()?;
7626        let sql = cur.long_text()?;
7627        let statement = cur.long_text()?;
7628        let width = cur.u16()? as usize;
7629        let mut aliases = Vec::with_capacity(width);
7630        for _ in 0..width {
7631            aliases.push(cur.text()?);
7632        }
7633        let width = cur.u16()? as usize;
7634        let mut columns = Vec::with_capacity(width);
7635        for _ in 0..width {
7636            let name = cur.text()?;
7637            let ty = read_type(&mut cur)?;
7638            let not_null = match cur.u8()? {
7639                0 => false,
7640                1 => true,
7641                _ => return Err(invalid("nullability flag differs")),
7642            };
7643            columns.push(Field { name, ty, not_null });
7644        }
7645        // The same rule the tables above get, and for the same reason. Two entries under one name
7646        // is a catalog nothing can answer a lookup from, and finding that out here is better than
7647        // finding it out from whichever of the two a search happened to reach first.
7648        if views.iter().any(|held| held.name == name) {
7649            return Err(invalid("two views in the catalog have the same name"));
7650        }
7651        if entries.iter().any(|held| held.name == name) {
7652            return Err(invalid("a table and a view in the catalog have the same name"));
7653        }
7654        views.push(ViewEntry { name, sql, statement, aliases, columns });
7655    }
7656    if !cur.done() {
7657        if cur.take(8)? != NONZERO_COUNTS {
7658            return Err(invalid("catalog extension magic differs"));
7659        }
7660        for entry in &mut entries {
7661            for (field, count) in entry.fields.iter().zip(&mut entry.nonzero) {
7662                *count = match cur.u8()? {
7663                    0 => None,
7664                    1 if matches!(
7665                        field.ty,
7666                        LogicalType::TinyInt
7667                            | LogicalType::SmallInt
7668                            | LogicalType::Integer
7669                            | LogicalType::BigInt
7670                            | LogicalType::UTinyInt
7671                            | LogicalType::USmallInt
7672                            | LogicalType::UInteger
7673                            | LogicalType::UBigInt
7674                    ) =>
7675                    {
7676                        let value = cur.u64()?;
7677                        if value > entry.rows as u64 {
7678                            return Err(invalid("nonzero count exceeds rows"));
7679                        }
7680                        Some(value)
7681                    }
7682                    _ => return Err(invalid("nonzero count tag or column type differs")),
7683                };
7684            }
7685        }
7686    }
7687    if !cur.done() {
7688        if cur.take(8)? != AGGREGATE_SUMS {
7689            return Err(invalid("aggregate catalog extension magic differs"));
7690        }
7691        for entry in &mut entries {
7692            for (field, summary) in entry.fields.iter().zip(&mut entry.aggregates) {
7693                *summary = match cur.u8()? {
7694                    0 => None,
7695                    1 if signed_integer(&field.ty) => {
7696                        let sum = i128::from_le_bytes(
7697                            cur.take(16)?
7698                                .try_into()
7699                                .map_err(|_| invalid("aggregate sum is truncated"))?,
7700                        );
7701                        let count = cur.u64()?;
7702                        if count > entry.rows as u64 {
7703                            return Err(invalid("aggregate count exceeds table rows"));
7704                        }
7705                        Some((sum, count))
7706                    }
7707                    _ => return Err(invalid("aggregate sum tag or column type differs")),
7708                };
7709            }
7710        }
7711    }
7712    if !cur.done() {
7713        if cur.take(8)? != DISTINCT_COUNTS {
7714            return Err(invalid("distinct catalog extension magic differs"));
7715        }
7716        for entry in &mut entries {
7717            for count in &mut entry.distincts {
7718                *count = match cur.u8()? {
7719                    0 => None,
7720                    1 => {
7721                        let value = cur.u64()?;
7722                        if value > entry.rows as u64 {
7723                            return Err(invalid("distinct count exceeds table rows"));
7724                        }
7725                        Some(value)
7726                    }
7727                    _ => return Err(invalid("distinct count tag differs")),
7728                };
7729            }
7730        }
7731    }
7732    if !cur.done() {
7733        if cur.take(8)? != INTEGER_EXTREMES {
7734            return Err(invalid("integer extremes catalog extension magic differs"));
7735        }
7736        for entry in &mut entries {
7737            for (field, extremes) in entry.fields.iter().zip(&mut entry.extremes) {
7738                *extremes = match cur.u8()? {
7739                    0 => None,
7740                    1 if integer_or_date(&field.ty) => Some(None),
7741                    2 if integer_or_date(&field.ty) => {
7742                        let low = i128::from_le_bytes(
7743                            cur.take(16)?
7744                                .try_into()
7745                                .map_err(|_| invalid("minimum is truncated"))?,
7746                        );
7747                        let high = i128::from_le_bytes(
7748                            cur.take(16)?
7749                                .try_into()
7750                                .map_err(|_| invalid("maximum is truncated"))?,
7751                        );
7752                        if low > high {
7753                            return Err(invalid("integer extremes are reversed"));
7754                        }
7755                        Some(Some((low, high)))
7756                    }
7757                    _ => return Err(invalid("integer extremes tag or type differs")),
7758                };
7759            }
7760        }
7761    }
7762    if !cur.done() {
7763        if cur.take(8)? != COMPLETE_FREQUENCIES {
7764            return Err(invalid("numeric frequency catalog extension magic differs"));
7765        }
7766        for entry in &mut entries {
7767            for (field, frequencies) in entry.fields.iter().zip(&mut entry.frequencies) {
7768                *frequencies = match cur.u8()? {
7769                    0 => None,
7770                    1 if integer_or_date(&field.ty) => {
7771                        let len = cur.u8()? as usize;
7772                        if len > MAX_CATALOG_FREQUENCIES {
7773                            return Err(invalid("too many catalog numeric frequencies"));
7774                        }
7775                        let mut values = Vec::with_capacity(len);
7776                        let mut total = 0_u64;
7777                        for _ in 0..len {
7778                            let value = match cur.u8()? {
7779                                0 => None,
7780                                1 => Some(i128::from_le_bytes(cur.take(16)?.try_into().map_err(
7781                                    |_| invalid("numeric frequency value is truncated"),
7782                                )?)),
7783                                _ => return Err(invalid("numeric frequency value tag differs")),
7784                            };
7785                            if values.iter().any(|(held, _)| *held == value) {
7786                                return Err(invalid("numeric frequency value repeats"));
7787                            }
7788                            let count = cur.u64()?;
7789                            total = total
7790                                .checked_add(count)
7791                                .ok_or_else(|| invalid("numeric frequency count overflows"))?;
7792                            values.push((value, count));
7793                        }
7794                        if total != entry.rows as u64 {
7795                            return Err(invalid("numeric frequencies do not cover table rows"));
7796                        }
7797                        Some(values)
7798                    }
7799                    _ => return Err(invalid("numeric frequency tag or type differs")),
7800                };
7801            }
7802        }
7803    }
7804    if !cur.done() {
7805        return Err(invalid("catalog has trailing bytes"));
7806    }
7807    Ok((entries, views))
7808}
7809
7810/// Reads the fields of a directory or a catalog in order, off bytes in memory or out of the file.
7811///
7812/// A catalog is small and is read whole. A table directory is not: at ten million rows of `hits` it
7813/// is nearly a megabyte, and holding that buffer while the table it describes is built out of it
7814/// put both at the peak of every query. Out of the file, the cursor holds one window of
7815/// [`DIRECTORY_WINDOW`] bytes and moves it forward as the fields are read, so what a directory
7816/// costs at open is what it decodes into and not that plus its own bytes.
7817struct Cursor<'a> {
7818    bytes: &'a [u8],
7819    at: usize,
7820    window: Option<Window<'a>>,
7821}
7822
7823/// The part of a directory in the file that a [`Cursor`] has read in.
7824struct Window<'a> {
7825    file: &'a File,
7826    offset: u64,
7827    length: usize,
7828    /// Where `held` starts, counted from the start of the directory.
7829    start: usize,
7830    held: Vec<u8>,
7831    /// How much to read at once, which is [`DIRECTORY_WINDOW`] outside the tests.
7832    size: usize,
7833}
7834
7835/// How much of a directory a cursor reading one out of the file holds at once.
7836const DIRECTORY_WINDOW: usize = 64 << 10;
7837
7838impl<'a> Cursor<'a> {
7839    fn new(bytes: &'a [u8]) -> Self {
7840        Self { bytes, at: 0, window: None }
7841    }
7842
7843    /// A cursor over `length` bytes of `file` from `offset`, which it reads a window at a time.
7844    fn over(file: &'a File, offset: u64, length: usize) -> Self {
7845        let window =
7846            Window { file, offset, length, start: 0, held: Vec::new(), size: DIRECTORY_WINDOW };
7847        Self { bytes: &[], at: 0, window: Some(window) }
7848    }
7849
7850    /// How many bytes the cursor walks in all.
7851    fn len(&self) -> usize {
7852        self.window.as_ref().map_or(self.bytes.len(), |window| window.length)
7853    }
7854
7855    /// Makes sure the next `len` bytes are in memory.
7856    fn ensure(&mut self, len: usize) -> Result<()> {
7857        let end = self.at.checked_add(len).ok_or_else(|| invalid("directory offset overflow"))?;
7858        if end > self.len() {
7859            return Err(invalid("directory is truncated"));
7860        }
7861        let Some(window) = &mut self.window else { return Ok(()) };
7862        if self.at < window.start || end > window.start + window.held.len() {
7863            let want = len.max(window.size).min(window.length - self.at);
7864            window.start = self.at;
7865            window.held.resize(want, 0);
7866            read_at(window.file, window.offset + self.at as u64, &mut window.held)?;
7867        }
7868        Ok(())
7869    }
7870
7871    /// `len` bytes from `at`, which [`Self::ensure`] has already brought in.
7872    fn held(&self, at: usize, len: usize) -> &[u8] {
7873        match &self.window {
7874            Some(window) => &window.held[at - window.start..at - window.start + len],
7875            None => &self.bytes[at..at + len],
7876        }
7877    }
7878
7879    /// The next `len` bytes, without moving past them.
7880    #[inline]
7881    fn peek(&mut self, len: usize) -> Result<&[u8]> {
7882        if self.window.is_none() {
7883            let bytes = self.bytes;
7884            return Ok(&bytes[self.at..self.end(len)?]);
7885        }
7886        self.ensure(len)?;
7887        Ok(self.held(self.at, len))
7888    }
7889
7890    /// The next `len` bytes, moving past them.
7891    ///
7892    /// Every data page is decoded through this, a byte or a word at a time, so a cursor over bytes
7893    /// already in memory takes them here and never reaches [`Self::ensure`]. With the window check
7894    /// on every call, q06 on TPC-H spent a seventh of its instructions in it.
7895    #[inline]
7896    fn take(&mut self, len: usize) -> Result<&[u8]> {
7897        if self.window.is_none() {
7898            let bytes = self.bytes;
7899            let (at, end) = (self.at, self.end(len)?);
7900            self.at = end;
7901            return Ok(&bytes[at..end]);
7902        }
7903        self.take_windowed(len)
7904    }
7905
7906    /// Moves over a checked field without reading its payload from a windowed directory.
7907    fn skip(&mut self, len: usize) -> Result<()> {
7908        let end = self.at.checked_add(len).ok_or_else(|| invalid("directory offset overflow"))?;
7909        if end > self.len() {
7910            return Err(invalid("directory is truncated"));
7911        }
7912        self.at = end;
7913        Ok(())
7914    }
7915
7916    fn skip_bound(&mut self) -> Result<()> {
7917        match self.u8()? {
7918            0 => Ok(()),
7919            1 => self.skip(16),
7920            2 => self.skip(8),
7921            3 => {
7922                let length = self.u32()? as usize;
7923                self.skip(length)
7924            }
7925            4 => self.skip(17),
7926            _ => Err(invalid("a stored bound has an unknown tag")),
7927        }
7928    }
7929
7930    /// Where `len` bytes from here end, when they end inside the bytes.
7931    #[inline]
7932    fn end(&self, len: usize) -> Result<usize> {
7933        let end = self.at.checked_add(len).ok_or_else(|| invalid("directory offset overflow"))?;
7934        if end > self.bytes.len() {
7935            return Err(invalid("directory is truncated"));
7936        }
7937        Ok(end)
7938    }
7939
7940    /// [`Self::take`] out of the file, a window at a time.
7941    #[inline(never)]
7942    fn take_windowed(&mut self, len: usize) -> Result<&[u8]> {
7943        self.ensure(len)?;
7944        self.at += len;
7945        Ok(self.held(self.at - len, len))
7946    }
7947    #[inline]
7948    fn u8(&mut self) -> Result<u8> {
7949        Ok(self.take(1)?[0])
7950    }
7951    #[inline]
7952    fn u16(&mut self) -> Result<u16> {
7953        Ok(u16::from_le_bytes(self.take(2)?.try_into().expect("two bytes")))
7954    }
7955    #[inline]
7956    fn u32(&mut self) -> Result<u32> {
7957        Ok(u32::from_le_bytes(self.take(4)?.try_into().expect("four bytes")))
7958    }
7959    #[inline]
7960    fn u64(&mut self) -> Result<u64> {
7961        Ok(u64::from_le_bytes(self.take(8)?.try_into().expect("eight bytes")))
7962    }
7963    fn var_u64(&mut self) -> Result<u64> {
7964        let mut value = 0_u64;
7965        for shift in (0..=63).step_by(7) {
7966            let byte = self.u8()?;
7967            let part = u64::from(byte & 0x7f);
7968            if shift == 63 && part > 1 {
7969                return Err(invalid("frequency ordinal varint overflows"));
7970            }
7971            value |= part << shift;
7972            if byte & 0x80 == 0 {
7973                return Ok(value);
7974            }
7975        }
7976        Err(invalid("frequency ordinal varint is too long"))
7977    }
7978    /// A zone map's end, in the layout `rudb_common::bounds` defines.
7979    ///
7980    /// The bytes are the ones this directory has written since format 10 and the codec moved to
7981    /// rank zero rather than being copied, because a column summary now writes the same two ends
7982    /// and two encodings of one type is how the two quietly stop agreeing.
7983    ///
7984    /// A bound's length is in the bound, so out of the file the cursor offers the codec a few bytes
7985    /// and offers it twice as many whenever it runs out before the directory does.
7986    fn bound(&mut self) -> Result<Option<Bound>> {
7987        let rest = self.len().saturating_sub(self.at);
7988        let mut want = 32;
7989        loop {
7990            let offered = self.peek(want.min(rest))?;
7991            let mut used = 0;
7992            match bounds::get(offered, &mut used) {
7993                Ok(bound) => {
7994                    self.at += used;
7995                    return Ok(bound);
7996                }
7997                Err(_) if want < rest => want *= 2,
7998                Err(error) => return Err(error),
7999            }
8000        }
8001    }
8002    fn text(&mut self) -> Result<String> {
8003        let len = self.u16()? as usize;
8004        String::from_utf8(self.take(len)?.to_vec()).map_err(|_| invalid("name is not UTF-8"))
8005    }
8006    /// Whether everything has been read, which is how a section that an older file does not have at
8007    /// all is told from one that is there and empty.
8008    fn done(&self) -> bool {
8009        self.at >= self.len()
8010    }
8011    /// The same, for text that is a query rather than a name.
8012    ///
8013    /// A name fits in sixteen bits of length and a view body does not have to. Nobody writes a 64
8014    /// kilobyte identifier by accident and people do write generated queries that long, and a view
8015    /// that could not be written down because its body was too big would be a limit invented here
8016    /// rather than one anything else in the engine has.
8017    fn long_text(&mut self) -> Result<String> {
8018        let len = self.u32()? as usize;
8019        String::from_utf8(self.take(len)?.to_vec()).map_err(|_| invalid("text is not UTF-8"))
8020    }
8021}
8022
8023/// One column's frequency synopsis, or `None` for a column that has none, checked against the column.
8024fn decode_summary(
8025    cur: &mut Cursor<'_>,
8026    field: &Field,
8027    rows: usize,
8028    values: bool,
8029) -> Result<Option<FrequencySummary>> {
8030    Ok(match cur.u8()? {
8031        0 => None,
8032        1 => {
8033            let omitted_max = cur.u64()?;
8034            let count = cur.u32()? as usize;
8035            if count > FREQUENCY_ENTRIES {
8036                return Err(invalid("frequency entry count exceeds its bound"));
8037            }
8038            let mut entries = Vec::with_capacity(count);
8039            // row at a time: directory decoding validates each persisted bounded frequency entry.
8040            for _ in 0..count {
8041                let value = match cur.u8()? {
8042                    0 => FrequencyValue::Null,
8043                    1 => FrequencyValue::Integer(i128::from_le_bytes(
8044                        cur.take(16)?.try_into().expect("sixteen bytes"),
8045                    )),
8046                    2 => FrequencyValue::Code(cur.u32()?),
8047                    _ => return Err(invalid("frequency value tag differs")),
8048                };
8049                let valid = matches!(
8050                    (&field.ty, value),
8051                    (_, FrequencyValue::Null)
8052                        | (LogicalType::Varchar, FrequencyValue::Code(_))
8053                        | (
8054                            LogicalType::TinyInt
8055                                | LogicalType::SmallInt
8056                                | LogicalType::Integer
8057                                | LogicalType::BigInt
8058                                | LogicalType::UTinyInt
8059                                | LogicalType::USmallInt
8060                                | LogicalType::UInteger
8061                                | LogicalType::UBigInt
8062                                | LogicalType::Date
8063                                | LogicalType::Timestamp,
8064                            FrequencyValue::Integer(_),
8065                        )
8066                );
8067                if !valid {
8068                    return Err(invalid("frequency value does not match its column"));
8069                }
8070                let count = cur.u64()?;
8071                if count == 0 || count > rows as u64 {
8072                    return Err(invalid("frequency count is outside the table"));
8073                }
8074                entries.push(FrequencyEntry { value, count });
8075            }
8076            if entries.windows(2).any(|pair| pair[0].count < pair[1].count) {
8077                return Err(invalid("frequency entries are not descending"));
8078            }
8079            let ordinals = {
8080                let ordinal_count = cur.u32()? as usize;
8081                if ordinal_count > FREQUENCY_ORDINALS || ordinal_count > rows {
8082                    return Err(invalid("frequency ordinal count exceeds its bound"));
8083                }
8084                let mut ordinals = Vec::with_capacity(ordinal_count);
8085                let mut previous = 0_u64;
8086                for at in 0..ordinal_count {
8087                    let delta = cur.var_u64()?;
8088                    if at != 0 && delta == 0 {
8089                        return Err(invalid("frequency ordinals are not increasing"));
8090                    }
8091                    let ordinal = if at == 0 {
8092                        delta
8093                    } else {
8094                        previous
8095                            .checked_add(delta)
8096                            .ok_or_else(|| invalid("frequency ordinal overflows"))?
8097                    };
8098                    if ordinal >= rows as u64 {
8099                        return Err(invalid("frequency ordinal is outside the table"));
8100                    }
8101                    ordinals.push(ordinal);
8102                    previous = ordinal;
8103                }
8104                ordinals
8105            };
8106            let ordinal_entries = if values {
8107                let mut ordinal_entries = Vec::with_capacity(ordinals.len());
8108                for _ in 0..ordinals.len() {
8109                    let entry = cur.u16()?;
8110                    if entry as usize >= entries.len() {
8111                        return Err(invalid("frequency ordinal value is outside its entries"));
8112                    }
8113                    ordinal_entries.push(entry);
8114                }
8115                ordinal_entries
8116            } else {
8117                Vec::new()
8118            };
8119            Some(FrequencySummary { entries, omitted_max, ordinals, ordinal_entries })
8120        }
8121        _ => return Err(invalid("frequency summary tag differs")),
8122    })
8123}
8124
8125/// Skips a synopsis whose column the caller does not need. The directory checksum was checked
8126/// before this walk, and the fields still need their lengths and tags checked to find the next one.
8127fn skip_summary(cur: &mut Cursor<'_>, values: bool, rows: usize) -> Result<()> {
8128    match cur.u8()? {
8129        0 => Ok(()),
8130        1 => {
8131            cur.skip(8)?;
8132            let entries = cur.u32()? as usize;
8133            if entries > FREQUENCY_ENTRIES {
8134                return Err(invalid("frequency entry count exceeds its bound"));
8135            }
8136            for _ in 0..entries {
8137                match cur.u8()? {
8138                    0 => {}
8139                    1 => cur.skip(16)?,
8140                    2 => cur.skip(4)?,
8141                    _ => return Err(invalid("frequency value tag differs")),
8142                }
8143                cur.skip(8)?;
8144            }
8145            let ordinals = cur.u32()? as usize;
8146            if ordinals > FREQUENCY_ORDINALS || ordinals > rows {
8147                return Err(invalid("frequency ordinal count exceeds its bound"));
8148            }
8149            for _ in 0..ordinals {
8150                cur.var_u64()?;
8151            }
8152            if values {
8153                cur.skip(ordinals * 2)?;
8154            }
8155            Ok(())
8156        }
8157        _ => Err(invalid("frequency summary tag differs")),
8158    }
8159}
8160
8161/// Reads only the catalog, stripe null counts, and one frequency synopsis. This is the cold path
8162/// for a summary-backed count; constructing every page descriptor and zone map would make it cost
8163/// the size of the table directory even when no row is read.
8164fn quick_nonzero(
8165    mut cur: Cursor<'_>,
8166    name: &str,
8167    fields: &[Field],
8168    rows: usize,
8169    wanted: usize,
8170) -> Result<Option<u64>> {
8171    if cur.take(8)? != DIRECTORY || cur.text()? != name {
8172        return Err(invalid("table directory differs from the catalog"));
8173    }
8174    let width = cur.u16()? as usize;
8175    if width != fields.len() {
8176        return Err(invalid("table directory width differs from the catalog"));
8177    }
8178    for field in fields {
8179        let stored =
8180            Field { name: cur.text()?, ty: read_type(&mut cur)?, not_null: cur.u8()? != 0 };
8181        if &stored != field {
8182            return Err(invalid("table directory schema differs from the catalog"));
8183        }
8184    }
8185    let mut dictionaries = Vec::with_capacity(width);
8186    for _ in 0..width {
8187        let held = match cur.u8()? {
8188            0 => false,
8189            1 => {
8190                cur.skip(20)?;
8191                true
8192            }
8193            _ => return Err(invalid("dictionary page tag differs")),
8194        };
8195        dictionaries.push(held);
8196    }
8197    for _ in 0..width {
8198        match cur.u8()? {
8199            0 => {}
8200            1 => cur.skip(8)?,
8201            _ => return Err(invalid("distinct count tag differs")),
8202        }
8203    }
8204    if cur.u64()? != rows as u64 {
8205        return Err(invalid("table row count differs from the catalog"));
8206    }
8207    let stripes = cur.u32()? as usize;
8208    let mut total = 0_usize;
8209    let mut nulls = 0_u64;
8210    for _ in 0..stripes {
8211        let parts = cur.u32()? as usize;
8212        if parts == 0 || parts > STRIPE_PARTS {
8213            return Err(invalid("stripe part count is outside its bound"));
8214        }
8215        let mut stripe_rows = 0_usize;
8216        for _ in 0..parts {
8217            stripe_rows = stripe_rows
8218                .checked_add(cur.u32()? as usize)
8219                .ok_or_else(|| invalid("stripe row count overflow"))?;
8220        }
8221        total =
8222            total.checked_add(stripe_rows).ok_or_else(|| invalid("stripe row count overflow"))?;
8223        cur.skip(12 + width * 12)?;
8224        for (field, held) in fields.iter().zip(&dictionaries) {
8225            if field.ty == LogicalType::Varchar && *held {
8226                cur.skip(20)?;
8227            }
8228        }
8229        for _ in 0..width * 2 {
8230            match cur.u8()? {
8231                0 => {}
8232                1 => cur.skip(20)?,
8233                _ => return Err(invalid("stripe page tag differs")),
8234            }
8235        }
8236        for column in 0..width {
8237            cur.skip_bound()?;
8238            cur.skip_bound()?;
8239            let count = cur.u32()? as u64;
8240            if count > stripe_rows as u64 {
8241                return Err(invalid("null count exceeds stripe rows"));
8242            }
8243            if column == wanted {
8244                nulls = nulls.checked_add(count).ok_or_else(|| invalid("null count overflow"))?;
8245            }
8246            cur.skip(1)?;
8247            match cur.u8()? {
8248                0 => {}
8249                1 => cur.skip(16)?,
8250                _ => return Err(invalid("a stripe sum has an unknown tag")),
8251            }
8252        }
8253    }
8254    if total != rows {
8255        return Err(invalid("table row count differs from stripes"));
8256    }
8257    if cur.done() {
8258        return Ok(None);
8259    }
8260    let magic = cur.take(8)?;
8261    let values = magic == FREQUENCIES;
8262    if !values && magic != FREQUENCIES_V2 {
8263        return Err(invalid("directory extension magic differs"));
8264    }
8265    if cur.u16()? as usize != width {
8266        return Err(invalid("frequency column count differs"));
8267    }
8268    for _ in 0..wanted {
8269        skip_summary(&mut cur, values, rows)?;
8270    }
8271    let Some(summary) = decode_summary(&mut cur, &fields[wanted], rows, values)? else {
8272        return Ok(None);
8273    };
8274    let zero = summary
8275        .entries
8276        .iter()
8277        .find(|entry| entry.value == FrequencyValue::Integer(0))
8278        .map(|entry| entry.count)
8279        .or_else(|| (summary.omitted_max == 0).then_some(0));
8280    Ok(zero.and_then(|zero| (rows as u64).checked_sub(nulls)?.checked_sub(zero)))
8281}
8282
8283fn decode_directory(bytes: &[u8], size: u64) -> Result<Table> {
8284    read_directory(Cursor::new(bytes), size, None)
8285}
8286
8287/// A directory out of `cur`, which is a whole one in memory or one being read out of the file.
8288///
8289/// `stored_at` is where the directory starts in the file when it is being read out of it, and then
8290/// every frequency synopsis is checked and left there, as [`Frequencies::Stored`].
8291fn read_directory(mut cur: Cursor<'_>, size: u64, stored_at: Option<u64>) -> Result<Table> {
8292    if cur.take(8)? != DIRECTORY {
8293        return Err(invalid("directory magic differs"));
8294    }
8295    let name = cur.text()?;
8296    let width = cur.u16()? as usize;
8297    let mut fields = Vec::with_capacity(width);
8298    for _ in 0..width {
8299        let name = cur.text()?;
8300        let ty = read_type(&mut cur)?;
8301        let not_null = match cur.u8()? {
8302            0 => false,
8303            1 => true,
8304            _ => return Err(invalid("nullability flag differs")),
8305        };
8306        fields.push(Field { name, ty, not_null });
8307    }
8308    let mut dictionaries = Vec::with_capacity(width);
8309    for _ in 0..width {
8310        dictionaries.push(match cur.u8()? {
8311            0 => None,
8312            1 => {
8313                let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
8314                let end = page
8315                    .offset
8316                    .checked_add(u64::from(page.length))
8317                    .ok_or_else(|| invalid("dictionary page offset overflow"))?;
8318                // A global dictionary covers a whole column, not one bounded stripe. Its lazy
8319                // payload is intentionally allowed to grow past `MAX_PAGE`; only ordinary column
8320                // pages are capped there. `Writer::finish` has already bounded this length by the
8321                // on-disk `u32`, and the range check below keeps it inside the file.
8322                if page.offset < HEADER || end > size {
8323                    return Err(invalid("dictionary page range is outside the file"));
8324                }
8325                Some(page)
8326            }
8327            _ => return Err(invalid("dictionary page tag differs")),
8328        });
8329    }
8330    let mut distincts = Vec::with_capacity(width);
8331    for _ in 0..width {
8332        distincts.push(match cur.u8()? {
8333            0 => None,
8334            1 => Some(cur.u64()?),
8335            _ => return Err(invalid("distinct count tag differs")),
8336        });
8337    }
8338    let rows = usize::try_from(cur.u64()?).map_err(|_| invalid("row count does not fit"))?;
8339    let count = cur.u32()? as usize;
8340    let mut stripes = Vec::with_capacity(count);
8341    let mut total = 0_usize;
8342    for _ in 0..count {
8343        let count = cur.u32()? as usize;
8344        if count == 0 || count > STRIPE_PARTS {
8345            return Err(invalid("stripe part count is outside its bound"));
8346        }
8347        let mut parts = Vec::with_capacity(count);
8348        let mut stripe_rows = 0_usize;
8349        for _ in 0..count {
8350            let rows = cur.u32()?;
8351            if rows == 0 {
8352                return Err(invalid("empty part"));
8353            }
8354            parts.push(rows);
8355            stripe_rows = stripe_rows
8356                .checked_add(rows as usize)
8357                .ok_or_else(|| invalid("stripe row count overflow"))?;
8358        }
8359        total =
8360            total.checked_add(stripe_rows).ok_or_else(|| invalid("stripe row count overflow"))?;
8361        let index = Span { offset: cur.u64()?, length: cur.u32()? };
8362        let section = index_section(count)?;
8363        let wanted = section
8364            .checked_mul(width)
8365            .and_then(|bytes| u32::try_from(bytes).ok())
8366            .ok_or_else(|| invalid("index page length overflow"))?;
8367        let end = index
8368            .offset
8369            .checked_add(u64::from(index.length))
8370            .ok_or_else(|| invalid("index page offset overflow"))?;
8371        if index.offset < HEADER || end > size || index.length != wanted {
8372            return Err(invalid("index page range is outside the file"));
8373        }
8374        let mut pages = Vec::with_capacity(width);
8375        for _ in 0..width {
8376            let offset = cur.u64()?;
8377            let length = cur.u32()?;
8378            let end = offset
8379                .checked_add(u64::from(length))
8380                .ok_or_else(|| invalid("page offset overflow"))?;
8381            if offset < HEADER || end > size || length as usize > MAX_PAGE {
8382                return Err(invalid("page range is outside the file"));
8383            }
8384            pages.push(Span { offset, length });
8385        }
8386        let mut memberships = vec![None; width];
8387        for (column, field) in fields.iter().enumerate() {
8388            if field.ty != LogicalType::Varchar || dictionaries[column].is_none() {
8389                continue;
8390            }
8391            let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
8392            let end = page
8393                .offset
8394                .checked_add(u64::from(page.length))
8395                .ok_or_else(|| invalid("membership page offset overflow"))?;
8396            if page.offset < HEADER || end > size || page.length as usize > MAX_PAGE {
8397                return Err(invalid("membership page range is outside the file"));
8398            }
8399            memberships[column] = Some(page);
8400        }
8401        let mut sieves = vec![None; width];
8402        for sieve in sieves.iter_mut().take(width) {
8403            match cur.u8()? {
8404                0 => continue,
8405                1 => {}
8406                _ => return Err(invalid("a sieve page has an unknown tag")),
8407            }
8408            let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
8409            let end = page
8410                .offset
8411                .checked_add(u64::from(page.length))
8412                .ok_or_else(|| invalid("sieve page offset overflow"))?;
8413            if page.offset < HEADER || end > size || page.length as usize > MAX_PAGE {
8414                return Err(invalid("sieve page range is outside the file"));
8415            }
8416            *sieve = Some(page);
8417        }
8418        let mut part_ranges = vec![None; width];
8419        for held in part_ranges.iter_mut().take(width) {
8420            match cur.u8()? {
8421                0 => continue,
8422                1 => {}
8423                _ => return Err(invalid("a part range page has an unknown tag")),
8424            }
8425            let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
8426            let end = page
8427                .offset
8428                .checked_add(u64::from(page.length))
8429                .ok_or_else(|| invalid("part range page offset overflow"))?;
8430            if page.offset < HEADER || end > size || page.length as usize > MAX_PAGE {
8431                return Err(invalid("part range page range is outside the file"));
8432            }
8433            *held = Some(page);
8434        }
8435        let mut ranges = Vec::with_capacity(width);
8436        for column in 0..width {
8437            let low = cur.bound()?;
8438            let high = cur.bound()?;
8439            let nulls = cur.u32()? as usize;
8440            if nulls > stripe_rows {
8441                return Err(invalid("null count exceeds stripe rows"));
8442            }
8443            let exact = cur.u8()? != 0;
8444            let sum = match cur.u8()? {
8445                0 => None,
8446                1 => Some(i128::from_le_bytes(
8447                    cur.take(16)?.try_into().map_err(|_| invalid("a stripe sum is truncated"))?,
8448                )),
8449                _ => return Err(invalid("a stripe sum has an unknown tag")),
8450            };
8451            // Files written before the ends of a decimal or a timestamp column carried their power
8452            // of ten hold a bare integer here, and that integer is the one the column holds, which
8453            // is what the power is over. So the type puts it back on the way in and an old file
8454            // prunes as well as a new one. A file that already wrote the power keeps it, because
8455            // this leaves anything that is not an integer alone.
8456            let ty = &fields.get(column).ok_or_else(|| invalid("a stripe range has no column"))?.ty;
8457            let low = low.map(|bound| scaled_as(bound, ty));
8458            let high = high.map(|bound| scaled_as(bound, ty));
8459            ranges.push(Range { low, high, nulls, exact, sum });
8460        }
8461        stripes.push(Stripe {
8462            rows: stripe_rows,
8463            parts,
8464            index,
8465            pages,
8466            memberships: Pages::from_slots(memberships)?,
8467            sieves: Pages::from_slots(sieves)?,
8468            part_ranges: Pages::from_slots(part_ranges)?,
8469            zone: Zone::from_ranges(ranges),
8470        });
8471    }
8472    if total != rows {
8473        return Err(invalid("table row count differs from stripes"));
8474    }
8475    // How many entries each column's synopsis lists, which is all a pair summary is checked against,
8476    // kept apart because the synopses themselves may be left in the file.
8477    let mut entry_counts = vec![0; width];
8478    let frequencies = if cur.done() {
8479        vec![None; width]
8480    } else {
8481        let frequency_magic = cur.take(8)?;
8482        let frequency_values = frequency_magic == FREQUENCIES;
8483        if !frequency_values && frequency_magic != FREQUENCIES_V2 {
8484            return Err(invalid("directory extension magic differs"));
8485        }
8486        if cur.u16()? as usize != width {
8487            return Err(invalid("frequency column count differs"));
8488        }
8489        let mut frequencies = Vec::with_capacity(width);
8490        for (field, entry_count) in fields.iter().zip(&mut entry_counts) {
8491            let start = cur.at;
8492            let summary = decode_summary(&mut cur, field, rows, frequency_values)?;
8493            *entry_count = summary.as_ref().map_or(0, |summary| summary.entries.len());
8494            frequencies.push(match (summary, stored_at) {
8495                (None, _) => None,
8496                (Some(summary), None) => Some(Frequencies::Held(summary)),
8497                (Some(_), Some(offset)) => Some(Frequencies::Stored {
8498                    span: Span {
8499                        offset: offset + start as u64,
8500                        length: u32::try_from(cur.at - start)
8501                            .map_err(|_| invalid("a frequency synopsis is too long"))?,
8502                    },
8503                    values: frequency_values,
8504                }),
8505            });
8506        }
8507        frequencies
8508    };
8509    // There are two optional trailing blocks now rather than one, so the reader dispatches on the
8510    // magic it finds rather than on where the bytes ran out. That is what lets the two arrive
8511    // independently: a format 22 directory ends here and has neither, a directory written before
8512    // the section table has only the clustering declaration, and each one still opens without a
8513    // rewrite. It is the G1 exit criterion, which is that a reader that knows about sections opens
8514    // a file that predates them and answers every query, only without the graph path.
8515    //
8516    // A repeated block is refused rather than allowed to win, because two clustering declarations
8517    // in one directory is a torn directory and the only question is which of them is the lie.
8518    let mut clustering = None;
8519    let mut sections = Vec::new();
8520    let mut pair_frequencies = Vec::new();
8521    let mut seen_pair_frequencies = false;
8522    let mut frequency_texts = vec![Vec::new(); width];
8523    let mut seen_frequency_texts = false;
8524    let mut host_groups = None;
8525    let mut seen_sections = false;
8526    let mut dictionary_payloads = Vec::new();
8527    let mut seen_payloads = false;
8528    // Zero until a section table says otherwise, which is what a format 22 table gets and what
8529    // makes every section stamp fail to match on one, because real generations start at one.
8530    let mut generation = 0;
8531    while !cur.done() {
8532        let mut tag = [0u8; 8];
8533        tag.copy_from_slice(cur.take(8)?);
8534        if &tag == PAIR_FREQUENCIES {
8535            if seen_pair_frequencies {
8536                return Err(invalid("directory names two pair frequency blocks"));
8537            }
8538            seen_pair_frequencies = true;
8539            let count = cur.u16()? as usize;
8540            if count > MAX_PAIR_FREQUENCIES {
8541                return Err(invalid("pair frequency count exceeds its bound"));
8542            }
8543            pair_frequencies = Vec::with_capacity(count);
8544            for _ in 0..count {
8545                let first = cur.u16()?;
8546                let second = cur.u16()?;
8547                let first_at = first as usize;
8548                let second_at = second as usize;
8549                if frequencies.get(first_at).and_then(Option::as_ref).is_none() {
8550                    return Err(invalid("pair frequency first column has no synopsis"));
8551                }
8552                let first_entries = entry_counts[first_at];
8553                if !matches!(fields.get(second_at), Some(field) if field.ty == LogicalType::Varchar)
8554                    || dictionaries.get(second_at).copied().flatten().is_none()
8555                {
8556                    return Err(invalid("pair frequency second column has no stable dictionary"));
8557                }
8558                if pair_frequencies
8559                    .iter()
8560                    .any(|held: &PairFrequencySummary| held.first == first && held.second == second)
8561                {
8562                    return Err(invalid("directory repeats a pair frequency summary"));
8563                }
8564                let omitted_max = cur.u64()?;
8565                if omitted_max > rows as u64 {
8566                    return Err(invalid("pair frequency omitted count exceeds the table"));
8567                }
8568                let entries_count = cur.u16()? as usize;
8569                if entries_count > FREQUENCY_ENTRIES {
8570                    return Err(invalid("pair frequency entry count exceeds its bound"));
8571                }
8572                let mut entries = Vec::with_capacity(entries_count);
8573                for _ in 0..entries_count {
8574                    let first_entry = cur.u16()?;
8575                    if first_entry as usize >= first_entries {
8576                        return Err(invalid("pair frequency anchor is outside its synopsis"));
8577                    }
8578                    let second = match cur.u8()? {
8579                        0 => None,
8580                        1 => Some(cur.u32()?),
8581                        _ => return Err(invalid("pair frequency string tag differs")),
8582                    };
8583                    let count = cur.u64()?;
8584                    if count == 0 || count > rows as u64 {
8585                        return Err(invalid("pair frequency count is outside the table"));
8586                    }
8587                    entries.push(PairFrequencyEntry { first_entry, second, count });
8588                }
8589                if entries.windows(2).any(|pair| pair[0].count < pair[1].count) {
8590                    return Err(invalid("pair frequency entries are not descending"));
8591                }
8592                pair_frequencies.push(PairFrequencySummary { first, second, entries, omitted_max });
8593            }
8594        } else if &tag == FREQUENCY_TEXTS {
8595            if seen_frequency_texts {
8596                return Err(invalid("directory names two frequency text blocks"));
8597            }
8598            seen_frequency_texts = true;
8599            let columns = cur.u16()? as usize;
8600            if columns > width {
8601                return Err(invalid("frequency text column count exceeds the schema"));
8602            }
8603            for _ in 0..columns {
8604                let column = cur.u16()? as usize;
8605                if !frequency_texts.get(column).is_some_and(Vec::is_empty) {
8606                    return Err(invalid("frequency text column is repeated or out of range"));
8607                }
8608                if !matches!(fields.get(column), Some(field) if field.ty == LogicalType::Varchar)
8609                    || dictionaries.get(column).copied().flatten().is_none()
8610                    || frequencies.get(column).and_then(Option::as_ref).is_none()
8611                {
8612                    return Err(invalid("frequency texts belong to a non-string synopsis"));
8613                }
8614                let count = cur.u16()? as usize;
8615                if count == 0 || count != entry_counts[column] {
8616                    return Err(invalid("frequency text count differs from its synopsis"));
8617                }
8618                let mut texts = Vec::with_capacity(count);
8619                for _ in 0..count {
8620                    texts.push(match cur.u8()? {
8621                        0 => None,
8622                        1 => {
8623                            let length = cur.u32()? as usize;
8624                            let bytes = cur.take(length)?.to_vec();
8625                            std::str::from_utf8(&bytes)
8626                                .map_err(|_| invalid("frequency text is not UTF-8"))?;
8627                            Some(bytes)
8628                        }
8629                        _ => return Err(invalid("frequency text tag differs")),
8630                    });
8631                }
8632                frequency_texts[column] = texts;
8633            }
8634        } else if &tag == HOST_GROUPS {
8635            if host_groups.is_some() {
8636                return Err(invalid("directory names two host group blocks"));
8637            }
8638            let column = cur.u16()? as usize;
8639            if !matches!(fields.get(column), Some(field) if field.ty == LogicalType::Varchar)
8640                || dictionaries.get(column).copied().flatten().is_none()
8641            {
8642                return Err(invalid("host groups belong to a non-string dictionary"));
8643            }
8644            let omitted_max = cur.u64()?;
8645            if omitted_max > rows as u64 {
8646                return Err(invalid("host group bound exceeds the table"));
8647            }
8648            let count = cur.u16()? as usize;
8649            if count > host::CAPACITY {
8650                return Err(invalid("host group count exceeds its bound"));
8651            }
8652            let mut entries = Vec::with_capacity(count);
8653            let mut bytes = 0_usize;
8654            for _ in 0..count {
8655                let host_len = cur.u32()? as usize;
8656                bytes =
8657                    bytes.checked_add(host_len).ok_or_else(|| invalid("host bytes overflow"))?;
8658                if bytes > host::BYTE_BUDGET {
8659                    return Err(invalid("host groups exceed their byte budget"));
8660                }
8661                let host = std::str::from_utf8(cur.take(host_len)?)
8662                    .map_err(|_| invalid("host is not UTF-8"))?
8663                    .to_owned();
8664                let count = cur.u64()?;
8665                if count == 0 || count > rows as u64 {
8666                    return Err(invalid("host group count exceeds the table"));
8667                }
8668                let bytes_sum = i128::from_le_bytes(
8669                    cur.take(16)?
8670                        .try_into()
8671                        .map_err(|_| invalid("host length sum is truncated"))?,
8672                );
8673                if bytes_sum < 0 {
8674                    return Err(invalid("host length sum is negative"));
8675                }
8676                let minimum_len = cur.u32()? as usize;
8677                bytes =
8678                    bytes.checked_add(minimum_len).ok_or_else(|| invalid("host bytes overflow"))?;
8679                if bytes > host::BYTE_BUDGET {
8680                    return Err(invalid("host groups exceed their byte budget"));
8681                }
8682                let minimum = std::str::from_utf8(cur.take(minimum_len)?)
8683                    .map_err(|_| invalid("host minimum is not UTF-8"))?
8684                    .to_owned();
8685                entries.push(host::HostEntry { host, count, bytes_sum, minimum });
8686            }
8687            if entries.windows(2).any(|pair| pair[0].count < pair[1].count)
8688                || entries.iter().any(|entry| entry.host.is_empty() || entry.minimum.is_empty())
8689            {
8690                return Err(invalid("host groups are not in certified order"));
8691            }
8692            host_groups = Some(host::HostSummary { column, omitted_max, entries });
8693        } else if &tag == CLUSTERING {
8694            if clustering.is_some() {
8695                return Err(invalid("directory names two clustering declarations"));
8696            }
8697            let bucket = Width::from_tag(cur.u8()?)
8698                .ok_or_else(|| invalid("clustering width tag differs"))?;
8699            let count = cur.u16()? as usize;
8700            let mut columns = Vec::with_capacity(count.min(fields.len()));
8701            for _ in 0..count {
8702                columns.push(u32::from(cur.u16()?));
8703            }
8704            // Through the constructor and not built by hand, so that a file claiming a column the
8705            // table does not have is caught at open rather than at the first scan that trusted it.
8706            clustering = Some(Clustering::new(columns, bucket, &fields).map_err(|_| {
8707                invalid("stored clustering declaration does not match the table it is on")
8708            })?);
8709        } else if &tag == SECTIONS {
8710            if seen_sections {
8711                return Err(invalid("directory names two section tables"));
8712            }
8713            seen_sections = true;
8714            generation = cur.u64()?;
8715            let count = cur.u16()? as usize;
8716            if count > MAX_SECTIONS {
8717                return Err(invalid("section count exceeds its bound"));
8718            }
8719            sections = Vec::with_capacity(count);
8720            // entry at a time: a malformed section entry is refused rather than turned into an
8721            // offset.
8722            for _ in 0..count {
8723                sections.push(Section::decode(cur.take(section::ENTRY_BYTES)?)?);
8724            }
8725            for held in &sections {
8726                let Some(end) = held.extent_page.checked_add(u64::from(held.extent_bytes)) else {
8727                    return Err(invalid("a section's extent table overflows the file"));
8728                };
8729                // The bound check is here and not in `section`, because only the caller knows how
8730                // big the file is. A section pointing past the end is a torn directory, and reading
8731                // the payload it names would be reading whatever else is at that offset.
8732                if held.extent_bytes != 0 && (held.extent_page < HEADER || end > size) {
8733                    return Err(invalid("a section's extent table is outside the file"));
8734                }
8735                if held.extents == 0 && held.extent_bytes != 0 {
8736                    return Err(invalid("a section with no extents names an extent table"));
8737                }
8738            }
8739        } else if &tag == DICTIONARY_PAYLOADS {
8740            if seen_payloads {
8741                return Err(invalid("directory names two dictionary payload blocks"));
8742            }
8743            seen_payloads = true;
8744            let count = cur.u16()? as usize;
8745            if count != fields.len() {
8746                return Err(invalid("dictionary payload block does not match the table's columns"));
8747            }
8748            dictionary_payloads = Vec::with_capacity(count);
8749            for _ in 0..count {
8750                let bytes = cur.u64()?;
8751                if bytes > size {
8752                    return Err(invalid("a dictionary payload is larger than the file"));
8753                }
8754                dictionary_payloads.push(bytes);
8755            }
8756        } else {
8757            return Err(invalid("directory extension magic differs"));
8758        }
8759    }
8760    if !cur.done() {
8761        return Err(invalid("directory has trailing bytes"));
8762    }
8763    Ok(Table {
8764        name,
8765        fields,
8766        stripes,
8767        rows,
8768        dictionaries,
8769        dictionary_payloads,
8770        distincts,
8771        frequencies,
8772        pair_frequencies,
8773        frequency_texts,
8774        host_groups,
8775        clustering,
8776        generation,
8777        sections,
8778    })
8779}
8780
8781/// A zone map's end, in the layout `rudb_common::bounds` defines. See [`Cursor::bound`].
8782fn put_bound(out: &mut Vec<u8>, bound: Option<&Bound>) -> Result<()> {
8783    bounds::put(out, bound)
8784}
8785
8786/// Which cascades are worth trying on a run of dictionary codes.
8787///
8788/// The exhaustive chooser encodes every candidate at every level of a cascade three deep and keeps
8789/// the smallest, which on a part of 1024 codes is around a hundred full encodes to decide something
8790/// three candidates were always going to win. It is the right default for a crate that does not
8791/// know what it is looking at. Here we do know. Codes are counted from zero in the order the values
8792/// were first seen, so a part of them is one value, or a narrow band, or a few long runs, and those
8793/// are constant, frame of reference and run length. Nothing else has ever come first on this data.
8794///
8795/// A dictionary of dictionary codes is the one candidate that can never pay, because the codes are
8796/// already the dictionary, and it is also the most expensive one to try. Below the top level the
8797/// streams are an RLE's run values and run lengths, which are integers in their own right with no
8798/// runs left in them, so only the two flat candidates go down there.
8799///
8800/// This is size given up for time on purpose, and the ablation is this chooser against
8801/// [`chooser::EXHAUSTIVE`] on the same file.
8802#[derive(Debug)]
8803struct Codes;
8804
8805impl chooser::Chooser for Codes {
8806    fn name(&self) -> &'static str {
8807        "codes"
8808    }
8809
8810    fn narrow_strings(
8811        &self,
8812        _values: &[&[u8]],
8813        offered: &[string::Kind],
8814        _depth: u8,
8815    ) -> Vec<string::Kind> {
8816        // Never reached, because nothing here encodes strings through the cascade. The trait asks
8817        // for it and the honest answer to a question we have no opinion on is the whole list.
8818        offered.to_vec()
8819    }
8820
8821    fn narrow_integers(
8822        &self,
8823        _values: &[i64],
8824        offered: &[integer::Kind],
8825        depth: u8,
8826    ) -> Vec<integer::Kind> {
8827        // The contract is a non empty subset, and a chunk that offers none of the three is a chunk
8828        // this has no opinion about rather than one that cannot be written.
8829        narrowed_to(Codes::keep(depth), offered)
8830    }
8831
8832    fn considers_integer(&self, kind: integer::Kind, depth: u8) -> bool {
8833        Codes::keep(depth).contains(&kind)
8834    }
8835}
8836
8837impl Codes {
8838    fn keep(depth: u8) -> &'static [integer::Kind] {
8839        if depth == 0 {
8840            &[integer::Kind::Constant, integer::Kind::Packed, integer::Kind::Rle]
8841        } else {
8842            &[integer::Kind::Constant, integer::Kind::Packed]
8843        }
8844    }
8845}
8846
8847/// The kinds of `offered` that are in `keep`, or all of `offered` when none of them are.
8848///
8849/// `Packed` applies to every chunk and both choosers keep it, so the fallback is never taken on a
8850/// chunk the cascade offers. It is there because the contract is a non empty subset and a chooser
8851/// that returned nothing would be a chunk that cannot be written. It is also why saying no to a kind
8852/// in `considers_integer` is safe: a kind that is never offered could only have been kept through
8853/// this fallback, and the fallback is never reached.
8854fn narrowed_to(keep: &[integer::Kind], offered: &[integer::Kind]) -> Vec<integer::Kind> {
8855    let narrowed: Vec<integer::Kind> =
8856        offered.iter().copied().filter(|kind| keep.contains(kind)).collect();
8857    if narrowed.is_empty() { offered.to_vec() } else { narrowed }
8858}
8859
8860/// Which cascades are worth trying on a part of plain integers.
8861///
8862/// Wider than [`Codes`] because the values are not codes and carry whatever shape the column has.
8863/// A timestamp column climbs, so delta is the one that matters and is the reason this exists at
8864/// all: three timestamp columns in ClickBench were coming out at exactly eight bytes a row with
8865/// nothing asked of them. The same three columns are why the stride is here, since a timestamp
8866/// loaded from a source that recorded whole seconds is microseconds with twenty zero bits under
8867/// every value. A column that is one value with a handful of exceptions is sparse. What is still
8868/// left out is the dictionary, for the same reason as in [`Codes`]: it is the most
8869/// expensive candidate to try and this file already puts the columns that want one through a
8870/// dictionary of their own before they ever reach here.
8871#[derive(Debug)]
8872struct Fixed;
8873
8874impl chooser::Chooser for Fixed {
8875    fn name(&self) -> &'static str {
8876        "fixed"
8877    }
8878
8879    fn narrow_strings(
8880        &self,
8881        _values: &[&[u8]],
8882        offered: &[string::Kind],
8883        _depth: u8,
8884    ) -> Vec<string::Kind> {
8885        offered.to_vec()
8886    }
8887
8888    fn narrow_integers(
8889        &self,
8890        _values: &[i64],
8891        offered: &[integer::Kind],
8892        depth: u8,
8893    ) -> Vec<integer::Kind> {
8894        narrowed_to(Fixed::keep(depth), offered)
8895    }
8896
8897    fn considers_integer(&self, kind: integer::Kind, depth: u8) -> bool {
8898        Fixed::keep(depth).contains(&kind)
8899    }
8900}
8901
8902impl Fixed {
8903    fn keep(depth: u8) -> &'static [integer::Kind] {
8904        if depth == 0 {
8905            &[
8906                integer::Kind::Constant,
8907                integer::Kind::Packed,
8908                integer::Kind::Delta,
8909                integer::Kind::Rle,
8910                integer::Kind::Sparse,
8911                integer::Kind::Strided,
8912            ]
8913        } else {
8914            &[integer::Kind::Constant, integer::Kind::Packed, integer::Kind::Delta]
8915        }
8916    }
8917}
8918
8919/// Every value of an integer part as an `i64`, or `None` for a part this cannot widen without
8920/// losing one.
8921///
8922/// `UBIGINT` is the only integer type left out, because half its range does not fit and a page that
8923/// silently wrapped would be worse than a page that stays plain. Booleans and strings are not
8924/// integers and have their own ways of being small.
8925fn widened(data: &Data) -> Option<Vec<i64>> {
8926    match data {
8927        Data::Int8(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
8928        Data::UInt8(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
8929        Data::Int16(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
8930        Data::UInt16(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
8931        Data::Int32(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
8932        Data::UInt32(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
8933        Data::Int64(values) => Some(values.to_vec()),
8934        _ => None,
8935    }
8936}
8937
8938/// An integer type a cascaded page can be read back into, and how to tell whether a value fits.
8939///
8940/// This exists so that the check and the conversion can be two loops instead of one. `TryFrom` puts
8941/// them together, which is the right shape for one value and the wrong one for a page: a fallible
8942/// conversion a value at a time is a branch a value at a time, the branch decides whether the loop
8943/// keeps going, and a loop like that is one no compiler will widen.
8944trait Narrow: Copy {
8945    /// How wide this type is, and what to add to a value to put its range at the bottom of a `u64`.
8946    ///
8947    /// Half the width for a signed type, which is what moves its smallest value to zero, and nothing
8948    /// for an unsigned one, whose smallest value is already there.
8949    const BIASED: (u32, u64);
8950
8951    /// The value narrowed, which the caller has already shown fits.
8952    fn narrow(value: i64) -> Self;
8953}
8954
8955/// The bits of `value` a `T` cannot hold, and zero when the value fits.
8956///
8957/// The question is asked this way round because the answers or together. A page fits when every
8958/// residue in it is zero, so the loop is an or into an accumulator and the decision is one test
8959/// after it, where asking whether each value is between a floor and a ceiling gives an answer that
8960/// does not combine and turns into a running minimum and maximum.
8961///
8962/// Biasing and shifting is what the answer is made of, rather than anything that reads more like the
8963/// question, because those are the operations a machine has four of. A 64 bit integer minimum is
8964/// AVX-512. So is a 64 bit arithmetic shift right, which is how the sign extension this could be
8965/// written as would have to be done. An add and a logical shift right are AVX2 and are on every
8966/// machine this runs on, so this is the form that gets four values a cycle instead of one.
8967///
8968/// Adding the bias moves the type's range to `0..=2^bits`, wrapping, so everything in range shifts
8969/// away to nothing and everything outside it leaves something behind. A negative value under an
8970/// unsigned type is caught by the same shift, because a negative `i64` read as a `u64` is enormous.
8971#[allow(clippy::cast_sign_loss, reason = "a residue is a bit pattern and not a number")]
8972fn residue<T: Narrow>(value: i64) -> u64 {
8973    let (bits, bias) = T::BIASED;
8974    (value as u64).wrapping_add(bias) >> bits
8975}
8976
8977/// Says a primitive integer narrows with `as`, and where the bottom of its range is.
8978///
8979/// `as` is a truncation and is the right operation here only because [`fit`] has already found every
8980/// residue zero, and it is what makes the second loop a narrowing store with no branch in it.
8981macro_rules! narrows {
8982    ($($ty:ty => $bias:expr),* $(,)?) => {$(
8983        impl Narrow for $ty {
8984            const BIASED: (u32, u64) = (<$ty>::BITS, $bias);
8985
8986            #[allow(
8987                clippy::cast_possible_truncation,
8988                clippy::cast_sign_loss,
8989                reason = "the caller has checked the bits this truncates away"
8990            )]
8991            fn narrow(value: i64) -> Self {
8992                value as Self
8993            }
8994        }
8995    )*};
8996}
8997
8998narrows! {
8999    i8 => 1 << 7,
9000    u8 => 0,
9001    i16 => 1 << 15,
9002    u16 => 0,
9003    i32 => 1 << 31,
9004    u32 => 0,
9005}
9006
9007/// Narrows a page's values, refusing the page if any of them does not fit.
9008///
9009/// The check first and the conversion second, rather than a fallible conversion a value at a time.
9010/// Both loops here are ones a compiler widens: [`residue`] is three instructions a lane and a
9011/// narrowing store is one. The version before this was a `TryFrom` and a `collect` into a `Result`,
9012/// which is a compare, a branch and a short circuit a value at a time, and on ClickBench 39 it was
9013/// seven percent of the query. The version after that kept a running minimum and maximum, which is
9014/// the obvious way to ask and needs a 64 bit integer minimum that AVX2 does not have, so it stayed
9015/// a value at a time and was still ten percent of the same query.
9016///
9017/// An empty page has nothing to refuse, which falls out of the accumulator starting at zero rather
9018/// than needing a case of its own.
9019fn fit<T: Narrow>(values: &[i64]) -> Result<Vec<T>> {
9020    let mut spilled = 0u64;
9021    for value in values {
9022        spilled |= residue::<T>(*value);
9023    }
9024    if spilled != 0 {
9025        return Err(invalid("page value is not of its type"));
9026    }
9027    Ok(values.iter().map(|value| T::narrow(*value)).collect())
9028}
9029
9030/// The same values back in the width the column is declared at.
9031///
9032/// A value that does not fit is a page that disagrees with the directory about what the column is,
9033/// which is a damaged file rather than a caller error, so it is refused rather than truncated.
9034fn narrowed(ty: &LogicalType, values: Vec<i64>) -> Result<Data> {
9035    Ok(match ty {
9036        LogicalType::TinyInt => Data::Int8(fit::<i8>(&values)?.into()),
9037        LogicalType::UTinyInt => Data::UInt8(fit::<u8>(&values)?.into()),
9038        LogicalType::SmallInt => Data::Int16(fit::<i16>(&values)?.into()),
9039        LogicalType::USmallInt => Data::UInt16(fit::<u16>(&values)?.into()),
9040        LogicalType::Integer | LogicalType::Date => Data::Int32(fit::<i32>(&values)?.into()),
9041        LogicalType::UInteger => Data::UInt32(fit::<u32>(&values)?.into()),
9042        LogicalType::BigInt
9043        | LogicalType::Timestamp
9044        | LogicalType::Time
9045        | LogicalType::TimeTz
9046        | LogicalType::TimestampTz
9047        | LogicalType::TimestampS
9048        | LogicalType::TimestampMs
9049        | LogicalType::TimestampNs => Data::Int64(values.into()),
9050        // A decimal is an integer of unscaled units, so the cascade reads back into whichever
9051        // integer the declared width says the column is stored as.
9052        LogicalType::Decimal { .. } => match ty.physical() {
9053            PhysicalType::Int16 => Data::Int16(fit::<i16>(&values)?.into()),
9054            PhysicalType::Int32 => Data::Int32(fit::<i32>(&values)?.into()),
9055            PhysicalType::Int64 => Data::Int64(values.into()),
9056            _ => return Err(invalid("cascade codec belongs to a decimal that is not an integer")),
9057        },
9058        _ => return Err(invalid("cascade codec belongs to a page that is not integers")),
9059    })
9060}
9061
9062/// How many bytes a part of this type costs written out plainly, which is what the cascade has to
9063/// beat before it is worth the decode.
9064fn plain_width(ty: &LogicalType) -> Option<usize> {
9065    Some(match ty {
9066        LogicalType::TinyInt | LogicalType::UTinyInt => 1,
9067        LogicalType::SmallInt | LogicalType::USmallInt => 2,
9068        LogicalType::Integer | LogicalType::UInteger | LogicalType::Date => 4,
9069        LogicalType::BigInt
9070        | LogicalType::Timestamp
9071        | LogicalType::Time
9072        | LogicalType::TimeTz
9073        | LogicalType::TimestampTz
9074        | LogicalType::TimestampS
9075        | LogicalType::TimestampMs
9076        | LogicalType::TimestampNs => 8,
9077        LogicalType::Decimal { .. } => match ty.physical() {
9078            PhysicalType::Int16 => 2,
9079            PhysicalType::Int32 => 4,
9080            PhysicalType::Int64 => 8,
9081            // The widest decimals are stored as `i128`, which the cascade does not widen into, so
9082            // they take the plain path and there is nothing here to compare against.
9083            _ => return None,
9084        },
9085        _ => return None,
9086    })
9087}
9088
9089/// A part's plain integers through the cascade, or `None` when nothing it offers is worth it.
9090///
9091/// What it has to beat is whatever the page would otherwise have cost, which is the bit packed form
9092/// where there is one and the plain width where there is not. Both are cheaper to decode than a
9093/// cascade, so a tie goes to them.
9094fn cascaded(
9095    flat: &Vector,
9096    ty: &LogicalType,
9097    packed: Option<&Packed<'_>>,
9098    settling: &mut Settling,
9099) -> Result<Option<Vec<u8>>> {
9100    let (Some(width), Some(data)) = (plain_width(ty), flat.data()) else { return Ok(None) };
9101    let Some(values) = widened(data) else { return Ok(None) };
9102    let plain = values.len().saturating_mul(width);
9103    let best = match packed {
9104        // The tag, the base, the word count and the words, which is what the codec 2 branch writes.
9105        Some(packed) => plain.min(21 + size_of_val(packed.words())),
9106        None => plain,
9107    };
9108    let out = settling.encode(&values)?;
9109    Ok((out.len() < best).then_some(out))
9110}
9111
9112/// How often the parts of one column in one stripe search the cascade again, in parts.
9113///
9114/// A stripe is 64 parts, so this is four searches a stripe where there were 64. The search is
9115/// what the cascade costs: on ClickBench `hits` the integer cascade was about a tenth of the load's
9116/// CPU and nearly all of it under `encode_pages`, trying six trees on every part to keep the one
9117/// the part before had kept.
9118const SEARCH_EVERY: usize = 16;
9119
9120/// What the parts of one column in one stripe have settled on in the integer cascade.
9121///
9122/// One of these per column per stripe, used in part order, so what a part comes out as depends on
9123/// the stripe and not on which thread wrote it or on how many there were.
9124#[derive(Debug, Default)]
9125struct Settling {
9126    /// The shape of the last part that was searched, with what its top level offered, its length
9127    /// and its row count, which is the size a replay is held to.
9128    shape: Option<Shape>,
9129    /// Parts replayed since that search.
9130    since: usize,
9131}
9132
9133impl Settling {
9134    /// A part's integers through the cascade, replaying the settled shape where there is one.
9135    ///
9136    /// The replay is kept when it held and came out no more than a quarter bigger a row than the
9137    /// part the shape was searched on. Past that the column has changed under it and the part is
9138    /// searched. A replay that stopped fitting partway has already searched from where it stopped,
9139    /// so its shape is taken as the new one rather than searched a second time.
9140    fn encode(&mut self, values: &[i64]) -> Result<Vec<u8>> {
9141        if let Some(shape) = self.shape.as_ref().filter(|_| self.since < SEARCH_EVERY) {
9142            let replay = chooser::Replay::new(&shape.kinds, &Fixed).expecting(&shape.offered);
9143            let out = integer::encode_with(values, &replay)?;
9144            if !replay.held() {
9145                self.settle(&out, values.len(), replay.first_offered())?;
9146                return Ok(out);
9147            }
9148            let grown = (out.len() as u128) * (shape.rows as u128) * 4;
9149            if grown <= (shape.len as u128) * (values.len() as u128) * 5 {
9150                self.since += 1;
9151                return Ok(out);
9152            }
9153        }
9154        // A replay of nothing is the search, and says what the top level offered on the way.
9155        let search = chooser::Replay::new(&[], &Fixed);
9156        let out = integer::encode_with(values, &search)?;
9157        self.settle(&out, values.len(), search.first_offered())?;
9158        Ok(out)
9159    }
9160
9161    fn settle(&mut self, out: &[u8], rows: usize, offered: Vec<integer::Kind>) -> Result<()> {
9162        let kinds = integer::shape(out)?;
9163        self.shape = Some(Shape { kinds, offered, len: out.len().max(1), rows: rows.max(1) });
9164        self.since = 0;
9165        Ok(())
9166    }
9167}
9168
9169/// A searched part's cascade, what its top level was offered, and what it came to.
9170#[derive(Debug)]
9171struct Shape {
9172    kinds: Vec<integer::Kind>,
9173    offered: Vec<integer::Kind>,
9174    len: usize,
9175    rows: usize,
9176}
9177
9178/// A part's dictionary codes through the integer cascade, or `None` when the cascade did not pay.
9179///
9180/// Until now this stream was a `u32` a row with nothing asked of it, and on ClickBench that was
9181/// 400,185,326 bytes for every one of the 28 varchar columns, the same count for `URL` as for a
9182/// column holding the empty string in nearly every row. Codes are dense integers counted from zero
9183/// and a part holds 1024 of them, which is the shape frame of reference is best at, and a column
9184/// with one value everywhere comes back a constant costing nothing per row rather than four bytes.
9185///
9186/// The result is taken only when it is smaller than the plain form. A cascade is allowed to come
9187/// out larger on a part whose codes are genuinely wide, `URL` has about sixty million distinct
9188/// values, and there is no reason to pay for the decode when it does.
9189/// A varchar page as one FSST layer, or `None` when it did not pay.
9190///
9191/// Until now a varchar page that neither the global dictionary nor the per page dictionary claimed
9192/// was written out raw: four bytes of offset a row and then the bytes. That is the right answer for
9193/// a page of values with nothing in common and the wrong one for a page of English, and a column of
9194/// comments is the case this exists for.
9195///
9196/// One layer and not the full string cascade, which is what the payload blocks of a global
9197/// dictionary go through. The cascade is a search: it encodes the page under every candidate it has
9198/// and recurses into the integer cascade for the lengths of each one, and on TPC-H `orders` that
9199/// took the write from 6.9 s to 48.3 s. It reads back no faster than the dictionary it replaced
9200/// either, 1.807 G instructions against 1.810 G for `select o_comment from orders`, because
9201/// unpicking a nest of layers a value at a time costs what the dictionary's payload block decode
9202/// cost. Raw pages of the same column read in 0.686 G, which says the whole of the difference is
9203/// what the page has to be put back together from.
9204///
9205/// FSST alone keeps most of what the cascade found and gives all of that back. Decoding it is one
9206/// pass over the payload into one buffer, the values are laid end to end in it the way the raw form
9207/// already lays them out, and what the reader hands a chunk is views over that buffer.
9208///
9209/// The page dictionary gets first refusal because it is cheaper still, and it wins on a page whose
9210/// values repeat. What is left for this is the page whose values mostly do not, which is exactly the
9211/// page that was being written raw.
9212///
9213/// Taken only when it comes out smaller than the raw form, so a page of incompressible values pays
9214/// nothing at read time for having been offered.
9215fn text_compressed(flat: &Vector) -> Result<Option<Vec<u8>>> {
9216    let mut values: Vec<&[u8]> = Vec::with_capacity(flat.len());
9217    let mut payload = 0_usize;
9218    for row in 0..flat.len() {
9219        // bytes_at: the rows were checked for UTF-8 on the way in, and checking them again here
9220        // was most of what the loop cost.
9221        let text = flat.bytes_at(row).unwrap_or(b"");
9222        payload = payload.saturating_add(text.len());
9223        values.push(text);
9224    }
9225    // What codec 0 writes for a varchar page: an offset a row and one more, then the payload.
9226    let plain = (flat.len() + 1).saturating_mul(4).saturating_add(payload);
9227    let Some(out) = string::encode_only(string::Kind::Fsst, &values)? else {
9228        return Ok(None);
9229    };
9230    Ok((out.len() < plain).then_some(out))
9231}
9232
9233fn encoded_codes(codes: &[u32]) -> Result<Option<Vec<u8>>> {
9234    let wide: Vec<i64> = codes.iter().map(|code| i64::from(*code)).collect();
9235    let coded = integer::encode_with(&wide, &Codes)?;
9236    let plain = codes.len().saturating_mul(size_of::<u32>());
9237    Ok((coded.len() < plain).then_some(coded))
9238}
9239
9240/// The validity of a page, which is a flag and then, when some rows are null and some are not, a
9241/// bit a row with the valid ones set.
9242fn push_validity(out: &mut Vec<u8>, flat: &Vector) {
9243    let flag = match flat.validity() {
9244        Validity::AllValid => 0,
9245        Validity::AllInvalid => 1,
9246        Validity::Mask(_) => 2,
9247    };
9248    out.push(flag);
9249    if flag == 2 {
9250        for group in (0..flat.len()).step_by(8) {
9251            let mut bits = 0_u8;
9252            for bit in 0..8 {
9253                if group + bit < flat.len() && !flat.is_null_at(group + bit) {
9254                    bits |= 1 << bit;
9255                }
9256            }
9257            out.push(bits);
9258        }
9259    }
9260}
9261
9262/// One part of a column coded against its global dictionary as a page, from the codes and the
9263/// validity [`push_validity`] wrote for it.
9264///
9265/// The codes go through the integer cascade when that comes out smaller than four bytes a code,
9266/// which on a column that repeats itself it nearly always does, and are written as they are when it
9267/// does not.
9268fn coded_page(codes: &[u32], validity: &[u8]) -> Result<Vec<u8>> {
9269    let coded = encoded_codes(codes)?;
9270    let mut out = Vec::with_capacity(
9271        1 + validity.len() + coded.as_ref().map_or(size_of_val(codes), Vec::len),
9272    );
9273    out.push(if coded.is_some() { 4 } else { 3 });
9274    out.extend_from_slice(validity);
9275    match coded {
9276        Some(coded) => out.extend_from_slice(&coded),
9277        None => {
9278            for &code in codes {
9279                put_u32(&mut out, code);
9280            }
9281        }
9282    }
9283    Ok(out)
9284}
9285
9286/// One part of one column as a page, for every column that is not coded against a global
9287/// dictionary. Those are built by [`coded_page`] from codes [`prepare`] handed out.
9288fn encode(vector: &Vector, settling: &mut Settling) -> Result<Vec<u8>> {
9289    let ty = vector.logical_type();
9290    // flatten: the file writer needs a uniform scalar page and does it once per loaded chunk.
9291    let flat = vector.flatten()?;
9292    let mut out = Vec::new();
9293    let dictionary = if ty == &LogicalType::Varchar { string_dictionary(&flat)? } else { None };
9294    let compressed_text = if dictionary.is_none() && ty == &LogicalType::Varchar {
9295        text_compressed(&flat)?
9296    } else {
9297        None
9298    };
9299    let packed_vector = if dictionary.is_none() { Some(flat.bit_packed()?) } else { None };
9300    let packed = packed_vector.as_ref().and_then(Vector::packed_parts);
9301    // Only where nothing else has claimed the page, which is the plain integer case. A packed part
9302    // is still on the table because the cascade has to beat it too: the bit pack takes a part only
9303    // when it halves it, so a column that shrinks by a third was coming out whole.
9304    let cascade =
9305        if dictionary.is_none() { cascaded(&flat, ty, packed.as_ref(), settling)? } else { None };
9306    out.push(if cascade.is_some() {
9307        5
9308    } else if dictionary.is_some() {
9309        1
9310    } else if compressed_text.is_some() {
9311        6
9312    } else if packed.is_some() {
9313        2
9314    } else {
9315        0
9316    });
9317    push_validity(&mut out, &flat);
9318    if let Some(cascade) = cascade {
9319        out.extend_from_slice(&cascade);
9320        return Ok(out);
9321    }
9322    if let Some(dictionary) = dictionary {
9323        out.extend_from_slice(&dictionary);
9324        return Ok(out);
9325    }
9326    if let Some(compressed_text) = compressed_text {
9327        out.extend_from_slice(&compressed_text);
9328        return Ok(out);
9329    }
9330    if let Some(packed) = packed {
9331        if packed.offset() != 0 {
9332            return Err(invalid("writer received a sliced packed vector"));
9333        }
9334        out.push(u8::try_from(packed.width()).map_err(|_| invalid("packed width overflow"))?);
9335        out.extend_from_slice(&packed.base().to_le_bytes());
9336        put_u32(
9337            &mut out,
9338            u32::try_from(packed.words().len()).map_err(|_| invalid("too many packed words"))?,
9339        );
9340        for word in packed.words() {
9341            put_u64(&mut out, *word);
9342        }
9343        return Ok(out);
9344    }
9345    let data = flat.data().ok_or_else(|| invalid("scalar column did not flatten"))?;
9346    match (ty, data) {
9347        (LogicalType::TinyInt, Data::Int8(values)) => {
9348            for value in &**values {
9349                out.extend_from_slice(&value.to_le_bytes());
9350            }
9351        }
9352        (LogicalType::UTinyInt, Data::UInt8(values)) => {
9353            for value in &**values {
9354                out.extend_from_slice(&value.to_le_bytes());
9355            }
9356        }
9357        (LogicalType::SmallInt, Data::Int16(values)) => {
9358            for value in &**values {
9359                out.extend_from_slice(&value.to_le_bytes());
9360            }
9361        }
9362        (LogicalType::USmallInt, Data::UInt16(values)) => {
9363            for value in &**values {
9364                out.extend_from_slice(&value.to_le_bytes());
9365            }
9366        }
9367        (LogicalType::UInteger, Data::UInt32(values)) => {
9368            for value in &**values {
9369                out.extend_from_slice(&value.to_le_bytes());
9370            }
9371        }
9372        (LogicalType::UBigInt, Data::UInt64(values)) => {
9373            for value in &**values {
9374                out.extend_from_slice(&value.to_le_bytes());
9375            }
9376        }
9377        (LogicalType::Integer | LogicalType::Date, Data::Int32(values)) => {
9378            for value in &**values {
9379                out.extend_from_slice(&value.to_le_bytes());
9380            }
9381        }
9382        (
9383            LogicalType::BigInt
9384            | LogicalType::Timestamp
9385            | LogicalType::Time
9386            | LogicalType::TimeTz
9387            | LogicalType::TimestampTz
9388            | LogicalType::TimestampS
9389            | LogicalType::TimestampMs
9390            | LogicalType::TimestampNs,
9391            Data::Int64(values),
9392        ) => {
9393            for value in &**values {
9394                out.extend_from_slice(&value.to_le_bytes());
9395            }
9396        }
9397        // A hugeint and a uuid are both the 128 bit lane, and a uuid's bits are the ones the rest of
9398        // the engine already carries it in, so nothing about the value changes on the way down.
9399        (LogicalType::HugeInt | LogicalType::Uuid, Data::Int128(values)) => {
9400            for value in &**values {
9401                out.extend_from_slice(&value.to_le_bytes());
9402            }
9403        }
9404        (LogicalType::UHugeInt, Data::UInt128(values)) => {
9405            for value in &**values {
9406                out.extend_from_slice(&value.to_le_bytes());
9407            }
9408        }
9409        // Plainly, in the IEEE bytes. The integer encodings do not apply to a float and none of the
9410        // float codecs is worth having before somebody has measured a corpus of them.
9411        (LogicalType::Float, Data::Float32(values)) => {
9412            for value in &**values {
9413                out.extend_from_slice(&value.to_le_bytes());
9414            }
9415        }
9416        (LogicalType::Double, Data::Float64(values)) => {
9417            for value in &**values {
9418                out.extend_from_slice(&value.to_le_bytes());
9419            }
9420        }
9421        // Three counts and not one number. Months, days and microseconds stay apart on disk because
9422        // they are apart in the value: a month is not a fixed number of days and a day is not a
9423        // fixed number of microseconds, which is the whole reason the type has three fields.
9424        (LogicalType::Interval, Data::Interval(values)) => {
9425            for (months, days, micros) in &**values {
9426                out.extend_from_slice(&months.to_le_bytes());
9427                out.extend_from_slice(&days.to_le_bytes());
9428                out.extend_from_slice(&micros.to_le_bytes());
9429            }
9430        }
9431        (LogicalType::Boolean, Data::Bool(values)) => {
9432            for value in &**values {
9433                out.push(u8::from(*value));
9434            }
9435        }
9436        // The unscaled integer and nothing else. Scale is a property of the column and it is in the
9437        // directory already, so writing it a value at a time would be paying for it twice.
9438        (LogicalType::Decimal { .. }, Data::Int16(values)) => {
9439            for value in &**values {
9440                out.extend_from_slice(&value.to_le_bytes());
9441            }
9442        }
9443        (LogicalType::Decimal { .. }, Data::Int32(values)) => {
9444            for value in &**values {
9445                out.extend_from_slice(&value.to_le_bytes());
9446            }
9447        }
9448        (LogicalType::Decimal { .. }, Data::Int64(values)) => {
9449            for value in &**values {
9450                out.extend_from_slice(&value.to_le_bytes());
9451            }
9452        }
9453        (LogicalType::Decimal { .. }, Data::Int128(values)) => {
9454            for value in &**values {
9455                out.extend_from_slice(&value.to_le_bytes());
9456            }
9457        }
9458        // A blob and a bit string go down the way a varchar does, because the layout is the same
9459        // one: an offset a value and then the bytes. What is not the same is that nothing here may
9460        // read the payload as text, which is why this arm asks the column for bytes rather than for
9461        // a string, and why the codecs above that do read text are all asked of a varchar by name.
9462        (LogicalType::Varchar | LogicalType::Blob | LogicalType::Bit, Data::Varlen(values)) => {
9463            let mut bytes = Vec::new();
9464            put_u32(&mut out, 0);
9465            for row in 0..vector.len() {
9466                let value = values.bytes(row).ok_or_else(|| invalid("string view is invalid"))?;
9467                bytes.extend_from_slice(value);
9468                put_u32(
9469                    &mut out,
9470                    u32::try_from(bytes.len())
9471                        .map_err(|_| invalid("string payload exceeds 4GiB"))?,
9472                );
9473            }
9474            out.extend_from_slice(&bytes);
9475        }
9476        _ => return Err(Error::not_implemented(format!("native page for {ty}"))),
9477    }
9478    Ok(out)
9479}
9480
9481fn put_varint(out: &mut Vec<u8>, mut value: u32) {
9482    while value >= 0x80 {
9483        out.push((value as u8 & 0x7f) | 0x80);
9484        value >>= 7;
9485    }
9486    out.push(value as u8);
9487}
9488
9489/// The distinct codes of one part, which is what a stripe's membership index is merged from.
9490fn unique_codes(codes: &[u32]) -> Vec<u32> {
9491    let mut unique = codes.to_vec();
9492    unique.sort_unstable();
9493    unique.dedup();
9494    unique
9495}
9496
9497/// The union of the sorted distinct codes of every part in a stripe.
9498///
9499/// Pairwise up a tree rather than one long list concatenated and sorted. Both are the same order of
9500/// work on paper and the tree is the one that does not sort what is already in order: sixty four
9501/// sorted lists become one in six passes over the values.
9502fn merged_codes(lists: Vec<Vec<u32>>) -> Vec<u32> {
9503    let mut lists = lists;
9504    while lists.len() > 1 {
9505        let mut next = Vec::with_capacity(lists.len().div_ceil(2));
9506        for pair in lists.chunks(2) {
9507            match pair {
9508                [left, right] => next.push(merged_pair(left, right)),
9509                [only] => next.push(only.clone()),
9510                _ => {}
9511            }
9512        }
9513        lists = next;
9514    }
9515    lists.pop().unwrap_or_default()
9516}
9517
9518fn merged_pair(left: &[u32], right: &[u32]) -> Vec<u32> {
9519    let mut out = Vec::with_capacity(left.len().saturating_add(right.len()));
9520    let mut at = 0;
9521    let mut to = 0;
9522    while at < left.len() && to < right.len() {
9523        match left[at].cmp(&right[to]) {
9524            Ordering::Less => {
9525                out.push(left[at]);
9526                at += 1;
9527            }
9528            Ordering::Greater => {
9529                out.push(right[to]);
9530                to += 1;
9531            }
9532            Ordering::Equal => {
9533                out.push(left[at]);
9534                at += 1;
9535                to += 1;
9536            }
9537        }
9538    }
9539    out.extend_from_slice(&left[at..]);
9540    out.extend_from_slice(&right[to..]);
9541    out
9542}
9543
9544/// The widest bounds and the total null count of a stripe, from the bounds of its parts.
9545///
9546/// A bound that is missing from any part is missing from the stripe, because a missing bound means
9547/// nothing is known and a stripe that holds an unknown cannot claim one.
9548fn merged_range(ranges: impl Iterator<Item = Range>) -> Range {
9549    let mut merged = Range::default();
9550    let mut first = true;
9551    for range in ranges {
9552        merged.nulls = merged.nulls.saturating_add(range.nulls);
9553        // Both of these have to survive every part, so one part that could not say anything makes
9554        // the stripe unable to say it either. A sum is dropped on overflow rather than wrapped,
9555        // which leaves the stripe with exact ends and no total, which is a true thing to say.
9556        merged.sum = match (merged.sum.take(), range.sum) {
9557            (Some(held), Some(next)) if !first => held.checked_add(next),
9558            (_, next) if first => next,
9559            _ => None,
9560        };
9561        merged.exact = if first { range.exact } else { merged.exact && range.exact };
9562        if first {
9563            merged.low = range.low;
9564            merged.high = range.high;
9565            first = false;
9566            continue;
9567        }
9568        merged.low = match (merged.low.take(), range.low) {
9569            (Some(held), Some(next)) => Some(held.smaller(next)),
9570            _ => None,
9571        };
9572        merged.high = match (merged.high.take(), range.high) {
9573            (Some(held), Some(next)) => Some(held.larger(next)),
9574            _ => None,
9575        };
9576    }
9577    merged
9578}
9579
9580/// One stripe's sieves for one column: the part count, a length for each part, then their bytes.
9581///
9582/// One page for the whole stripe rather than one per part, because a part's sieve is a few hundred
9583/// bytes and sixty four of those are sixty four directory entries and sixty four reads for something
9584/// a scan walks straight through. A part with no sieve writes a length of zero and costs four bytes.
9585/// `bound` cut down to [`PART_BOUND_BYTES`], still a bound of the side it was.
9586///
9587/// A prefix of a string sorts at or before the string, so cutting one down leaves a low end that is
9588/// still a low end. A high end has to go the other way, so the cut prefix is stepped up at the last
9589/// byte that can carry it, and a prefix of nothing but `0xFF` has no such byte and gives up the
9590/// bound rather than claiming one that is too small. Anything that is not a string is already a
9591/// fixed width and is left alone.
9592fn shortened(bound: Option<Bound>, high: bool) -> Option<Bound> {
9593    match bound {
9594        Some(Bound::Bytes(mut value)) if value.len() > PART_BOUND_BYTES => {
9595            value.truncate(PART_BOUND_BYTES);
9596            if !high {
9597                return Some(Bound::Bytes(value));
9598            }
9599            while let Some(last) = value.pop() {
9600                if last < u8::MAX {
9601                    value.push(last + 1);
9602                    return Some(Bound::Bytes(value));
9603                }
9604            }
9605            None
9606        }
9607        other => other,
9608    }
9609}
9610
9611/// The ranges of one column's parts of one stripe, as a page.
9612///
9613/// The two ends and the null count, and not `exact` or the total. Those two answer a `MIN` or a
9614/// `SUM` out of the directory, and the directory already answers those per stripe, where the same
9615/// number costs sixty times less to keep. What a part range is for is skipping the part, and
9616/// skipping needs the ends. So a range read back from here says it is not exact, which is true of a
9617/// string end that was cut down anyway.
9618fn encode_part_ranges(ranges: &[Range]) -> Result<Vec<u8>> {
9619    let mut out = Vec::new();
9620    put_u32(
9621        &mut out,
9622        u32::try_from(ranges.len()).map_err(|_| invalid("too many parts in a stripe"))?,
9623    );
9624    for range in ranges {
9625        put_bound(&mut out, shortened(range.low.clone(), false).as_ref())?;
9626        put_bound(&mut out, shortened(range.high.clone(), true).as_ref())?;
9627        put_u32(&mut out, u32::try_from(range.nulls).map_err(|_| invalid("null count overflow"))?);
9628    }
9629    Ok(out)
9630}
9631
9632/// The ranges one encoded page holds, one entry per part of the stripe.
9633fn decode_part_ranges(bytes: &[u8]) -> Result<Vec<Range>> {
9634    let mut cur = Cursor::new(bytes);
9635    let parts = cur.u32()? as usize;
9636    let mut out = Vec::new();
9637    for _ in 0..parts {
9638        let low = cur.bound()?;
9639        let high = cur.bound()?;
9640        let nulls = cur.u32()? as usize;
9641        out.push(Range { low, high, nulls, exact: false, sum: None });
9642    }
9643    Ok(out)
9644}
9645
9646fn encode_sieves<'a>(sieves: impl Iterator<Item = &'a Option<Sieve>>) -> Result<Vec<u8>> {
9647    let held: Vec<&Option<Sieve>> = sieves.collect();
9648    let mut out = Vec::new();
9649    put_u32(
9650        &mut out,
9651        u32::try_from(held.len()).map_err(|_| invalid("too many parts in a stripe"))?,
9652    );
9653    for sieve in &held {
9654        let length = sieve.as_ref().map_or(0, Sieve::len);
9655        put_u32(&mut out, u32::try_from(length).map_err(|_| invalid("sieve length overflow"))?);
9656    }
9657    // flatten: a part with no sieve wrote a length of zero above and contributes no bytes here.
9658    for sieve in held.into_iter().flatten() {
9659        out.extend_from_slice(&sieve.to_bytes());
9660    }
9661    Ok(out)
9662}
9663
9664/// The sieves one encoded page holds, one entry per part of the stripe.
9665///
9666/// A part whose bytes are not a sieve this version understands comes back as `None`, which is a part
9667/// that gets read. That is how a file written by a later version of the sieve stays readable rather
9668/// than being a corrupt page.
9669fn decode_sieves(bytes: &[u8]) -> Result<Vec<Option<Sieve>>> {
9670    let parts = u32::from_le_bytes(
9671        bytes
9672            .get(..4)
9673            .ok_or_else(|| invalid("sieve page is truncated"))?
9674            .try_into()
9675            .map_err(|_| invalid("sieve page is truncated"))?,
9676    ) as usize;
9677    let mut lengths = Vec::with_capacity(parts);
9678    for part in 0..parts {
9679        let at = 4 + part * 4;
9680        let field = bytes.get(at..at + 4).ok_or_else(|| invalid("sieve page is truncated"))?;
9681        lengths.push(u32::from_le_bytes(
9682            field.try_into().map_err(|_| invalid("sieve page is truncated"))?,
9683        ) as usize);
9684    }
9685    let mut at = 4 + parts * 4;
9686    let mut out = Vec::with_capacity(parts);
9687    for length in lengths {
9688        if length == 0 {
9689            out.push(None);
9690            continue;
9691        }
9692        let end = at.checked_add(length).ok_or_else(|| invalid("sieve page is truncated"))?;
9693        let field = bytes.get(at..end).ok_or_else(|| invalid("sieve page is truncated"))?;
9694        out.push(Sieve::from_bytes(field));
9695        at = end;
9696    }
9697    if at != bytes.len() {
9698        return Err(invalid("sieve page has trailing bytes"));
9699    }
9700    Ok(out)
9701}
9702
9703/// One stripe's membership index: the code count and then the codes as ascending deltas.
9704///
9705/// The codes have to be sorted and distinct already, which is what [`unique_codes`] and
9706/// [`merged_codes`] hand over. Anything else decodes as different codes, so neither of those two is
9707/// a step a caller can skip.
9708fn encode_membership(unique: &[u32]) -> Vec<u8> {
9709    let mut out = Vec::with_capacity(unique.len().saturating_mul(2).saturating_add(5));
9710    put_varint(&mut out, u32::try_from(unique.len()).unwrap_or(u32::MAX));
9711    let mut previous = 0;
9712    for (at, &code) in unique.iter().enumerate() {
9713        put_varint(&mut out, if at == 0 { code } else { code - previous });
9714        previous = code;
9715    }
9716    out
9717}
9718
9719fn take_varint(bytes: &[u8], at: &mut usize) -> Result<u32> {
9720    let mut value = 0_u32;
9721    for shift in (0..35).step_by(7) {
9722        let byte = *bytes.get(*at).ok_or_else(|| invalid("membership varint is truncated"))?;
9723        *at += 1;
9724        let part = u32::from(byte & 0x7f);
9725        if shift == 28 && part > 0x0f {
9726            return Err(invalid("membership varint overflow"));
9727        }
9728        value = value
9729            .checked_add(
9730                part.checked_shl(shift).ok_or_else(|| invalid("membership varint overflow"))?,
9731            )
9732            .ok_or_else(|| invalid("membership varint overflow"))?;
9733        if byte & 0x80 == 0 {
9734            return Ok(value);
9735        }
9736    }
9737    Err(invalid("membership varint is too long"))
9738}
9739
9740fn decode_membership(bytes: &[u8]) -> Result<Vec<u32>> {
9741    let mut at = 0;
9742    let count = take_varint(bytes, &mut at)? as usize;
9743    let mut codes = Vec::with_capacity(count);
9744    let mut previous = 0_u32;
9745    for index in 0..count {
9746        let delta = take_varint(bytes, &mut at)?;
9747        let code = if index == 0 {
9748            delta
9749        } else {
9750            previous.checked_add(delta).ok_or_else(|| invalid("membership code overflow"))?
9751        };
9752        if index > 0 && code <= previous {
9753            return Err(invalid("membership codes are not increasing"));
9754        }
9755        codes.push(code);
9756        previous = code;
9757    }
9758    if at != bytes.len() {
9759        return Err(invalid("membership page has trailing bytes"));
9760    }
9761    Ok(codes)
9762}
9763
9764fn string_dictionary(vector: &Vector) -> Result<Option<Vec<u8>>> {
9765    let mut by_text = HashMap::new();
9766    let mut values = Vec::new();
9767    let mut codes = Vec::with_capacity(vector.len());
9768    let mut plain_bytes = 0_usize;
9769    for row in 0..vector.len() {
9770        let text = vector.bytes_at(row).unwrap_or(b"");
9771        plain_bytes = plain_bytes.saturating_add(text.len());
9772        let code = match by_text.get(text) {
9773            Some(&code) => code,
9774            None => {
9775                let code = u32::try_from(values.len())
9776                    .map_err(|_| invalid("too many dictionary values"))?;
9777                by_text.insert(text, code);
9778                values.push(text);
9779                code
9780            }
9781        };
9782        codes.push(code);
9783    }
9784    let dictionary_bytes = values.iter().map(|value| value.len()).sum::<usize>();
9785    let encoded = 8_usize
9786        .saturating_add((values.len() + 1).saturating_mul(4))
9787        .saturating_add(dictionary_bytes)
9788        .saturating_add(codes.len().saturating_mul(4));
9789    let plain = (vector.len() + 1).saturating_mul(4).saturating_add(plain_bytes);
9790    if encoded >= plain {
9791        return Ok(None);
9792    }
9793    let mut out = Vec::with_capacity(encoded);
9794    put_u32(
9795        &mut out,
9796        u32::try_from(values.len()).map_err(|_| invalid("too many dictionary values"))?,
9797    );
9798    put_u32(
9799        &mut out,
9800        u32::try_from(dictionary_bytes).map_err(|_| invalid("dictionary payload exceeds 4GiB"))?,
9801    );
9802    let mut offset = 0_u32;
9803    put_u32(&mut out, offset);
9804    for value in &values {
9805        offset = offset
9806            .checked_add(
9807                u32::try_from(value.len()).map_err(|_| invalid("dictionary value is too long"))?,
9808            )
9809            .ok_or_else(|| invalid("dictionary payload exceeds 4GiB"))?;
9810        put_u32(&mut out, offset);
9811    }
9812    for value in values {
9813        out.extend_from_slice(value);
9814    }
9815    for code in codes {
9816        put_u32(&mut out, code);
9817    }
9818    Ok(Some(out))
9819}
9820
9821/// The room one closing dictionary takes under [`CLOSE_DICTIONARY_BYTES`], given back when dropped.
9822struct Room<'a, T> {
9823    state: &'a Mutex<(T, usize)>,
9824    finished: &'a Condvar,
9825    bytes: usize,
9826}
9827
9828impl<T> Drop for Room<'_, T> {
9829    fn drop(&mut self) {
9830        let mut held = self.state.lock().unwrap_or_else(PoisonError::into_inner);
9831        held.1 -= self.bytes;
9832        drop(held);
9833        self.finished.notify_all();
9834    }
9835}
9836
9837/// What [`Writer::close_dictionary`] builds for one column and [`Writer::close`] writes.
9838struct ClosedDictionary {
9839    distinct: u64,
9840    frequencies: FrequencySummary,
9841    texts: Vec<Option<Vec<u8>>>,
9842    hosts: Option<host::HostSummary>,
9843    encoded: EncodedDictionary,
9844    /// The bytes of the column's payload blocks, which are already in the file.
9845    payload: u64,
9846}
9847
9848struct EncodedDictionary {
9849    index: Vec<u8>,
9850    ranks: Vec<u8>,
9851    grams: Vec<u8>,
9852}
9853
9854/// Sorts codes into the byte order of the values they name, eight bytes of depth at a time.
9855///
9856/// # What the shape of the data does to a comparison sort
9857///
9858/// Distinct values against distinct prefixes, on the eight million row `hits`:
9859///
9860/// ```text
9861///   distinct   first 8   first 16   first 32   column
9862///  2,266,417        50      8,892    232,630   URL
9863///  2,346,025        49      8,534    204,060   Referer
9864///  1,357,764    81,362    348,340    861,579   Title
9865/// ```
9866///
9867/// Two and a quarter million URLs have fifty distinct first eight bytes between them, because they
9868/// all begin `http://` and then a host and there are not many hosts. So a sort that leads with
9869/// those eight bytes settles almost nothing on `URL` and `Referer`, whatever the comment on it used
9870/// to say, and almost every pair falls through to a comparison of whole values that agree for most
9871/// of their length. `Title` is free text and separates at eight bytes, which is why the design
9872/// looked right when it was written.
9873///
9874/// # What is done about it
9875///
9876/// Sort on eight bytes of the value at the current depth, held beside the code, and then take each
9877/// run that those eight bytes leave tied and sort it again on the next eight. A value is fetched
9878/// from the payload once per eight bytes of depth rather than once per comparison, and the sort
9879/// itself runs over an array of integers that is in cache rather than over pointers into a payload
9880/// that is hundreds of megabytes.
9881///
9882/// That is the whole trick, and it matters because the payload touch is the expensive part. The
9883/// bytes themselves are nearly free once the line is in cache, so reading eight at a time and
9884/// throwing away the ones that were not needed beats going back for each one.
9885///
9886/// # Why the length has to be carried
9887///
9888/// The eight bytes are padded with zero when the value has fewer than eight left, and a zero byte
9889/// can appear in a value, so equal keys do not mean equal bytes. What is true is that a value which
9890/// ran out inside the window is a prefix of any other value with the same key, and a prefix sorts
9891/// first, so how many of the eight bytes were real is the tie break and nothing further is needed.
9892/// A run is only worth another pass when all eight were real, because otherwise the run is one
9893/// value: a dictionary holds a value once.
9894fn sort_by_value<'a>(codes: &mut [u32], values: impl Fn(u32) -> &'a [u8]) {
9895    let mut work = vec![(0, codes.len(), 0)];
9896    let mut keyed: Vec<(u64, u8, u32)> = Vec::new();
9897    while let Some((from, to, depth)) = work.pop() {
9898        let part = &mut codes[from..to];
9899        keyed.clear();
9900        keyed.extend(part.iter().map(|&code| {
9901            let value = values(code);
9902            let rest = value.get(depth..).unwrap_or_default();
9903            (head(rest), rest.len().min(8) as u8, code)
9904        }));
9905        keyed.sort_unstable();
9906        for (slot, entry) in part.iter_mut().zip(keyed.iter()) {
9907            *slot = entry.2;
9908        }
9909        let mut start = 0;
9910        while start < keyed.len() {
9911            let (key, taken, _) = keyed[start];
9912            let mut end = start + 1;
9913            while end < keyed.len() && keyed[end].0 == key && keyed[end].1 == taken {
9914                end += 1;
9915            }
9916            if taken == 8 && end - start > 1 {
9917                work.push((from + start, from + end, depth + 8));
9918            }
9919            start = end;
9920        }
9921    }
9922}
9923
9924/// How few codes are worth sorting on more than one thread.
9925const PARALLEL_SORT_MIN: usize = 1 << 16;
9926
9927/// How many buckets a thread gets in [`sort_by_value_across`], so that a thread that drew a slow
9928/// bucket is not what the others wait for.
9929const BUCKETS_PER_WORKER: usize = 4;
9930
9931/// How many sampled codes stand for each bucket when the splitters are picked.
9932const SAMPLES_PER_BUCKET: usize = 32;
9933
9934/// [`sort_by_value`] over `workers` threads, with the same answer.
9935///
9936/// A sample sort. A sample of the codes is sorted and cut into as many equal runs as there are
9937/// buckets, and the values at the cuts are the splitters. Every code goes to the bucket its value
9938/// falls in by a binary search of the splitters, the buckets are laid end to end in splitter order,
9939/// and each bucket is then sorted on its own by whichever thread takes it. Every value in a bucket
9940/// sorts after every value in the bucket before, so the buckets sorted one by one are the codes
9941/// sorted.
9942///
9943/// The answer is the one [`sort_by_value`] gives down to the order of equal values, not only the
9944/// order of different ones. A global dictionary holds each value once, so there are none, but the
9945/// sort does not rely on it: equal values land in the same bucket in code order, which is the order
9946/// [`sort_by_value`] leaves them in, since the code is the last thing it sorts on.
9947///
9948/// On the 10m ClickBench sample the close sorts five columns of one to three and a half million
9949/// distinct values, one column at a time, and until this each sort ran on one thread while the
9950/// other thirty one waited for it.
9951fn sort_by_value_across<'a>(
9952    codes: &mut [u32],
9953    values: impl Fn(u32) -> &'a [u8] + Sync,
9954    workers: usize,
9955) {
9956    if workers <= 1 || codes.len() < PARALLEL_SORT_MIN {
9957        sort_by_value(codes, values);
9958        return;
9959    }
9960    let buckets = workers * BUCKETS_PER_WORKER;
9961    let wanted = buckets * SAMPLES_PER_BUCKET;
9962    let mut sample = (0..wanted).map(|at| codes[at * codes.len() / wanted]).collect::<Vec<_>>();
9963    sort_by_value(&mut sample, &values);
9964    let splitters =
9965        (1..buckets).map(|cut| values(sample[cut * sample.len() / buckets])).collect::<Vec<_>>();
9966    let values = &values;
9967    let splitters = &splitters;
9968    let per = codes.len().div_ceil(workers);
9969    // Which bucket each code goes to, a run of the codes per thread.
9970    let places = std::thread::scope(|scope| {
9971        codes
9972            .chunks(per)
9973            .map(|run| {
9974                scope.spawn(move || {
9975                    run.iter()
9976                        .map(|&code| {
9977                            let value = values(code);
9978                            splitters.partition_point(|splitter| *splitter <= value) as u32
9979                        })
9980                        .collect::<Vec<_>>()
9981                })
9982            })
9983            .collect::<Vec<_>>()
9984            .into_iter()
9985            .flat_map(|handle| {
9986                handle.join().unwrap_or_else(|panic| std::panic::resume_unwind(panic))
9987            })
9988            .collect::<Vec<_>>()
9989    });
9990    let mut starts = vec![0_usize; buckets + 1];
9991    for &place in &places {
9992        starts[place as usize + 1] += 1;
9993    }
9994    for bucket in 0..buckets {
9995        starts[bucket + 1] += starts[bucket];
9996    }
9997    let mut laid = vec![0_u32; codes.len()];
9998    let mut next = starts.clone();
9999    for (&code, &place) in codes.iter().zip(&places) {
10000        laid[next[place as usize]] = code;
10001        next[place as usize] += 1;
10002    }
10003    drop(places);
10004    let mut runs = Vec::with_capacity(buckets);
10005    let mut rest = laid.as_mut_slice();
10006    for bucket in 0..buckets {
10007        let (run, after) = rest.split_at_mut(starts[bucket + 1] - starts[bucket]);
10008        runs.push(run);
10009        rest = after;
10010    }
10011    // The largest buckets first, since they are taken from the back.
10012    runs.sort_by_key(|run| run.len());
10013    let queue = Mutex::new(runs);
10014    std::thread::scope(|scope| {
10015        for _ in 0..workers {
10016            scope.spawn(|| {
10017                loop {
10018                    let taken = queue.lock().unwrap_or_else(PoisonError::into_inner).pop();
10019                    let Some(run) = taken else { break };
10020                    sort_by_value(run, values);
10021                }
10022            });
10023        }
10024    });
10025    codes.copy_from_slice(&laid);
10026}
10027
10028/// The first eight bytes of a value as an integer that sorts the way the bytes sort.
10029fn head(bytes: &[u8]) -> u64 {
10030    let mut word = [0; 8];
10031    let take = bytes.len().min(8);
10032    word[..take].copy_from_slice(&bytes[..take]);
10033    u64::from_be_bytes(word)
10034}
10035
10036/// One column's dictionary page, which is its index and its sorted order.
10037///
10038/// The payload is not in it. Its blocks are in the file already, written as each was encoded, and
10039/// `places` says where, in block order. With `scattered` set the index records each block's start
10040/// and length, so a reader can find one wherever it went.
10041///
10042/// `scattered` false lays the blocks out the way a file written before format 26 has them, one
10043/// behind the next with only the ends recorded. Nothing in the writer asks for that any more. It is
10044/// kept because [`open_global_dictionary`] still reads those files and a reading path that nothing
10045/// can produce is a reading path nothing tests.
10046fn encode_global_dictionary(
10047    dictionary: &GlobalDictionary,
10048    order: &[(u64, u32)],
10049    places: &[Placed],
10050    scattered: bool,
10051) -> Result<EncodedDictionary> {
10052    let values = dictionary.values();
10053    if order.len() != values {
10054        return Err(invalid("global dictionary order does not cover its values"));
10055    }
10056    let blocks = values.div_ceil(TEXT_PAYLOAD_VALUES);
10057    if places.len() != blocks {
10058        return Err(invalid("global dictionary payload is not the blocks it says it is"));
10059    }
10060    if dictionary.grams.len() != blocks {
10061        return Err(invalid("global dictionary signatures do not cover its blocks"));
10062    }
10063    let (ranks, rank_ends) = encode_ranks(order, code_width(values))?;
10064    let rank_blocks = values.div_ceil(TEXT_RANK_BLOCK);
10065    let offset_bits = offset_width(&dictionary.ends);
10066    let payload_words = if scattered { 3 } else { 2 };
10067    let index_len = DICTIONARY_HEADER
10068        .checked_add(offset_bytes(values, offset_bits))
10069        .and_then(|len| len.checked_add(blocks.checked_mul(payload_words * 8)?))
10070        .and_then(|len| len.checked_add(rank_blocks.checked_mul(16)?))
10071        .and_then(|len| len.checked_add(8))
10072        .ok_or_else(|| invalid("global dictionary index length overflow"))?;
10073    let mut index = Vec::with_capacity(index_len);
10074    put_u32(
10075        &mut index,
10076        u32::try_from(values).map_err(|_| invalid("global dictionary has too many values"))?,
10077    );
10078    put_u32(&mut index, TEXT_PAYLOAD_VALUES as u32);
10079    put_u32(
10080        &mut index,
10081        u32::try_from(blocks).map_err(|_| invalid("global dictionary has too many blocks"))?,
10082    );
10083    let flag = (if scattered { DICTIONARY_SCATTERED } else { 0 })
10084        | DICTIONARY_GRAMS
10085        | DICTIONARY_WIDE_GRAMS;
10086    put_u32(&mut index, offset_bits as u32 | flag);
10087    encode_offsets(&dictionary.ends, offset_bits, &mut index)?;
10088    // Where each block is and how long it is, so a reader can find one. The stored blocks are
10089    // shorter than the decoded ones and by a different amount each, so their lengths are the one
10090    // thing the offsets above no longer say, and where they start is no longer arithmetic on the
10091    // block before once a block is written the moment it is encoded.
10092    let mut end = 0_u64;
10093    for place in places {
10094        if scattered {
10095            put_u64(&mut index, place.start);
10096            put_u64(&mut index, place.length);
10097        } else {
10098            end = end
10099                .checked_add(place.length)
10100                .ok_or_else(|| invalid("global dictionary payload overflow"))?;
10101            put_u64(&mut index, end);
10102        }
10103    }
10104    for place in places {
10105        put_u64(&mut index, place.hash);
10106    }
10107    // The same two lists for the sorted order. A rank block is packed at whatever width its own
10108    // heads need, so where one ends is no longer arithmetic on the block number.
10109    if rank_ends.len() != rank_blocks {
10110        return Err(invalid("global dictionary order is not the blocks it says it is"));
10111    }
10112    for end in &rank_ends {
10113        put_u64(&mut index, *end);
10114    }
10115    let mut at = 0_usize;
10116    for end in &rank_ends {
10117        let end = usize::try_from(*end).map_err(|_| invalid("global dictionary order overflow"))?;
10118        put_u64(&mut index, checksum(&ranks[at..end]));
10119        at = end;
10120    }
10121    let gram_len = blocks
10122        .checked_mul(TEXT_GRAM_BYTES)
10123        .ok_or_else(|| invalid("global dictionary signature count overflow"))?;
10124    let mut grams = Vec::with_capacity(gram_len);
10125    for block in &dictionary.grams {
10126        grams.extend_from_slice(block);
10127    }
10128    put_u64(&mut index, checksum(&grams));
10129    if index.len() != index_len {
10130        return Err(invalid("global dictionary index is not the length it was laid out for"));
10131    }
10132    Ok(EncodedDictionary { index, ranks, grams })
10133}
10134
10135/// How many blocks of the payload the shape is settled on.
10136///
10137/// Eight blocks is 8,192 values, which is the sample `chooser::Sampled` draws and is that size for
10138/// the same reason. They are spread across the dictionary rather than taken off the front, because
10139/// a dictionary is in the order values were first seen and the front of it is the first morsel of
10140/// the load.
10141const PAYLOAD_SAMPLE_BLOCKS: usize = 8;
10142
10143/// The shapes the payload encoder picks between.
10144///
10145/// Narrow on purpose. The exhaustive search encodes every candidate at every level and runs at two
10146/// to six megabytes a second on this data, which over the twelve gigabytes of dictionary `hits`
10147/// carries is about an hour of processor time, so it cannot be what a load does. Each of these
10148/// settles the outer level and the one below it, which is where almost all of that hour goes, and
10149/// leaves the levels under them to the exhaustive search where the chunks are small enough for it
10150/// to cost nothing.
10151///
10152/// Measured on the five ClickBench columns that have a dictionary worth the name, at 1,024 values a
10153/// block, against the exhaustive search over the same blocks:
10154///
10155/// | column | exhaustive | FRONT then LZ | LZ then FSST | LZ then PLAIN |
10156/// |---|---|---|---|---|
10157/// | 2 | 2.923 at 4.3 MB/s | 2.587 at 21.2 | 2.593 at 36.1 | 2.538 at 53.6 |
10158/// | 13 | 3.093 at 3.1 | 3.029 at 36.4 | 2.921 at 35.7 | 2.770 at 82.9 |
10159/// | 14 | 2.330 at 2.1 | 2.283 at 24.3 | 2.213 at 23.5 | 2.113 at 67.6 |
10160/// | 39 | 2.459 at 5.3 | 2.147 at 10.6 | 2.145 at 29.3 | 2.088 at 43.1 |
10161/// | 56 | 4.694 at 6.3 | 4.381 at 51.0 | 4.172 at 50.6 | 3.983 at 86.8 |
10162///
10163/// The best of the three per column is 98 percent of the exhaustive ratio for a tenth of the time.
10164/// `FSST` and `PLAIN` on their own are in the list as a floor rather than to win. `FSST` is the
10165/// right answer for text that does not share prefixes with its neighbours, and `PLAIN` is there so
10166/// that a column nothing compresses is found out in the sample and written at a gigabyte a second
10167/// rather than searched for an answer that does not exist.
10168fn payload_shapes() -> Vec<chooser::Settled> {
10169    let integers = vec![integer::Kind::Packed];
10170    [
10171        vec![string::Kind::Front, string::Kind::Lz],
10172        vec![string::Kind::Lz, string::Kind::Fsst],
10173        vec![string::Kind::Lz, string::Kind::Plain],
10174        vec![string::Kind::Fsst],
10175        vec![string::Kind::Plain],
10176    ]
10177    .into_iter()
10178    .map(|strings| chooser::Settled::new(strings, integers.clone()))
10179    .collect()
10180}
10181
10182/// Syncs the file, and counts the sync and how long it took as a publish wait when a load is being
10183/// profiled.
10184///
10185/// A wait rather than time, because the time is already in the publish span around it. What the
10186/// wait columns add is how much of publish was the device, which on the WSL2 disk of the gaming PC
10187/// is most of it: a sync there costs about two milliseconds (see `rudb_device_card`).
10188fn synced(file: &File, profile: Option<&LoadProfile>) -> Result<()> {
10189    let started = profile.map(|_| std::time::Instant::now());
10190    file.sync_all().map_err(io)?;
10191    if let (Some(profile), Some(started)) = (profile, started) {
10192        profile.waited(
10193            Stage::Publish,
10194            u64::try_from(started.elapsed().as_nanos()).unwrap_or(u64::MAX),
10195        );
10196    }
10197    Ok(())
10198}
10199
10200/// One sealed dictionary block on its way to being encoded outside the writer's lock.
10201///
10202/// Handed out by the merge that sealed it and encoded with the pages of the same stripe. See
10203/// [`GlobalDictionary::hand_out`].
10204#[derive(Debug)]
10205pub(crate) struct Unencoded {
10206    column: usize,
10207    at: usize,
10208    ends: Vec<u32>,
10209    bytes: Vec<u8>,
10210    shape: chooser::Settled,
10211}
10212
10213impl Unencoded {
10214    /// The encoded block and its signature.
10215    pub(crate) fn encode(&self) -> Result<EncodedBlock> {
10216        let values = block_values(&self.ends, &self.bytes);
10217        Ok((string::encode_with(&values, &self.shape)?, block_grams(&values)))
10218    }
10219
10220    /// The column and the block number the encoded block goes back to.
10221    pub(crate) fn place(&self) -> (usize, usize) {
10222        (self.column, self.at)
10223    }
10224}
10225
10226/// One encoded dictionary block and the signature of the values in it.
10227///
10228/// Boxed because it is carried around in things that are otherwise small.
10229pub(crate) type EncodedBlock = (Vec<u8>, Box<[u8; TEXT_GRAM_BYTES]>);
10230
10231/// The conservative four-byte substring signature of one block's values.
10232fn block_grams(values: &[&[u8]]) -> Box<[u8; TEXT_GRAM_BYTES]> {
10233    let mut grams = Box::new([0_u8; TEXT_GRAM_BYTES]);
10234    for value in values {
10235        for gram in value.windows(4) {
10236            for bit in gram_bits(gram, TEXT_GRAM_BYTES) {
10237                grams[bit / 8] |= 1 << (bit % 8);
10238            }
10239        }
10240    }
10241    grams
10242}
10243
10244/// The values of one block, given where each of them ends relative to the block.
10245fn block_values<'a>(ends: &[u32], bytes: &'a [u8]) -> Vec<&'a [u8]> {
10246    let mut out = Vec::with_capacity(ends.len());
10247    let mut from = 0;
10248    for &to in ends {
10249        out.push(&bytes[from..to as usize]);
10250        from = to as usize;
10251    }
10252    out
10253}
10254
10255/// Encodes every block still raw at the end of a load: the part block each column ends on and,
10256/// for a column too small to have settled a shape, every block it has.
10257///
10258/// Across threads, the way [`encode_ready`] does it. This ran one column at a time on the thread
10259/// closing the table, and a column that never settled a shape encodes each block by trying every
10260/// candidate, so on a million rows of `hits` it was most of the load's CPU on one core.
10261fn finish_dictionaries(dictionaries: &mut [Option<GlobalDictionary>]) -> Result<()> {
10262    for dictionary in dictionaries.iter_mut().flatten() {
10263        if !dictionary.early.is_empty() {
10264            return Err(Error::internal("a dictionary block handed out never came back"));
10265        }
10266        dictionary.seal_rest();
10267    }
10268    encode_waiting(dictionaries)?;
10269    // A block handed out and never given back leaves a gap nothing above would notice when it was
10270    // the last one, so the count is checked against the values as well.
10271    if dictionaries
10272        .iter()
10273        .flatten()
10274        .any(|dictionary| dictionary.encoded() != dictionary.values().div_ceil(TEXT_PAYLOAD_VALUES))
10275    {
10276        return Err(Error::internal("a dictionary block handed out never came back"));
10277    }
10278    Ok(())
10279}
10280
10281/// Encodes the waiting blocks of every dictionary across threads, and appends them to their columns
10282/// in order.
10283fn encode_waiting(dictionaries: &mut [Option<GlobalDictionary>]) -> Result<()> {
10284    let jobs = dictionaries
10285        .iter()
10286        .enumerate()
10287        .flat_map(|(column, held)| {
10288            (0..held.as_ref().map_or(0, |held| held.waiting.len())).map(move |at| (column, at))
10289        })
10290        .collect::<Vec<_>>();
10291    if jobs.is_empty() {
10292        return Ok(());
10293    }
10294    let one = |column: usize, at: usize| -> Result<(usize, usize, EncodedBlock)> {
10295        let held = dictionaries[column].as_ref().ok_or_else(|| Error::internal("no dictionary"))?;
10296        Ok((column, at, held.encode_waiting(at)?))
10297    };
10298    let workers = std::thread::available_parallelism()
10299        .map_or(1, usize::from)
10300        .min(MAX_FREQUENCY_WORKERS)
10301        .min(jobs.len());
10302    let made = if workers <= 1 {
10303        jobs.iter().map(|&(column, at)| one(column, at)).collect::<Result<Vec<_>>>()?
10304    } else {
10305        let next = AtomicUsize::new(0);
10306        let jobs = &jobs;
10307        let pieces = std::thread::scope(|scope| {
10308            (0..workers)
10309                .map(|_| {
10310                    scope.spawn(|| {
10311                        let mut mine = Vec::new();
10312                        loop {
10313                            let job = next.fetch_add(1, Atomic::Relaxed);
10314                            let Some(&(column, at)) = jobs.get(job) else { break };
10315                            mine.push(one(column, at)?);
10316                        }
10317                        Ok(mine)
10318                    })
10319                })
10320                .collect::<Vec<_>>()
10321                .into_iter()
10322                .map(|handle| {
10323                    handle
10324                        .join()
10325                        .map_err(|_| Error::internal("a dictionary encode worker panicked"))?
10326                })
10327                .collect::<Result<Vec<_>>>()
10328        })?;
10329        pieces.into_iter().flatten().collect()
10330    };
10331    let mut done: Vec<Vec<(usize, EncodedBlock)>> =
10332        (0..dictionaries.len()).map(|_| Vec::new()).collect();
10333    for (column, at, bytes) in made {
10334        done[column].push((at, bytes));
10335    }
10336    for (column, mut made) in done.into_iter().enumerate() {
10337        if made.is_empty() {
10338            continue;
10339        }
10340        let Some(held) = dictionaries[column].as_mut() else { continue };
10341        made.sort_by_key(|(at, _)| *at);
10342        let waiting = std::mem::take(&mut held.waiting);
10343        for ((at, _), (_, block)) in waiting.into_iter().zip(made) {
10344            if held.encoded() != at {
10345                return Err(Error::internal("a dictionary block was encoded out of order"));
10346            }
10347            held.push_block(block);
10348        }
10349    }
10350    Ok(())
10351}
10352
10353/// Which of [`payload_shapes`] comes out smallest over a sample of the blocks.
10354///
10355/// Every shape is encoded over the same sample and the smallest wins, which is the exhaustive
10356/// search moved up a level: over shapes of a column rather than over candidates of a chunk. The
10357/// sample is spread across the dictionary so that the first and last blocks are both in it, because
10358/// a dictionary written in first seen order has its common values at the front and its long tail at
10359/// the back, and those do not compress alike. Which blocks those are is
10360/// [`GlobalDictionary::seal`]'s to decide, because by the time this is called the rest of them have
10361/// been encoded and the raw bytes are gone.
10362fn settle_shape(sample: &[Vec<&[u8]>]) -> Result<chooser::Settled> {
10363    let mut best: Option<(chooser::Settled, usize)> = None;
10364    for shape in payload_shapes() {
10365        let mut size = 0;
10366        for block in sample {
10367            size += string::encode_with(block, &shape)?.len();
10368        }
10369        if best.as_ref().is_none_or(|(_, smallest)| size < *smallest) {
10370            best = Some((shape, size));
10371        }
10372    }
10373    best.map(|(shape, _)| shape)
10374        .ok_or_else(|| invalid("no shape applies to a global dictionary payload"))
10375}
10376
10377/// The sorted order laid out the way a reader reads it, in blocks of [`TEXT_RANK_BLOCK`] entries.
10378///
10379/// Each block holds its heads first and then its codes, rather than pairing them, because a search
10380/// asks for a head at every probe and for a code about once a search. Keeping the heads together
10381/// means a probe touches eight bytes of a block rather than twelve spread over it, and the last few
10382/// probes of a search, which are the ones that land in the same block, touch the same cache line.
10383fn encode_ranks(order: &[(u64, u32)], code_bits: usize) -> Result<(Vec<u8>, Vec<u64>)> {
10384    let mut out = Vec::with_capacity(order.len() * 4);
10385    let mut ends = Vec::with_capacity(order.len().div_ceil(TEXT_RANK_BLOCK));
10386    let mut heads = Vec::with_capacity(TEXT_RANK_BLOCK);
10387    let mut codes = Vec::with_capacity(TEXT_RANK_BLOCK);
10388    for block in order.chunks(TEXT_RANK_BLOCK) {
10389        // The order is sorted by value and a head is a prefix of a value, so the heads of a block
10390        // rise, the smallest is the first and the largest is the last.
10391        let base = block.first().map_or(0, |&(head, _)| head);
10392        let span = block.last().map_or(0, |&(head, _)| head.wrapping_sub(base));
10393        let width = (u64::BITS - span.leading_zeros()) as usize;
10394        heads.clear();
10395        codes.clear();
10396        for &(head, code) in block {
10397            heads.push(head.wrapping_sub(base));
10398            codes.push(u64::from(code));
10399        }
10400        put_u64(&mut out, base);
10401        out.push(width as u8);
10402        bitpack::pack_tail(&heads, width, &mut out)
10403            .map_err(|_| invalid("global dictionary heads do not pack"))?;
10404        bitpack::pack_tail(&codes, code_bits, &mut out)
10405            .map_err(|_| invalid("global dictionary codes do not pack"))?;
10406        ends.push(out.len() as u64);
10407    }
10408    Ok((out, ends))
10409}
10410
10411/// Opens a column's global dictionary, which reads its index and none of its payload.
10412///
10413/// `keep_budget` is how many decoded payload bytes this dictionary may hold on to, and every
10414/// caller bar the test of the ceiling passes [`TEXT_KEEP_BUDGET`]. It is a parameter rather than
10415/// the constant read where it is used because a test of a ceiling that cannot be moved has to build
10416/// a quarter of a gigabyte of dictionary to reach it.
10417fn open_global_dictionary(
10418    file: Arc<File>,
10419    page: Page,
10420    ty: &LogicalType,
10421    keep_budget: usize,
10422) -> Result<Vector> {
10423    if ty != &LogicalType::Varchar {
10424        return Err(invalid("global dictionary belongs to a non-string column"));
10425    }
10426    let mut header = [0; DICTIONARY_HEADER];
10427    read_at(&file, page.offset, &mut header)?;
10428    let count = u32::from_le_bytes(header[0..4].try_into().expect("four bytes")) as usize;
10429    let per_block = u32::from_le_bytes(header[4..8].try_into().expect("four bytes")) as usize;
10430    let blocks = u32::from_le_bytes(header[8..12].try_into().expect("four bytes")) as usize;
10431    let width = u32::from_le_bytes(header[12..16].try_into().expect("four bytes"));
10432    let scattered = width & DICTIONARY_SCATTERED != 0;
10433    let has_grams = width & DICTIONARY_GRAMS != 0;
10434    let gram_width =
10435        if width & DICTIONARY_WIDE_GRAMS != 0 { TEXT_GRAM_BYTES } else { NARROW_GRAM_BYTES };
10436    let offset_bits = (width & !DICTIONARY_FLAGS) as usize;
10437    if per_block != TEXT_PAYLOAD_VALUES {
10438        return Err(invalid("global dictionary block width differs"));
10439    }
10440    if blocks != count.div_ceil(TEXT_PAYLOAD_VALUES) {
10441        return Err(invalid("global dictionary block count differs from its value count"));
10442    }
10443    if offset_bits > u32::BITS as usize {
10444        return Err(invalid("global dictionary packs offsets past a payload"));
10445    }
10446    let offset_len = offset_bytes(count, offset_bits);
10447    // The sorted order is kept out of the index on purpose. The index is read and checksummed in
10448    // full the moment the column is first touched, and the order is half again the size of the
10449    // offsets, so putting it there would make every query that reads a string column pay for a
10450    // search that most of them never make.
10451    let ranks = count;
10452    let rank_blocks = ranks.div_ceil(TEXT_RANK_BLOCK);
10453    // Three words a payload block, for where it starts, how long it is and what it hashes to, or
10454    // two of them on a file that has the blocks back to back and needs no start. Two a rank block
10455    // either way, since those are still one run.
10456    let payload_words = if scattered { 3 } else { 2 };
10457    let hash_len = blocks
10458        .checked_mul(payload_words * 8)
10459        .and_then(|len| len.checked_add(rank_blocks.checked_mul(16)?))
10460        .and_then(|len| len.checked_add(usize::from(has_grams) * 8))
10461        .ok_or_else(|| invalid("global dictionary block count overflow"))?;
10462    let gram_len = if has_grams {
10463        blocks
10464            .checked_mul(gram_width)
10465            .ok_or_else(|| invalid("global dictionary signature count overflow"))?
10466    } else {
10467        0
10468    };
10469    let index_len = DICTIONARY_HEADER
10470        .checked_add(offset_len)
10471        .and_then(|len| len.checked_add(hash_len))
10472        .ok_or_else(|| invalid("global dictionary header overflow"))?;
10473    if index_len > page.length as usize {
10474        return Err(invalid("global dictionary offset index exceeds its page"));
10475    }
10476    let mut index = vec![0; index_len];
10477    index[..DICTIONARY_HEADER].copy_from_slice(&header);
10478    read_at(&file, page.offset + DICTIONARY_HEADER as u64, &mut index[DICTIONARY_HEADER..])?;
10479    if checksum(&index) != page.hash {
10480        return Err(invalid("global dictionary index checksum differs"));
10481    }
10482    let offsets = index[DICTIONARY_HEADER..DICTIONARY_HEADER + offset_len].to_vec();
10483    let word_end = index_len - usize::from(has_grams) * 8;
10484    let gram_hash = has_grams
10485        .then(|| u64::from_le_bytes(index[word_end..index_len].try_into().expect("eight bytes")));
10486    let mut words = index[DICTIONARY_HEADER + offset_len..word_end]
10487        .chunks_exact(8)
10488        .map(|part| u64::from_le_bytes(part.try_into().expect("eight bytes")))
10489        .collect::<Vec<_>>();
10490    let mut rest = words.split_off(blocks * payload_words);
10491    let rank_hashes = rest.split_off(rank_blocks);
10492    let rank_ends = rest;
10493    // A rank block packs its heads at whatever width its own values need, so its length is no longer
10494    // arithmetic on the block number and the reader has to be told where each one ends.
10495    if rank_ends.windows(2).any(|pair| pair[0] >= pair[1]) {
10496        return Err(invalid("global dictionary order blocks do not rise"));
10497    }
10498    let rank_len = usize::try_from(rank_ends.last().copied().unwrap_or_default())
10499        .map_err(|_| invalid("global dictionary rank overflow"))?;
10500    let body_len = index_len
10501        .checked_add(rank_len)
10502        .ok_or_else(|| invalid("global dictionary header overflow"))?;
10503    if body_len > page.length as usize {
10504        return Err(invalid("global dictionary order exceeds its page"));
10505    }
10506    let gram_end = body_len
10507        .checked_add(gram_len)
10508        .ok_or_else(|| invalid("global dictionary signature length overflow"))?;
10509    if gram_end > page.length as usize {
10510        return Err(invalid("global dictionary signatures exceed their page"));
10511    }
10512    let grams = gram_hash.map(|hash| NativeGrams {
10513        start: page.offset + body_len as u64,
10514        length: gram_len,
10515        width: gram_width,
10516        hash,
10517        verdicts: Mutex::new(Vec::new()),
10518    });
10519    let hashes = words.split_off(blocks * (payload_words - 1));
10520    let (starts, lengths) = if scattered {
10521        let mut starts = Vec::with_capacity(blocks);
10522        let mut lengths = Vec::with_capacity(blocks);
10523        for pair in words.chunks_exact(2) {
10524            starts.push(pair[0]);
10525            lengths.push(pair[1]);
10526        }
10527        (starts, lengths)
10528    } else {
10529        // A file written before the blocks said where they were has them behind one another at the
10530        // end of the page, so the base is where the sorted order stops and each end is the start of
10531        // the one after it. Turning them round here is what lets everything below take one shape.
10532        let base = page.offset + gram_end as u64;
10533        let mut starts = Vec::with_capacity(blocks);
10534        let mut lengths = Vec::with_capacity(blocks);
10535        let mut at = 0_u64;
10536        for &end in &words {
10537            let len = end
10538                .checked_sub(at)
10539                .ok_or_else(|| invalid("global dictionary block ends before it starts"))?;
10540            starts.push(base + at);
10541            lengths.push(len);
10542            at = end;
10543        }
10544        (starts, lengths)
10545    };
10546    // What the offsets bound is the decoded payload, and what the page length counts is the stored
10547    // one, so on a format 26 file the block lengths adding up to the rest of the page is the one
10548    // thing that ties the index to the page. From format 27 the blocks are written during the load
10549    // and the page is only the index and the order, so there the most that can be said is that
10550    // every block is somewhere in the file past its header.
10551    let stored_len = page.length as u64 - gram_end as u64;
10552    if scattered && stored_len == 0 {
10553        let size = file.metadata().map_err(io)?.len();
10554        let inside = starts.iter().zip(&lengths).all(|(&start, &len)| {
10555            start >= HEADER && start.checked_add(len).is_some_and(|end| end <= size)
10556        });
10557        if !inside {
10558            return Err(invalid("global dictionary block lies outside the file"));
10559        }
10560    } else if lengths.iter().try_fold(0_u64, |sum, len| sum.checked_add(*len)) != Some(stored_len) {
10561        return Err(invalid("global dictionary blocks do not bound the payload"));
10562    }
10563    Vector::external_text(
10564        LogicalType::Varchar,
10565        Arc::new(NativeText {
10566            file,
10567            values: count,
10568            offsets,
10569            offset_bits,
10570            value_ends: OnceLock::new(),
10571            value_lens: OnceLock::new(),
10572            ends_asked: AtomicUsize::new(0),
10573            ranks,
10574            rank_at: page.offset + index_len as u64,
10575            rank_ends,
10576            rank_hashes,
10577            rank_blocks: (0..rank_blocks).map(|_| OnceLock::new()).collect(),
10578            code_bits: code_width(count),
10579            code_ranks: OnceLock::new(),
10580            starts,
10581            lengths,
10582            hashes,
10583            grams,
10584            blocks: (0..blocks).map(|_| OnceLock::new()).collect(),
10585            keep_budget,
10586            payload_kept: AtomicUsize::new(0),
10587            swept: (0..blocks).map(|_| AtomicBool::new(false)).collect(),
10588            searched: Mutex::new(HashMap::new()),
10589        }),
10590    )
10591}
10592
10593/// What a stored page is, without decoding a value out of it.
10594///
10595/// Two layers, and both of them belong in the answer. The codec byte at the front of every page is
10596/// the format's own choice, and it is what says whether the column came back as codes into a table
10597/// wide dictionary, as a bit packed page, as an encoding cascade or as the bytes themselves. Under
10598/// the cascade codecs there is a second choice the encoder made per chunk, and that is what
10599/// [`integer::describe`] and [`string::describe`] already write out as `DICT(PACKED, PACKED)`.
10600///
10601/// This mirrors the tags [`decode`] reads and has to be kept beside it. A page whose header this
10602/// cannot walk comes back as text rather than as an error, because a caller asking what a file
10603/// looks like is usually asking because something is wrong with it, and a report that stops at the
10604/// first bad page is a report that says nothing about the other nine hundred.
10605fn page_encoding(ty: &LogicalType, rows: usize, bytes: &[u8]) -> String {
10606    /// The page header is the codec, the validity tag and, for a page that stores a mask, the mask.
10607    fn cascade_at(rows: usize, bytes: &[u8]) -> Result<(u8, usize)> {
10608        let mut cur = Cursor::new(bytes);
10609        let codec = cur.u8()?;
10610        if cur.u8()? == 2 {
10611            cur.take(rows.div_ceil(8))?;
10612        }
10613        Ok((codec, cur.at))
10614    }
10615    let Ok((codec, at)) = cascade_at(rows, bytes) else {
10616        return "UNREADABLE".to_string();
10617    };
10618    let tail = &bytes[at..];
10619    let described = |described: Result<String>| described.unwrap_or_else(|_| "UNREADABLE".into());
10620    match codec {
10621        0 => match ty {
10622            LogicalType::Varchar | LogicalType::Blob => "PLAIN".to_string(),
10623            _ => "FIXED".to_string(),
10624        },
10625        1 => "DICT(PLAIN)".to_string(),
10626        2 => "FOR+BITPACK".to_string(),
10627        3 => "TABLE DICT".to_string(),
10628        4 => format!("TABLE DICT({})", described(integer::describe(tail))),
10629        5 => described(integer::describe(tail)),
10630        6 => described(string::describe(tail)),
10631        other => format!("CODEC {other}"),
10632    }
10633}
10634
10635/// Selected stable dictionary codes from one page.
10636///
10637/// Pair-frequency construction needs at most the bounded heavy-hitter rows. Reading those code
10638/// positions directly avoids materializing every code in each part that contains a candidate.
10639fn decode_selected_stable_codes(
10640    rows: usize,
10641    bytes: &[u8],
10642    positions: &[usize],
10643    out: &mut Vec<Option<u32>>,
10644) -> Result<bool> {
10645    if positions.windows(2).any(|pair| pair[0] >= pair[1])
10646        || positions.last().is_some_and(|&position| position >= rows)
10647    {
10648        return Err(invalid("selected code positions are not sorted and in range"));
10649    }
10650    let mut cur = Cursor::new(bytes);
10651    let codec = cur.u8()?;
10652    if codec != 3 && codec != 4 {
10653        return Ok(false);
10654    }
10655    let flag = cur.u8()?;
10656    let mask = match flag {
10657        0 | 1 => None,
10658        2 => {
10659            let at = cur.at;
10660            let len = rows.div_ceil(8);
10661            cur.take(len)?;
10662            Some((at, len))
10663        }
10664        _ => return Err(invalid("page validity tag differs")),
10665    };
10666    let valid = |row: usize| match flag {
10667        0 => true,
10668        1 => false,
10669        2 => mask.is_some_and(|(at, _)| bytes[at + row / 8] >> (row % 8) & 1 == 1),
10670        _ => unreachable!("the validity tag was checked"),
10671    };
10672    if codec == 4 {
10673        let wide = integer::decode_selected(&bytes[cur.at..], positions)?;
10674        for (&row, code) in positions.iter().zip(wide) {
10675            let code = u32::try_from(code).map_err(|_| invalid("code is not a code"))?;
10676            out.push(valid(row).then_some(code));
10677        }
10678        return Ok(true);
10679    }
10680    let codes_at = cur.at;
10681    let codes_len = rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?;
10682    cur.take(codes_len)?;
10683    if cur.at != bytes.len() {
10684        return Err(invalid("global code page has trailing bytes"));
10685    }
10686    let codes = &bytes[codes_at..codes_at + codes_len];
10687    for &row in positions {
10688        let at = row.checked_mul(4).ok_or_else(|| invalid("dictionary code offset overflow"))?;
10689        let code = u32::from_le_bytes(
10690            codes[at..at + 4].try_into().map_err(|_| invalid("dictionary code is truncated"))?,
10691        );
10692        out.push(valid(row).then_some(code));
10693    }
10694    Ok(true)
10695}
10696
10697/// [`decode`] of only the rows at `positions`, which rise.
10698///
10699/// A compressed text page decompresses only those rows, see [`string::decode_flat_at`], and checks
10700/// only those rows are text. Every other page is decoded whole and gathered, since its values are
10701/// fixed width or its strings are shared through a dictionary, and there picking comes after.
10702fn decode_at(
10703    ty: &LogicalType,
10704    rows: usize,
10705    bytes: &[u8],
10706    global: Option<Arc<Vector>>,
10707    positions: &[u32],
10708) -> Result<Vector> {
10709    if positions.last().is_some_and(|&last| last as usize >= rows) {
10710        return Err(invalid("a position is past the end of the part"));
10711    }
10712    if bytes.first() != Some(&6) {
10713        return decode(ty, rows, bytes, global)?.gather(positions);
10714    }
10715    if ty != &LogicalType::Varchar {
10716        return Err(invalid("compressed text codec belongs to a non-string page"));
10717    }
10718    let mut cur = Cursor::new(bytes);
10719    cur.u8()?;
10720    let validity = match cur.u8()? {
10721        0 => Validity::AllValid,
10722        1 => Validity::AllInvalid,
10723        2 => {
10724            let mask = cur.take(rows.div_ceil(8))?;
10725            Validity::from_iter(positions.len(), |at| {
10726                let row = positions[at] as usize;
10727                mask[row / 8] >> (row % 8) & 1 == 1
10728            })
10729        }
10730        _ => return Err(invalid("page validity tag differs")),
10731    };
10732    let (payload, ends) = string::decode_flat_at(&bytes[cur.at..], positions)?.into_parts();
10733    let mut values = StringColumn::over(Buffer::from_vec(payload).into_page());
10734    let mut start = 0;
10735    for end in ends {
10736        let len = end
10737            .checked_sub(start)
10738            .ok_or_else(|| invalid("compressed text value ends before it starts"))?;
10739        values.push_in_place(start, len)?;
10740        start = end;
10741    }
10742    Ok(Vector::flat(ty.clone(), Data::Varlen(values))?.with_validity(validity))
10743}
10744
10745fn decode(
10746    ty: &LogicalType,
10747    rows: usize,
10748    bytes: &[u8],
10749    global: Option<Arc<Vector>>,
10750) -> Result<Vector> {
10751    let mut cur = Cursor::new(bytes);
10752    let codec = cur.u8()?;
10753    let flag = cur.u8()?;
10754    let validity = match flag {
10755        0 => Validity::AllValid,
10756        1 => Validity::AllInvalid,
10757        2 => {
10758            let mask = cur.take(rows.div_ceil(8))?;
10759            Validity::from_iter(rows, |row| mask[row / 8] >> (row % 8) & 1 == 1)
10760        }
10761        _ => return Err(invalid("page validity tag differs")),
10762    };
10763    if codec == 1 {
10764        if ty != &LogicalType::Varchar {
10765            return Err(invalid("dictionary codec belongs to a non-string page"));
10766        }
10767        let count = cur.u32()? as usize;
10768        let payload_len = cur.u32()? as usize;
10769        let offset_bytes = cur.take(
10770            (count + 1)
10771                .checked_mul(4)
10772                .ok_or_else(|| invalid("dictionary offset count overflow"))?,
10773        )?;
10774        let offsets = offset_bytes
10775            .chunks_exact(4)
10776            .map(|part| u32::from_le_bytes(part.try_into().expect("four bytes")))
10777            .collect::<Vec<_>>();
10778        let payload = cur.take(payload_len)?.to_vec();
10779        if offsets.first() != Some(&0)
10780            || offsets.last().copied().map(|last| last as usize) != Some(payload.len())
10781            || offsets.windows(2).any(|pair| pair[0] > pair[1])
10782        {
10783            return Err(invalid("dictionary offsets do not bound the payload"));
10784        }
10785        // A page, because every chunk cut out of this dictionary points at the same payload and a
10786        // page is what lets a cut be the views and nothing else.
10787        let mut strings = StringColumn::over(Buffer::from_vec(payload).into_page());
10788        for pair in offsets.windows(2) {
10789            strings.push_in_place(pair[0] as usize, (pair[1] - pair[0]) as usize)?;
10790        }
10791        let mut codes = Vec::with_capacity(rows);
10792        for _ in 0..rows {
10793            codes.push(cur.u32()?);
10794        }
10795        if codes.iter().any(|code| *code as usize >= count) {
10796            return Err(invalid("dictionary code is out of range"));
10797        }
10798        if cur.at != bytes.len() {
10799            return Err(invalid("dictionary page has trailing bytes"));
10800        }
10801        let dictionary = Vector::flat(LogicalType::Varchar, Data::Varlen(strings))?;
10802        return Ok(Vector::dictionary(codes, dictionary)?.with_validity(validity));
10803    }
10804    if codec == 3 || codec == 4 {
10805        let dictionary = global.ok_or_else(|| invalid("global code page has no dictionary"))?;
10806        let codes = if codec == 4 {
10807            // The cascade holds the whole tail of the page and says how long it is itself, so the
10808            // check that nothing is left over is the one the decoder already makes.
10809            let wide = integer::decode(&bytes[cur.at..])?;
10810            if wide.len() != rows {
10811                return Err(invalid("encoded code page holds the wrong number of rows"));
10812            }
10813            // Checked once for the page rather than a fallible conversion per code. Every code a
10814            // file holds is inside a `u32` or the file is corrupt, so or the codes together and the
10815            // answer has a bit set above the low thirty two, or the sign bit, exactly when one of
10816            // them did. The or and the narrowing are two passes because each is then a vector
10817            // loop. As one loop with a `push` a code, the length check and the store kept it scalar,
10818            // and it was sixteen instructions a row on the two flag columns of q1.
10819            let seen = wide.iter().fold(0_i64, |seen, &code| seen | code);
10820            if seen < 0 || seen > i64::from(u32::MAX) {
10821                return Err(invalid("code is not a code"));
10822            }
10823            wide.iter().map(|&code| code as u32).collect()
10824        } else {
10825            let mut codes = Vec::with_capacity(rows);
10826            for _ in 0..rows {
10827                codes.push(cur.u32()?);
10828            }
10829            if cur.at != bytes.len() {
10830                return Err(invalid("global code page has trailing bytes"));
10831            }
10832            codes
10833        };
10834        let highest = codes.iter().copied().max();
10835        return Ok(Vector::stable_dictionary_validated(codes, dictionary, highest)?
10836            .with_validity(validity));
10837    }
10838    if codec == 6 {
10839        if ty != &LogicalType::Varchar {
10840            return Err(invalid("compressed text codec belongs to a non-string page"));
10841        }
10842        // As codec 5, the layer holds the whole tail of the page and says how long it is itself.
10843        // It comes back as one buffer with the values laid end to end and where each one ends, which
10844        // is the raw form's layout, so what is left to do here is what codec 0 does.
10845        let (payload, ends) = string::decode_flat(&bytes[cur.at..])?.into_parts();
10846        if ends.len() != rows {
10847            return Err(invalid("compressed text page holds the wrong number of rows"));
10848        }
10849        // A page, because this is read once and handed out a chunk at a time, and a cut of a paged
10850        // payload moves views rather than bytes.
10851        let mut values = StringColumn::over(Buffer::from_vec(payload).into_page());
10852        let mut start = 0;
10853        for end in ends {
10854            let len = end
10855                .checked_sub(start)
10856                .ok_or_else(|| invalid("compressed text value ends before it starts"))?;
10857            values.push_in_place(start, len)?;
10858            start = end;
10859        }
10860        return Ok(Vector::flat(ty.clone(), Data::Varlen(values))?.with_validity(validity));
10861    }
10862    if codec == 5 {
10863        // The cascade holds the whole tail of the page and says how long it is itself.
10864        let values = integer::decode(&bytes[cur.at..])?;
10865        if values.len() != rows {
10866            return Err(invalid("cascade page holds the wrong number of rows"));
10867        }
10868        let data = narrowed(ty, values)?;
10869        return Ok(Vector::flat(ty.clone(), data)?.with_validity(validity));
10870    }
10871    if codec == 2 {
10872        let width = u32::from(cur.u8()?);
10873        let base = i128::from_le_bytes(cur.take(16)?.try_into().expect("sixteen bytes"));
10874        let count = cur.u32()? as usize;
10875        let length = count.checked_mul(8).ok_or_else(|| invalid("packed page is too long"))?;
10876        let words: Vec<u64> = cur
10877            .take(length)?
10878            .chunks_exact(8)
10879            .map(|word| u64::from_le_bytes(word.try_into().expect("eight bytes")))
10880            .collect();
10881        if cur.at != bytes.len() {
10882            return Err(invalid("packed page has trailing bytes"));
10883        }
10884        return Ok(Vector::packed(ty.clone(), words, width, base, rows)?.with_validity(validity));
10885    }
10886    if codec != 0 {
10887        return Err(invalid("page codec is unknown"));
10888    }
10889    let data = match ty {
10890        LogicalType::TinyInt => {
10891            let values = cur.take(rows)?;
10892            Data::Int8(values.iter().map(|item| *item as i8).collect::<Vec<_>>().into())
10893        }
10894        LogicalType::UTinyInt => Data::UInt8(cur.take(rows)?.to_vec().into()),
10895        LogicalType::SmallInt => {
10896            let values =
10897                cur.take(rows.checked_mul(2).ok_or_else(|| invalid("page size overflow"))?)?;
10898            Data::Int16(
10899                values
10900                    .chunks_exact(2)
10901                    .map(|item| i16::from_le_bytes(item.try_into().expect("two bytes")))
10902                    .collect::<Vec<_>>()
10903                    .into(),
10904            )
10905        }
10906        LogicalType::USmallInt => {
10907            let values =
10908                cur.take(rows.checked_mul(2).ok_or_else(|| invalid("page size overflow"))?)?;
10909            Data::UInt16(
10910                values
10911                    .chunks_exact(2)
10912                    .map(|item| u16::from_le_bytes(item.try_into().expect("two bytes")))
10913                    .collect::<Vec<_>>()
10914                    .into(),
10915            )
10916        }
10917        LogicalType::UInteger => {
10918            let values =
10919                cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
10920            Data::UInt32(
10921                values
10922                    .chunks_exact(4)
10923                    .map(|item| u32::from_le_bytes(item.try_into().expect("four bytes")))
10924                    .collect::<Vec<_>>()
10925                    .into(),
10926            )
10927        }
10928        LogicalType::UBigInt => {
10929            let values =
10930                cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
10931            Data::UInt64(
10932                values
10933                    .chunks_exact(8)
10934                    .map(|item| u64::from_le_bytes(item.try_into().expect("eight bytes")))
10935                    .collect::<Vec<_>>()
10936                    .into(),
10937            )
10938        }
10939        LogicalType::Integer | LogicalType::Date => {
10940            let values =
10941                cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
10942            Data::Int32(
10943                values
10944                    .chunks_exact(4)
10945                    .map(|item| i32::from_le_bytes(item.try_into().expect("four bytes")))
10946                    .collect::<Vec<_>>()
10947                    .into(),
10948            )
10949        }
10950        LogicalType::BigInt
10951        | LogicalType::Timestamp
10952        | LogicalType::Time
10953        | LogicalType::TimeTz
10954        | LogicalType::TimestampTz
10955        | LogicalType::TimestampS
10956        | LogicalType::TimestampMs
10957        | LogicalType::TimestampNs => {
10958            let values =
10959                cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
10960            Data::Int64(
10961                values
10962                    .chunks_exact(8)
10963                    .map(|item| i64::from_le_bytes(item.try_into().expect("eight bytes")))
10964                    .collect::<Vec<_>>()
10965                    .into(),
10966            )
10967        }
10968        LogicalType::HugeInt | LogicalType::Uuid => {
10969            let values =
10970                cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
10971            Data::Int128(
10972                values
10973                    .chunks_exact(16)
10974                    .map(|item| i128::from_le_bytes(item.try_into().expect("sixteen bytes")))
10975                    .collect::<Vec<_>>()
10976                    .into(),
10977            )
10978        }
10979        LogicalType::UHugeInt => {
10980            let values =
10981                cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
10982            Data::UInt128(
10983                values
10984                    .chunks_exact(16)
10985                    .map(|item| u128::from_le_bytes(item.try_into().expect("sixteen bytes")))
10986                    .collect::<Vec<_>>()
10987                    .into(),
10988            )
10989        }
10990        LogicalType::Float => {
10991            let values =
10992                cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
10993            Data::Float32(
10994                values
10995                    .chunks_exact(4)
10996                    .map(|item| f32::from_le_bytes(item.try_into().expect("four bytes")))
10997                    .collect::<Vec<_>>()
10998                    .into(),
10999            )
11000        }
11001        LogicalType::Double => {
11002            let values =
11003                cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
11004            Data::Float64(
11005                values
11006                    .chunks_exact(8)
11007                    .map(|item| f64::from_le_bytes(item.try_into().expect("eight bytes")))
11008                    .collect::<Vec<_>>()
11009                    .into(),
11010            )
11011        }
11012        LogicalType::Interval => {
11013            let values =
11014                cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
11015            Data::Interval(
11016                values
11017                    .chunks_exact(16)
11018                    .map(|item| {
11019                        (
11020                            i32::from_le_bytes(item[..4].try_into().expect("four bytes")),
11021                            i32::from_le_bytes(item[4..8].try_into().expect("four bytes")),
11022                            i64::from_le_bytes(item[8..].try_into().expect("eight bytes")),
11023                        )
11024                    })
11025                    .collect::<Vec<_>>()
11026                    .into(),
11027            )
11028        }
11029        LogicalType::Boolean => {
11030            let values = cur.take(rows)?;
11031            if values.iter().any(|value| *value > 1) {
11032                return Err(invalid("boolean page has another value"));
11033            }
11034            Data::Bool(values.iter().map(|value| *value == 1).collect::<Vec<_>>().into())
11035        }
11036        // Whichever integer the declared width says, which is the mapping the rest of the engine
11037        // already uses for a decimal in memory.
11038        LogicalType::Decimal { .. } => match ty.physical() {
11039            PhysicalType::Int16 => {
11040                let values =
11041                    cur.take(rows.checked_mul(2).ok_or_else(|| invalid("page size overflow"))?)?;
11042                Data::Int16(
11043                    values
11044                        .chunks_exact(2)
11045                        .map(|item| i16::from_le_bytes(item.try_into().expect("two bytes")))
11046                        .collect::<Vec<_>>()
11047                        .into(),
11048                )
11049            }
11050            PhysicalType::Int32 => {
11051                let values =
11052                    cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
11053                Data::Int32(
11054                    values
11055                        .chunks_exact(4)
11056                        .map(|item| i32::from_le_bytes(item.try_into().expect("four bytes")))
11057                        .collect::<Vec<_>>()
11058                        .into(),
11059                )
11060            }
11061            PhysicalType::Int64 => {
11062                let values =
11063                    cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
11064                Data::Int64(
11065                    values
11066                        .chunks_exact(8)
11067                        .map(|item| i64::from_le_bytes(item.try_into().expect("eight bytes")))
11068                        .collect::<Vec<_>>()
11069                        .into(),
11070                )
11071            }
11072            _ => {
11073                let values =
11074                    cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
11075                Data::Int128(
11076                    values
11077                        .chunks_exact(16)
11078                        .map(|item| i128::from_le_bytes(item.try_into().expect("sixteen bytes")))
11079                        .collect::<Vec<_>>()
11080                        .into(),
11081                )
11082            }
11083        },
11084        LogicalType::Varchar | LogicalType::Blob | LogicalType::Bit => {
11085            let offset_bytes = cur
11086                .take((rows + 1).checked_mul(4).ok_or_else(|| invalid("offset count overflow"))?)?;
11087            let offsets = offset_bytes
11088                .chunks_exact(4)
11089                .map(|part| u32::from_le_bytes(part.try_into().expect("four bytes")))
11090                .collect::<Vec<_>>();
11091            let payload = cur.take(bytes.len() - cur.at)?.to_vec();
11092            if offsets.first() != Some(&0)
11093                || offsets.last().copied().map(|last| last as usize) != Some(payload.len())
11094                || offsets.windows(2).any(|pair| pair[0] > pair[1])
11095            {
11096                return Err(invalid("string offsets do not bound the payload"));
11097            }
11098            // A page for the reason the dictionary payload above is one: the page is read once and
11099            // handed out a chunk at a time, and a cut of a paged payload moves views rather than
11100            // bytes.
11101            //
11102            // A varchar is checked for text on the way in and a blob and a bit string are not,
11103            // because the second pair never claimed to hold any. Reading them through the checking
11104            // seam would refuse a column for holding exactly what it was told to hold.
11105            let mut values = StringColumn::over(Buffer::from_vec(payload).into_page());
11106            let text = ty == &LogicalType::Varchar;
11107            for pair in offsets.windows(2) {
11108                let (at, len) = (pair[0] as usize, (pair[1] - pair[0]) as usize);
11109                if text {
11110                    values.push_in_place(at, len)?;
11111                } else {
11112                    values.push_bytes_in_place(at, len)?;
11113                }
11114            }
11115            Data::Varlen(values)
11116        }
11117        _ => return Err(Error::not_implemented(format!("native page for {ty}"))),
11118    };
11119    if cur.at != bytes.len() {
11120        return Err(invalid("page has trailing bytes"));
11121    }
11122    Ok(Vector::flat(ty.clone(), data)?.with_validity(validity))
11123}
11124
11125#[cfg(test)]
11126mod tests {
11127    use std::fs;
11128    use std::io::{Seek, SeekFrom, Write};
11129    use std::path::PathBuf;
11130    use std::time::{SystemTime, UNIX_EPOCH};
11131
11132    use rudb_common::Stat;
11133    use rudb_common::Value;
11134    use rudb_common::bounds::{Frequencies, Op, Remainder, Zones};
11135    use rudb_common::stat::Provenance;
11136
11137    use super::*;
11138
11139    #[test]
11140    fn a_name_taken_in_pieces_is_the_name_of_the_pieces_joined() {
11141        let bytes: Vec<u8> =
11142            (0..300_u32).map(|at| (at.wrapping_mul(2_654_435_761) >> 13) as u8).collect();
11143        for length in [0, 1, 7, 31, 32, 33, 63, 64, 65, 100, 300] {
11144            let whole = content_name(&bytes[..length]);
11145            for step in [1, 3, 8, 31, 32, 33, 64, 301] {
11146                let mut namer = ContentNamer::default();
11147                bytes[..length].chunks(step).for_each(|piece| namer.update(piece));
11148                assert_eq!(namer.finish(), whole, "{length} bytes in pieces of {step}");
11149            }
11150        }
11151    }
11152
11153    /// The chooser as it was before it could rule kinds out up front: the same narrowing, with every
11154    /// kind tested for. What it writes is what the file used to hold.
11155    #[derive(Debug)]
11156    struct TestsEverything<'a>(&'a dyn chooser::Chooser);
11157
11158    impl chooser::Chooser for TestsEverything<'_> {
11159        fn name(&self) -> &'static str {
11160            "tests everything"
11161        }
11162
11163        fn narrow_strings(
11164            &self,
11165            values: &[&[u8]],
11166            offered: &[string::Kind],
11167            depth: u8,
11168        ) -> Vec<string::Kind> {
11169            self.0.narrow_strings(values, offered, depth)
11170        }
11171
11172        fn narrow_integers(
11173            &self,
11174            values: &[i64],
11175            offered: &[integer::Kind],
11176            depth: u8,
11177        ) -> Vec<integer::Kind> {
11178            self.0.narrow_integers(values, offered, depth)
11179        }
11180    }
11181
11182    #[test]
11183    fn ruling_kinds_out_before_testing_for_them_writes_the_same_bytes() {
11184        let columns: Vec<Vec<i64>> = vec![
11185            vec![],
11186            vec![5; 1000],
11187            (0..1000).collect(),
11188            (0..1000).map(|row| 1_600_000_000_000_000 + row * 1_000_000).collect(),
11189            (0..1000).map(|row| row / 50).collect(),
11190            (0..1000).map(|row| if row % 97 == 0 { row } else { 0 }).collect(),
11191            (0..1000).map(|row| (row * 7919) % 13).collect(),
11192            (0..1000).map(|row| (row * 2_654_435_761) % 1_000_003).collect(),
11193            (0..1000).map(|row| [3, 3, 3, 9, 9, 1][row as usize % 6]).collect(),
11194            (0..1000).map(|row| i64::MIN + row % 3).collect(),
11195        ];
11196        let choosers: [&dyn chooser::Chooser; 2] = [&Fixed, &Codes];
11197        for column in &columns {
11198            for chooser in choosers {
11199                let quick = integer::encode_with(column, chooser).unwrap();
11200                let full = integer::encode_with(column, &TestsEverything(chooser)).unwrap();
11201                assert_eq!(
11202                    quick,
11203                    full,
11204                    "{} on {:?}",
11205                    chooser.name(),
11206                    &column[..column.len().min(8)]
11207                );
11208            }
11209        }
11210    }
11211
11212    /// Parts of a column that all look alike come out of a settled shape byte for byte as they
11213    /// come out of a search, because the search would have kept the same tree on every one.
11214    #[test]
11215    fn parts_that_look_alike_replay_to_the_bytes_a_search_writes() {
11216        let mut settling = Settling::default();
11217        for part in 0..STRIPE_PARTS as i64 {
11218            let values: Vec<i64> = (0..2048)
11219                .map(|row| 1_600_000_000_000_000 + (part * 2048 + row) * 1_000_000 + row % 7)
11220                .collect();
11221            let searched = integer::encode_with(&values, &Fixed).unwrap();
11222            assert_eq!(settling.encode(&values).unwrap(), searched, "part {part}");
11223        }
11224    }
11225
11226    /// A column that changes shape partway through a stripe still reads back, and no part comes
11227    /// out much bigger than a search would have made it, because a replay that stops fitting or
11228    /// grows past a quarter a row is searched.
11229    #[test]
11230    fn a_column_that_changes_under_the_shape_is_searched_again() {
11231        let mut state = 0x9e37_79b9_7f4a_7c15_u64;
11232        let mut noise = move || {
11233            state ^= state << 13;
11234            state ^= state >> 7;
11235            state ^= state << 17;
11236            (state % 1_000_000) as i64
11237        };
11238        let mut settling = Settling::default();
11239        for part in 0..STRIPE_PARTS as i64 {
11240            let values: Vec<i64> = match part / 16 {
11241                0 => (0..2048).map(|row| (part * 2048 + row) / 300).collect(),
11242                1 => (0..2048).map(|_| noise()).collect(),
11243                2 => (0..2048).map(|row| if row % 97 == 0 { row } else { 42 }).collect(),
11244                _ => (0..2048).map(|row| 5 + (part * 2048 + row) * 1_000_000).collect(),
11245            };
11246            let settled = settling.encode(&values).unwrap();
11247            assert_eq!(integer::decode(&settled).unwrap(), values, "part {part}");
11248            let searched = integer::encode_with(&values, &Fixed).unwrap();
11249            assert!(
11250                settled.len() * 4 <= searched.len() * 5,
11251                "part {part}: {} settled against {} searched, {} against {}",
11252                settled.len(),
11253                searched.len(),
11254                integer::describe(&settled).unwrap(),
11255                integer::describe(&searched).unwrap(),
11256            );
11257        }
11258    }
11259
11260    #[test]
11261    fn checksum_matches_fixed_vectors() {
11262        assert_eq!(checksum(b""), 0xef46_db37_51d8_e999);
11263        assert_eq!(checksum(b"a"), 0xd24e_c4f1_a98c_6e5b);
11264        assert_eq!(checksum(b"abc"), 0x44bc_2cf5_ad77_0999);
11265    }
11266
11267    #[test]
11268    fn sorting_across_threads_matches_sorting_on_one() {
11269        let mut state = 0x9e37_79b9_7f4a_7c15_u64;
11270        let mut next = move || {
11271            state ^= state << 13;
11272            state ^= state >> 7;
11273            state ^= state << 17;
11274            state
11275        };
11276        let mut values = Vec::new();
11277        for at in 0..150_000_u64 {
11278            let value = match next() % 6 {
11279                0 => Vec::new(),
11280                1 => format!("https://example.com/{}", next() % 5_000).into_bytes(),
11281                2 => format!("https://example.com/path/{at}").into_bytes(),
11282                3 => b"same".to_vec(),
11283                4 => vec![0xff; (next() % 12) as usize],
11284                _ => (0..next() % 20).map(|_| (next() % 3) as u8).collect(),
11285            };
11286            values.push(value);
11287        }
11288        let value = |code: u32| values[code as usize].as_slice();
11289        for workers in [1, 2, 3, 8, 32] {
11290            let mut one = (0..values.len() as u32).rev().collect::<Vec<_>>();
11291            let mut across = one.clone();
11292            sort_by_value(&mut one, value);
11293            sort_by_value_across(&mut across, value, workers);
11294            assert_eq!(one, across, "{workers} workers");
11295        }
11296        let mut sorted = (0..values.len() as u32).collect::<Vec<_>>();
11297        sort_by_value_across(&mut sorted, value, 8);
11298        assert!(sorted.windows(2).all(|pair| value(pair[0]) <= value(pair[1])));
11299    }
11300
11301    fn path(label: &str) -> PathBuf {
11302        let stamp = SystemTime::now().duration_since(UNIX_EPOCH).expect("time advances").as_nanos();
11303        std::env::temp_dir().join(format!("rudb-native-{label}-{}-{stamp}.rdb", std::process::id()))
11304    }
11305
11306    /// Every value of a dictionary in code order, which the tests have no other way to ask for now
11307    /// that a dictionary does not keep the bytes of the values it has seen.
11308    ///
11309    /// Only valid once `finish_blocks` has run, because until then the last part block is still raw.
11310    fn dictionary_values(dictionary: &GlobalDictionary) -> Vec<Vec<u8>> {
11311        let (flat, bases) = dictionary.decoded(None).expect("the blocks decode");
11312        (0..dictionary.values())
11313            .map(|code| {
11314                let (from, to) = GlobalDictionary::value_span(&dictionary.ends, &bases, code);
11315                flat[from..to].to_vec()
11316            })
11317            .collect()
11318    }
11319
11320    /// The sections a test put in the table, which is every one the writer did not.
11321    ///
11322    /// A table now carries a summary and a sketch per column out of the write itself, and a test
11323    /// about the section table is not about those. Filtering by kind rather than by count, so a
11324    /// table that turns out to have no room for its summaries does not quietly change what these
11325    /// tests are asserting over.
11326    fn attached(table: &Table) -> Vec<&Section> {
11327        table.sections().iter().filter(|held| !held.among(section::STATISTICS_KINDS)).collect()
11328    }
11329
11330    /// A read names the offset it wants, so a cursor somebody else moved cannot reach it.
11331    #[test]
11332    fn a_read_at_an_offset_ignores_where_another_thread_left_the_cursor() {
11333        const SPANS: usize = 64;
11334        const SPAN: usize = 512;
11335        let path = path("positional");
11336        let content: Vec<u8> =
11337            (0..SPANS).flat_map(|span| std::iter::repeat_n(span as u8, SPAN)).collect();
11338        fs::write(&path, &content).expect("the file is written");
11339        let file = Arc::new(File::open(&path).expect("the file opens"));
11340        std::thread::scope(|scope| {
11341            for _ in 0..8 {
11342                let file = Arc::clone(&file);
11343                scope.spawn(move || {
11344                    for _ in 0..64 {
11345                        for span in 0..SPANS {
11346                            let mut bytes = [0_u8; SPAN];
11347                            read_at(&file, (span * SPAN) as u64, &mut bytes)
11348                                .expect("the span reads");
11349                            assert!(
11350                                bytes.iter().all(|byte| *byte == span as u8),
11351                                "span {span} came back as {}",
11352                                bytes[0],
11353                            );
11354                        }
11355                    }
11356                });
11357            }
11358        });
11359        let mut past = [0_u8; SPAN];
11360        let end = (SPANS * SPAN) as u64;
11361        let error = read_at(&file, end, &mut past).expect_err("a read past the end is refused");
11362        assert!(error.message().contains("ends before its declared length"), "{error}");
11363        drop(file);
11364        let _ = fs::remove_file(&path);
11365    }
11366
11367    /// The writer records where it put a page and puts it there, whatever the cursor is doing.
11368    ///
11369    /// The cursor is moved between the steps that record an offset, which is what reading the pages
11370    /// back to build the frequencies does on a platform with no `pread`. Without the fix the
11371    /// directory lands on top of a page and the file fails to reopen.
11372    #[test]
11373    fn a_writer_puts_a_page_where_it_said_it_did_wherever_the_cursor_has_got_to() {
11374        let path = path("cursor");
11375        let mut writer = Writer::create(
11376            &path,
11377            "items",
11378            vec![
11379                Field::required("id", LogicalType::Integer),
11380                Field::new("text", LogicalType::Varchar),
11381            ],
11382        )
11383        .expect("new file");
11384        writer.append(&sample()).expect("first part");
11385        writer.file.seek(SeekFrom::Start(0)).expect("the cursor goes back to the header");
11386        writer.append(&sample()).expect("second part");
11387        writer.file.seek(SeekFrom::Start(1)).expect("and somewhere useless again");
11388        writer.finish().expect("commit");
11389        let reader = Reader::open(&path).expect("reopen from disk");
11390        assert_eq!(reader.table().rows(), 6);
11391        let ids = reader.read(0, &[0]).expect("the integer page reads back");
11392        assert_eq!(ids.value_at(0, 0), Value::Integer(4));
11393        assert_eq!(ids.value_at(2, 0), Value::Integer(-2));
11394        let text = reader.read(1, &[1]).expect("the text page reads back");
11395        assert_eq!(text.value_at(1, 0), Value::Null);
11396        assert_eq!(text.value_at(2, 0), Value::Varchar("long text after a slash".into()));
11397        // Nothing the directory points at may run past the end of the file, which is the shape the
11398        // failure took: a page recorded at an offset the directory had already been written over.
11399        let end = reader.table().stripes().iter().flat_map(|stripe| {
11400            stripe
11401                .pages
11402                .iter()
11403                .map(|page| page.offset + u64::from(page.length))
11404                .chain(std::iter::once(stripe.index.offset + u64::from(stripe.index.length)))
11405        });
11406        let last = end.fold(HEADER, u64::max);
11407        let directory = fs::metadata(&path).expect("the file is there").len();
11408        assert!(last <= directory, "a page runs to {last} in a file of {directory} bytes");
11409        fs::remove_file(path).expect("remove scratch file");
11410    }
11411
11412    /// How long a global dictionary index is, read out of the page's own header.
11413    ///
11414    /// The tests below damage a byte of the order or of the payload, so they need to know where each
11415    /// one starts, and working it out here rather than writing a number down means adding something
11416    /// to the index does not quietly turn one of them into a test that damages the index instead.
11417    fn dictionary_index_len(header: &[u8; DICTIONARY_HEADER]) -> u64 {
11418        let count = u64::from(u32::from_le_bytes(header[0..4].try_into().expect("four bytes")));
11419        let blocks = u64::from(u32::from_le_bytes(header[8..12].try_into().expect("four bytes")));
11420        let width = u32::from_le_bytes(header[12..16].try_into().expect("four bytes"));
11421        let bits = (width & !DICTIONARY_FLAGS) as usize;
11422        let payload_words = if width & DICTIONARY_SCATTERED == 0 { 2 } else { 3 };
11423        let rank_blocks = count.div_ceil(TEXT_RANK_BLOCK as u64);
11424        DICTIONARY_HEADER as u64
11425            + offset_bytes(count as usize, bits) as u64
11426            + blocks * payload_words * 8
11427            + rank_blocks * 16
11428            + if width & DICTIONARY_GRAMS == 0 { 0 } else { 8 }
11429    }
11430
11431    fn sample() -> Chunk {
11432        Chunk::new(vec![
11433            Vector::from_values(
11434                LogicalType::Integer,
11435                &[Value::Integer(4), Value::Integer(9), Value::Integer(-2)],
11436            )
11437            .expect("integers"),
11438            Vector::from_values(
11439                LogicalType::Varchar,
11440                &[
11441                    Value::Varchar("alpha".into()),
11442                    Value::Null,
11443                    Value::Varchar("long text after a slash".into()),
11444                ],
11445            )
11446            .expect("strings"),
11447        ])
11448        .expect("matching rows")
11449    }
11450
11451    fn sample_ids() -> Chunk {
11452        Chunk::new(vec![
11453            Vector::flat(LogicalType::Integer, Data::Int32(vec![7, 8, 9].into()))
11454                .expect("integers"),
11455        ])
11456        .expect("one column")
11457    }
11458
11459    #[test]
11460    fn the_planner_gets_the_null_count_off_the_same_directory_the_bounds_are_in() {
11461        // Six rows, two of them null. `IS NULL` used to get the same fifth any unreadable
11462        // condition gets, and the number was in the stripe entry next to the bounds all along.
11463        let path = path("nulls_for_the_planner");
11464        let mut writer =
11465            Writer::create(&path, "items", vec![Field::new("a", LogicalType::Integer)])
11466                .expect("new file");
11467        let rows = Chunk::new(vec![
11468            Vector::from_values(
11469                LogicalType::Integer,
11470                &[
11471                    Value::Integer(4),
11472                    Value::Null,
11473                    Value::Integer(9),
11474                    Value::Null,
11475                    Value::Integer(1),
11476                    Value::Integer(2),
11477                ],
11478            )
11479            .expect("integers"),
11480        ])
11481        .expect("one column");
11482        writer.append(&rows).expect("the only part");
11483        writer.finish().expect("commit");
11484        let reader = Reader::open(&path).expect("reopen from disk");
11485        let stripes = Stripes::new(reader);
11486        let column = stripes.column("a").expect("the file has that column");
11487        assert_eq!(stripes.nulls(column), Stat::exact(2, Provenance::NullCount));
11488        // A column the file does not have. Zero here would be a fact about a column that is not
11489        // there, which the planner would then divide by.
11490        assert_eq!(stripes.nulls(column + 1), Stat::Unknown);
11491        fs::remove_file(&path).expect("clean up");
11492    }
11493
11494    #[test]
11495    fn the_planner_gets_a_row_count_per_value_off_a_complete_synopsis() {
11496        // The whole of the frequency half of #1106, end to end over a real file. Six rows, three
11497        // of one value and two of another, and a complete synopsis because six rows is well inside
11498        // what the writer can account for. The estimate for `id = 4` is three rows rather than a
11499        // sixth of the table, and for a value the file does not hold it is none.
11500        let path = path("frequencies_for_the_planner");
11501        let mut writer =
11502            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
11503                .expect("new file");
11504        let rows = Chunk::new(vec![
11505            Vector::from_values(
11506                LogicalType::Integer,
11507                &[
11508                    Value::Integer(4),
11509                    Value::Integer(4),
11510                    Value::Integer(4),
11511                    Value::Integer(9),
11512                    Value::Integer(9),
11513                    Value::Integer(1),
11514                ],
11515            )
11516            .expect("integers"),
11517        ])
11518        .expect("one column");
11519        writer.append(&rows).expect("the only part");
11520        writer.finish().expect("commit");
11521        let reader = Reader::open(&path).expect("reopen from disk");
11522        let common = Common::new(reader);
11523        assert_eq!(common.rows(), 6);
11524        let column = common.column("id").expect("the file has that column");
11525        assert_eq!(common.column("nothing"), None);
11526        assert_eq!(
11527            common.rows_with(column, &Bound::Int(4)),
11528            Stat::exact(3, Provenance::FrequencySynopsis)
11529        );
11530        // Not in the file, and a synopsis that accounts for all six rows proves it.
11531        assert_eq!(
11532            common.rows_with(column, &Bound::Int(7)),
11533            Stat::exact(0, Provenance::FrequencySynopsis)
11534        );
11535        // A constant of another domain against an integer column. Nothing in the list compares
11536        // with it, so the zero above would be an artefact of the mismatch rather than a fact.
11537        assert_eq!(common.rows_with(column, &Bound::Bytes(b"four".to_vec())), Stat::Unknown);
11538        // A complete list has no remainder. Answering one of no rows over no values would hand the
11539        // caller a division to special case, and the counts above already answer this column.
11540        assert_eq!(common.remainder(column), None);
11541        fs::remove_file(&path).expect("clean up");
11542    }
11543
11544    #[test]
11545    fn string_frequency_estimates_do_not_open_the_global_dictionary() {
11546        let path = path("string_frequencies_for_the_planner");
11547        let mut writer =
11548            Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
11549                .expect("new file");
11550        let rows = Chunk::new(vec![
11551            Vector::from_values(
11552                LogicalType::Varchar,
11553                &[
11554                    Value::Varchar(String::new()),
11555                    Value::Varchar("alpha".into()),
11556                    Value::Varchar(String::new()),
11557                    Value::Varchar("beta".into()),
11558                    Value::Varchar(String::new()),
11559                ],
11560            )
11561            .expect("strings"),
11562        ])
11563        .expect("one column");
11564        writer.append(&rows).expect("the only part");
11565        writer.finish().expect("commit");
11566
11567        let reader = Reader::open(&path).expect("reopen from disk");
11568        assert_eq!(reader.reads().dictionaries, 0, "open reads only the directory");
11569        let common = Common::new(reader.clone());
11570        let column = common.column("text").expect("the file has that column");
11571        assert_eq!(
11572            common.rows_with(column, &Bound::Bytes(Vec::new())),
11573            Stat::exact(3, Provenance::FrequencySynopsis)
11574        );
11575        assert_eq!(
11576            common.rows_with(column, &Bound::Bytes(b"missing".to_vec())),
11577            Stat::exact(0, Provenance::FrequencySynopsis)
11578        );
11579        assert_eq!(
11580            reader.reads().dictionaries,
11581            0,
11582            "the bounded spellings answer without opening the dictionary index"
11583        );
11584        fs::remove_file(&path).expect("clean up");
11585    }
11586
11587    #[test]
11588    fn host_groups_certify_omitted_hosts_and_keep_exact_aggregates() {
11589        let path = path("certified_host_groups");
11590        let mut writer =
11591            Writer::create(&path, "hits", vec![Field::required("Referer", LogicalType::Varchar)])
11592                .expect("new file");
11593        let mut values = vec![Value::Varchar("http://www.example.com/a".into()); 150];
11594        values.extend(vec![Value::Varchar("https://example.com/b".into()); 70]);
11595        values.extend((0..550).map(|at| Value::Varchar(format!("https://site{at}.test/x"))));
11596        values.push(Value::Varchar(String::new()));
11597        for part in values.chunks(512) {
11598            writer
11599                .append(
11600                    &Chunk::new(vec![
11601                        Vector::from_values(LogicalType::Varchar, part).expect("strings"),
11602                    ])
11603                    .expect("one column"),
11604                )
11605                .expect("part written");
11606        }
11607        writer.finish().expect("commit");
11608        let reader = Reader::open(&path).expect("reopen");
11609        assert!(reader.table.host_groups.is_none(), "no query-specific host result is stored");
11610        fs::remove_file(&path).expect("clean up");
11611    }
11612
11613    /// A table directory with nothing in it but a name and one column, for the section tests.
11614    ///
11615    /// The section table is orthogonal to everything else in a directory, so the tests that pin it
11616    /// say so by starting from the emptiest table that encodes.
11617    fn bare_table(sections: Vec<Section>) -> Table {
11618        Table {
11619            name: "linked".to_owned(),
11620            fields: vec![Field::required("id", LogicalType::Integer)],
11621            stripes: Vec::new(),
11622            rows: 0,
11623            dictionaries: vec![None],
11624            dictionary_payloads: Vec::new(),
11625            distincts: vec![None],
11626            frequencies: vec![None],
11627            pair_frequencies: Vec::new(),
11628            frequency_texts: Vec::new(),
11629            host_groups: None,
11630            clustering: None,
11631            generation: 1,
11632            sections,
11633        }
11634    }
11635
11636    fn a_key_map_section() -> Section {
11637        Section {
11638            kind: *section::KEY_MAP,
11639            id: 1,
11640            generation: 3,
11641            extents: 1,
11642            extent_page: HEADER,
11643            extent_bytes: section::EXTENT_BYTES as u32,
11644            hash: 0x1234_5678_9abc_def0,
11645            flags: 0,
11646            header_bytes: 24,
11647        }
11648    }
11649
11650    #[test]
11651    fn a_section_table_round_trips_through_a_directory() {
11652        let mut later = a_key_map_section();
11653        later.kind = *b"RUDBZZ9\0";
11654        later.id = 2;
11655        let table = bare_table(vec![a_key_map_section(), later]);
11656        let directory = encode_directory(&table).expect("directory");
11657        let decoded = decode_directory(&directory, 1 << 20).expect("reopen");
11658        assert_eq!(decoded.sections(), &[a_key_map_section(), later]);
11659        // The second is a kind this build has no name for, and it survived the round trip anyway.
11660        // That is what keeps an old build from silently discarding a newer build's work when it
11661        // rewrites a directory.
11662        assert!(decoded.sections()[0].known());
11663        assert!(!decoded.sections()[1].known());
11664    }
11665
11666    #[test]
11667    fn a_directory_written_before_the_section_table_reads_as_a_table_with_none() {
11668        // The G1 exit criterion, at the directory level. A format 22 directory is exactly this
11669        // build's directory with the trailing section block cut off, so cutting it off is the
11670        // honest way to make one: no fixture to go stale, and no separate encoder to drift.
11671        let directory = encode_directory(&bare_table(Vec::new())).expect("directory");
11672        let block = SECTIONS.len() + size_of::<u64>() + size_of::<u16>();
11673        let older = &directory[..directory.len() - block];
11674        let decoded = decode_directory(older, 1 << 20).expect("a directory from before sections");
11675        assert!(decoded.sections().is_empty());
11676        assert_eq!(decoded.generation(), 0, "a format 22 table recorded no generation");
11677        assert_eq!(decoded.name(), "linked");
11678        assert_eq!(decoded.fields().len(), 1, "everything before the block still decodes");
11679    }
11680
11681    #[test]
11682    fn a_file_stamped_with_the_previous_format_still_opens_and_reads() {
11683        // The same criterion end to end, which is the one the milestone actually asks for: a build
11684        // that knows about sections opens a file written by a build that did not, with no rewrite
11685        // and no repair, and answers from it. The version field is patched rather than a file
11686        // committed by an old binary because the bytes either side of it are identical: format 22
11687        // and format 23 differ only in a trailing directory block, and a reader that stops before
11688        // that block gets a table with no sections.
11689        let path = path("format_twenty_two");
11690        let mut writer =
11691            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
11692                .expect("new file");
11693        let rows = Chunk::new(vec![
11694            Vector::from_values(
11695                LogicalType::Integer,
11696                &[Value::Integer(1), Value::Integer(2), Value::Integer(3)],
11697            )
11698            .expect("integers"),
11699        ])
11700        .expect("one column");
11701        writer.append(&rows).expect("the only part");
11702        writer.finish().expect("commit");
11703
11704        let file = OpenOptions::new().write(true).open(&path).expect("reopen to patch");
11705        write_at(&file, 8, &22_u32.to_le_bytes()).expect("stamp the older format");
11706        drop(file);
11707
11708        let reader = Reader::open(&path).expect("a format 22 file opens unchanged");
11709        assert_eq!(reader.table().rows(), 3);
11710        // The rows and not the section table, because the section block is found by the magic at
11711        // the end of the directory rather than by the number in the header, so stamping the header
11712        // back does not take away the summaries this writer put there. What the test is about is
11713        // that the version check accepts 22, and the rows coming back is what says it did.
11714        assert_eq!(reader.read(0, &[0]).expect("the part still reads").len(), 3);
11715
11716        // And a format this build has never written is still refused, so the accept set is a list
11717        // and not an absence of a check.
11718        let file = OpenOptions::new().write(true).open(&path).expect("reopen to patch");
11719        write_at(&file, 8, &21_u32.to_le_bytes()).expect("stamp an unreadable format");
11720        drop(file);
11721        let error = Reader::open(&path).expect_err("format 21 is not readable");
11722        assert!(error.to_string().contains("format 21"), "{error}");
11723
11724        fs::remove_file(&path).expect("clean up");
11725    }
11726
11727    #[test]
11728    fn a_section_whose_extent_table_is_outside_the_file_is_refused() {
11729        // The bound the format has to check and `section` cannot, because only the reader knows how
11730        // big the file is. Reading the payload a section like this names would be reading whatever
11731        // else happens to be at that offset, which is the one way a graph section could turn into a
11732        // wrong answer rather than a slow one.
11733        let mut past = a_key_map_section();
11734        past.extent_page = 1 << 30;
11735        let directory = encode_directory(&bare_table(vec![past])).expect("directory");
11736        let error = decode_directory(&directory, 1 << 20).expect_err("refused");
11737        assert!(error.to_string().contains("outside the file"), "{error}");
11738
11739        let mut inside_the_header = a_key_map_section();
11740        inside_the_header.extent_page = 8;
11741        let directory = encode_directory(&bare_table(vec![inside_the_header])).expect("directory");
11742        assert!(
11743            decode_directory(&directory, 1 << 20).is_err(),
11744            "a section may not overlap a header"
11745        );
11746    }
11747
11748    #[test]
11749    fn a_section_recorded_as_not_built_is_legal_and_names_no_bytes() {
11750        // Section 3.7: a relationship that does not fit the budget is recorded with its size so
11751        // that `rudb_links()` can report what a larger budget would buy. That record is a section
11752        // entry with no extents, so it has to survive a round trip while naming nothing.
11753        let not_built = Section {
11754            kind: *section::FORWARD_LINK,
11755            id: 9,
11756            generation: 3,
11757            extents: 0,
11758            extent_page: 0,
11759            extent_bytes: 0,
11760            hash: 0,
11761            flags: 0,
11762            header_bytes: 0,
11763        };
11764        let directory = encode_directory(&bare_table(vec![not_built])).expect("directory");
11765        let decoded = decode_directory(&directory, 1 << 20).expect("reopen");
11766        assert_eq!(decoded.sections(), &[not_built]);
11767
11768        // But a section with no extents that still names an extent table is incoherent, and an
11769        // incoherent entry is a torn directory rather than a relationship that was skipped.
11770        let mut incoherent = not_built;
11771        incoherent.extent_bytes = 28;
11772        incoherent.extent_page = HEADER;
11773        let directory = encode_directory(&bare_table(vec![incoherent])).expect("directory");
11774        assert!(decode_directory(&directory, 1 << 20).is_err());
11775    }
11776
11777    #[test]
11778    fn a_directory_naming_more_sections_than_the_bound_is_refused() {
11779        let directory = encode_directory(&bare_table(Vec::new())).expect("directory");
11780        let mut torn = directory.clone();
11781        let count_at = torn.len() - size_of::<u16>();
11782        torn[count_at..].copy_from_slice(&u16::MAX.to_le_bytes());
11783        // Not an allocation of sixty five thousand entries off a torn count: either the bound
11784        // refuses it or the bytes run out, and both are errors rather than a read past the end.
11785        assert!(decode_directory(&torn, 1 << 20).is_err());
11786    }
11787
11788    /// A committed one column file of `rows` integers, for the attach tests.
11789    fn linked_file(label: &str, rows: i32) -> PathBuf {
11790        let path = path(label);
11791        let mut writer =
11792            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
11793                .expect("new file");
11794        let values = (0..rows).map(Value::Integer).collect::<Vec<_>>();
11795        let chunk =
11796            Chunk::new(vec![Vector::from_values(LogicalType::Integer, &values).expect("integers")])
11797                .expect("one column");
11798        writer.append(&chunk).expect("the only part");
11799        writer.finish().expect("commit");
11800        path
11801    }
11802
11803    fn a_key_map_payload() -> Vec<u8> {
11804        // Shaped like one without being one: this crate never reads a payload, so what matters here
11805        // is that every byte comes back and that the header the entry measures is at the front.
11806        (0..512_u32).flat_map(u32::to_le_bytes).collect()
11807    }
11808
11809    #[test]
11810    fn a_section_attached_to_a_committed_file_reads_back_byte_for_byte() {
11811        let path = linked_file("attach", 64);
11812        let payload = a_key_map_payload();
11813        let table = attach(
11814            &path,
11815            "items",
11816            &[section::Attachment {
11817                kind: *section::KEY_MAP,
11818                id: 0,
11819                flags: 2,
11820                header_bytes: 40,
11821                bytes: &payload,
11822            }],
11823        )
11824        .expect("attach a key map");
11825        assert_eq!(attached(&table).len(), 1);
11826
11827        let reader = Reader::open(&path).expect("reopen after the attach");
11828        let held = attached(reader.table());
11829        assert_eq!(held.len(), 1);
11830        assert_eq!(held[0].kind, *section::KEY_MAP);
11831        assert_eq!(held[0].flags, 2, "the form a reader must not have to guess");
11832        assert_eq!(held[0].header_bytes, 40);
11833        // The generation is the one the pages were written at, not the one the attach committed at.
11834        // Attaching a section moved no row, so a section written by it is current, and a second
11835        // table added to this file later would not make it stale.
11836        assert_eq!(held[0].generation, 1);
11837        assert!(held[0].usable(reader.table().generation()));
11838        assert_eq!(reader.payload(held[0]).expect("read the payload"), payload);
11839        assert_eq!(reader.extents(held[0]).expect("extent table").len(), 1);
11840
11841        fs::remove_file(&path).expect("clean up");
11842    }
11843
11844    #[test]
11845    fn attaching_a_section_answers_every_row_exactly_as_before() {
11846        // Section 3.1 end to end, and the reason the whole layer is safe to build incrementally. A
11847        // file with a section in it and the same file without one have to agree row for row, so the
11848        // comparison is made against the answers taken before the attach rather than against a
11849        // constant somebody typed.
11850        let path = linked_file("attach_changes_nothing", 300);
11851        let before = Reader::open(&path).expect("open before");
11852        let rows = before.table().rows();
11853        let first = before.read(0, &[0]).expect("read before");
11854        let values = (0..rows).map(|at| first.value_at(at, 0)).collect::<Vec<_>>();
11855        let layout = before.layout().columns_total();
11856        drop(before);
11857
11858        let payload = a_key_map_payload();
11859        attach(
11860            &path,
11861            "items",
11862            &[section::Attachment {
11863                kind: *section::KEY_MAP,
11864                id: 0,
11865                flags: 0,
11866                header_bytes: 0,
11867                bytes: &payload,
11868            }],
11869        )
11870        .expect("attach");
11871
11872        let after = Reader::open(&path).expect("open after");
11873        assert_eq!(after.table().rows(), rows);
11874        let read = after.read(0, &[0]).expect("read after");
11875        for (at, value) in values.iter().enumerate() {
11876            assert_eq!(&read.value_at(at, 0), value, "row {at} moved");
11877        }
11878        assert_eq!(
11879            after.layout().columns_total(),
11880            layout,
11881            "an attach appends and does not rewrite a column page"
11882        );
11883
11884        fs::remove_file(&path).expect("clean up");
11885    }
11886
11887    #[test]
11888    fn a_rebuilt_section_replaces_the_one_it_supersedes() {
11889        // Rebuilding a key map has to be a write and not a question. If an attach added rather than
11890        // replaced, a table rebuilt a few times would name several maps for one column and a reader
11891        // would have to pick, which is a decision with no right answer in it.
11892        let path = linked_file("attach_twice", 32);
11893        let one = a_key_map_payload();
11894        let two = vec![7_u8; 1024];
11895        let entry = |bytes| section::Attachment {
11896            kind: *section::KEY_MAP,
11897            id: 4,
11898            flags: 1,
11899            header_bytes: 0,
11900            bytes,
11901        };
11902        attach(&path, "items", &[entry(&one)]).expect("first build");
11903        attach(&path, "items", &[entry(&two)]).expect("rebuild");
11904
11905        let reader = Reader::open(&path).expect("reopen");
11906        let held = attached(reader.table());
11907        assert_eq!(held.len(), 1, "one map per column and not one per build");
11908        assert_eq!(reader.payload(held[0]).expect("payload"), two);
11909
11910        fs::remove_file(&path).expect("clean up");
11911    }
11912
11913    #[test]
11914    fn an_attach_carries_through_a_kind_it_does_not_know() {
11915        // The first of section 3.2's three rules, at the point where it is easiest to break: a build
11916        // that rewrites a directory has to carry an entry it has no name for, or opening a file with
11917        // an older binary and attaching one section quietly deletes the work of a newer one.
11918        let path = linked_file("attach_unknown", 16);
11919        let payload = vec![3_u8; 96];
11920        attach(
11921            &path,
11922            "items",
11923            &[section::Attachment {
11924                kind: *b"RUDBZZ9\0",
11925                id: 1,
11926                flags: 0,
11927                header_bytes: 0,
11928                bytes: &payload,
11929            }],
11930        )
11931        .expect("a kind this build does not know still writes");
11932        let key_map = a_key_map_payload();
11933        attach(
11934            &path,
11935            "items",
11936            &[section::Attachment {
11937                kind: *section::KEY_MAP,
11938                id: 0,
11939                flags: 0,
11940                header_bytes: 0,
11941                bytes: &key_map,
11942            }],
11943        )
11944        .expect("attach beside it");
11945
11946        let reader = Reader::open(&path).expect("reopen");
11947        let held = attached(reader.table());
11948        assert_eq!(held.len(), 2, "the unfamiliar entry survived a directory rewrite");
11949        let unknown = held.iter().find(|one| !one.known()).expect("the unfamiliar one");
11950        assert_eq!(reader.payload(unknown).expect("its bytes are still there"), payload);
11951
11952        fs::remove_file(&path).expect("clean up");
11953    }
11954
11955    #[test]
11956    fn a_payload_of_nothing_is_a_relationship_recorded_as_not_built() {
11957        let path = linked_file("attach_not_built", 8);
11958        attach(
11959            &path,
11960            "items",
11961            &[section::Attachment {
11962                kind: *section::FORWARD_LINK,
11963                id: 2,
11964                flags: 0,
11965                header_bytes: 0,
11966                bytes: &[],
11967            }],
11968        )
11969        .expect("record a link that did not fit the budget");
11970
11971        let reader = Reader::open(&path).expect("reopen");
11972        let held = attached(reader.table());
11973        assert_eq!(held.len(), 1);
11974        assert_eq!(held[0].extents, 0);
11975        assert_eq!(held[0].extent_page, 0, "an entry that names no bytes points at none");
11976        assert!(reader.extents(held[0]).expect("no extent table").is_empty());
11977        assert!(reader.payload(held[0]).expect("no payload").is_empty());
11978
11979        fs::remove_file(&path).expect("clean up");
11980    }
11981
11982    #[test]
11983    fn a_payload_past_one_extent_is_split_and_joined_back() {
11984        // Issue #745's rule, exercised rather than argued. One byte past the bound is the smallest
11985        // payload that has to be two extents, and it is the case a split written for the common
11986        // size gets wrong.
11987        let path = linked_file("attach_two_extents", 8);
11988        let payload = vec![0x5a_u8; section::MAX_EXTENT as usize + 1];
11989        attach(
11990            &path,
11991            "items",
11992            &[section::Attachment {
11993                kind: *section::KEY_MAP,
11994                id: 0,
11995                flags: 0,
11996                header_bytes: 0,
11997                bytes: &payload,
11998            }],
11999        )
12000        .expect("attach a payload past the bound");
12001
12002        let reader = Reader::open(&path).expect("reopen");
12003        let held = attached(reader.table());
12004        let extents = reader.extents(held[0]).expect("extent table");
12005        assert_eq!(extents.len(), 2, "one byte past the bound is two extents");
12006        assert_eq!(extents[0].length, section::MAX_EXTENT);
12007        assert_eq!(extents[1].length, 1);
12008        assert_eq!(extents[1].first, u64::from(section::MAX_EXTENT));
12009        // And the extent the caller wants is readable on its own, which is the point of the split.
12010        assert_eq!(reader.extent(&extents[1]).expect("the last extent"), vec![0x5a]);
12011        assert_eq!(reader.payload(held[0]).expect("the whole payload").len(), payload.len());
12012
12013        fs::remove_file(&path).expect("clean up");
12014    }
12015
12016    #[test]
12017    fn a_torn_extent_is_refused_rather_than_decoded() {
12018        let path = linked_file("attach_torn", 8);
12019        let payload = a_key_map_payload();
12020        attach(
12021            &path,
12022            "items",
12023            &[section::Attachment {
12024                kind: *section::KEY_MAP,
12025                id: 0,
12026                flags: 0,
12027                header_bytes: 0,
12028                bytes: &payload,
12029            }],
12030        )
12031        .expect("attach");
12032
12033        let reader = Reader::open(&path).expect("reopen");
12034        let extent = reader.extents(&reader.table().sections()[0]).expect("extent table")[0];
12035        let file = OpenOptions::new().write(true).open(&path).expect("reopen to corrupt");
12036        write_at(&file, extent.offset + 7, &[0xff]).expect("flip a byte of the payload");
12037        drop(file);
12038
12039        let reader = Reader::open(&path).expect("the table still opens");
12040        let error = reader
12041            .payload(&reader.table().sections()[0])
12042            .expect_err("a corrupt payload is not handed out");
12043        assert!(error.to_string().contains("checksum"), "{error}");
12044        // And the table is still readable, which is section 3.1: a section that cannot be trusted
12045        // costs the query its shortcut and nothing else.
12046        assert_eq!(reader.read(0, &[0]).expect("the column is untouched").width(), 1);
12047
12048        fs::remove_file(&path).expect("clean up");
12049    }
12050
12051    #[test]
12052    fn attaching_to_a_file_of_the_previous_format_is_refused_rather_than_done() {
12053        // Readable is not writable. A format 22 directory has no section block, and adding one
12054        // without moving the number in the header would leave a file claiming a format it is not.
12055        let path = linked_file("attach_old_format", 8);
12056        let file = OpenOptions::new().write(true).open(&path).expect("reopen to patch");
12057        write_at(&file, 8, &22_u32.to_le_bytes()).expect("stamp the older format");
12058        drop(file);
12059
12060        let payload = a_key_map_payload();
12061        let error = attach(
12062            &path,
12063            "items",
12064            &[section::Attachment {
12065                kind: *section::KEY_MAP,
12066                id: 0,
12067                flags: 0,
12068                header_bytes: 0,
12069                bytes: &payload,
12070            }],
12071        )
12072        .expect_err("format 22 cannot gain a section");
12073        assert!(error.to_string().contains("format 22"), "{error}");
12074        assert!(Reader::open(&path).expect("and the file is untouched").table().rows() == 8);
12075
12076        fs::remove_file(&path).expect("clean up");
12077    }
12078
12079    #[test]
12080    fn a_section_header_longer_than_its_payload_is_refused_at_the_write() {
12081        let path = linked_file("attach_bad_header", 8);
12082        let error = attach(
12083            &path,
12084            "items",
12085            &[section::Attachment {
12086                kind: *section::KEY_MAP,
12087                id: 0,
12088                flags: 0,
12089                header_bytes: 40,
12090                bytes: &[1, 2, 3],
12091            }],
12092        )
12093        .expect_err("a writer's bug stops at the write");
12094        assert!(error.to_string().contains("header is longer"), "{error}");
12095
12096        fs::remove_file(&path).expect("clean up");
12097    }
12098
12099    #[test]
12100    fn attaching_to_a_name_the_file_does_not_hold_says_so() {
12101        let path = linked_file("attach_wrong_name", 8);
12102        let error = attach(&path, "orders", &[]).expect_err("no such table");
12103        assert!(error.to_string().contains("orders"), "{error}");
12104        fs::remove_file(&path).expect("clean up");
12105    }
12106
12107    #[test]
12108    fn the_planner_gets_an_exact_count_for_a_leading_value_of_an_incomplete_synopsis() {
12109        // The case a complete synopsis does not cover, and the one worth the most. 16,000 rows over
12110        // 601 distinct values, 10,000 of them holding a single value and the rest spread ten apiece
12111        // over six hundred more. The writer holds 512 values, so the list is a prefix and most of
12112        // the tail is outside it. The counts inside it are still exact, because the pass recounts
12113        // the candidates that survived it, so `id = 1` is ten thousand rows rather than the
12114        // twenty six a distinct count of 601 would divide its way to.
12115        let path = path("frequency_prefix_for_the_planner");
12116        let mut writer =
12117            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
12118                .expect("new file");
12119        let mut values = vec![Value::Integer(1); 10_000];
12120        for _ in 0..10 {
12121            values.extend((0..600).map(|tail| Value::Integer(1_000 + tail)));
12122        }
12123        // A vector holds 8,192 rows, so this goes in as several parts. The pass that takes the
12124        // synopsis walks the whole column rather than a part, so the counts are the same either way.
12125        for part in values.chunks(8_000) {
12126            let rows = Chunk::new(vec![
12127                Vector::from_values(LogicalType::Integer, part).expect("integers"),
12128            ])
12129            .expect("one column");
12130            writer.append(&rows).expect("a part");
12131        }
12132        writer.finish().expect("commit");
12133        let reader = Reader::open(&path).expect("reopen from disk");
12134        let prefix =
12135            reader.frequency_prefix(0).expect("a readable synopsis").expect("the column has one");
12136        // A prefix and not the whole column, and the writer said how many rows anything left out of
12137        // it can hold.
12138        assert_eq!(prefix.entries.len(), 512);
12139        assert_eq!(prefix.omitted_max, 10);
12140        let common = Common::new(reader);
12141        assert_eq!(common.rows(), 16_000);
12142        let column = common.column("id").expect("the file has that column");
12143        assert_eq!(
12144            common.rows_with(column, &Bound::Int(1)),
12145            Stat::exact(10_000, Provenance::FrequencySynopsis)
12146        );
12147        // In the prefix, because ties go to the smaller value and the prefix reaches 1,510.
12148        assert_eq!(
12149            common.rows_with(column, &Bound::Int(1_100)),
12150            Stat::exact(10, Provenance::FrequencySynopsis)
12151        );
12152        // Outside it, and a prefix says nothing about a value it does not list. Not zero, which is
12153        // what a complete list would say, and the file holds ten rows of this one.
12154        assert_eq!(common.rows_with(column, &Bound::Int(1_550)), Stat::Unknown);
12155        // Not in the file at all, and still nothing rather than a zero. A prefix cannot tell the
12156        // two apart, which is the whole of what it gives up.
12157        assert_eq!(common.rows_with(column, &Bound::Int(9_999)), Stat::Unknown);
12158        // What the prefix left out, which is what turns the unknown above into a number. The 512
12159        // entries account for 15,110 rows, so 890 are left for the 89 values the writer dropped,
12160        // and 890 over 89 is the ten rows each of them really holds.
12161        let remainder = common.remainder(column).expect("the list is a prefix");
12162        assert_eq!(remainder, Remainder { rows: 890, listed: 512, most: 10 });
12163        assert_eq!(remainder.rows / (601 - remainder.listed), 10);
12164        fs::remove_file(&path).expect("clean up");
12165    }
12166
12167    /// A file with no table in it is a file, and opening it says so rather than failing.
12168    #[test]
12169    fn a_file_holding_no_table_commits_and_opens_and_a_table_can_be_added_to_it() {
12170        let path = path("empty");
12171        Writer::empty(&path, &[]).expect("a file with nothing in it");
12172        let catalog = Catalog::open(&path).expect("the empty file opens");
12173        assert_eq!(catalog.len(), 0);
12174        assert!(catalog.is_empty());
12175        assert_eq!(catalog.names().count(), 0);
12176        // The next generation goes over the top of it the way it goes over any other, which is what
12177        // says this is a committed file and not a special case somebody has to know about.
12178        let mut writer =
12179            Writer::open(&path, "items", vec![Field::required("id", LogicalType::Integer)])
12180                .expect("a table goes into the empty file");
12181        writer.append(&sample_ids()).expect("rows");
12182        writer.finish().expect("commit");
12183        let catalog = Catalog::open(&path).expect("the file opens again");
12184        assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
12185        fs::remove_file(&path).expect("clean up");
12186    }
12187
12188    /// A committed table with no rows is a name the next generation takes over, and one with rows
12189    /// is a name it refuses.
12190    ///
12191    /// The refusal is what it always was and it is load bearing: carrying a table that holds rows
12192    /// forward means reading and rewriting its pages, and a writer that quietly wrote a second
12193    /// entry under the same name would leave a file with two tables a reader cannot tell apart. An
12194    /// empty one has no pages and no reader, so there is nothing to carry and nothing to lose, and
12195    /// taking its place is what lets a schema committed by an earlier session be loaded by a stream
12196    /// instead of through memory.
12197    #[test]
12198    fn a_committed_empty_table_gives_up_its_name_and_one_with_rows_does_not() {
12199        let path = path("empty-name");
12200        let field = || vec![Field::required("id", LogicalType::Integer)];
12201        Writer::create(&path, "items", field()).expect("new file").finish().expect("commit");
12202        let catalog = Catalog::open(&path).expect("the file opens");
12203        assert_eq!(catalog.rows().collect::<Vec<_>>(), vec![("items", 0)]);
12204
12205        let mut writer = Writer::open(&path, "items", field()).expect("the empty name is free");
12206        writer.append(&sample_ids()).expect("rows");
12207        writer.finish().expect("commit");
12208        let catalog = Catalog::open(&path).expect("the file opens again");
12209        // One entry and not two. The generation replaced the empty table rather than joining it.
12210        assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
12211        let held = catalog.rows().collect::<Vec<_>>();
12212        assert_eq!(held.len(), 1);
12213        assert!(held[0].1 > 0, "the rows that were appended are the ones the catalog counts");
12214
12215        // The same call against the same name now that it holds rows, which is still refused.
12216        let error = Writer::open(&path, "items", field()).expect_err("a name with rows is taken");
12217        assert!(error.to_string().contains("same name"), "{error}");
12218        fs::remove_file(&path).expect("clean up");
12219    }
12220
12221    /// A view, with everything about it that a reopened catalog has to be able to answer from.
12222    fn sample_view(name: &str) -> ViewEntry {
12223        ViewEntry {
12224            name: name.to_string(),
12225            sql: "SELECT id FROM items WHERE id > 0".to_string(),
12226            statement: format!("CREATE VIEW {name} AS SELECT id FROM items WHERE (id > 0);"),
12227            aliases: vec!["n".to_string()],
12228            columns: vec![Field::new("n", LogicalType::Integer)],
12229        }
12230    }
12231
12232    #[test]
12233    fn a_view_written_into_the_catalog_comes_back_whole() {
12234        let path = path("views");
12235        let mut writer =
12236            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
12237                .expect("new file");
12238        writer.append(&sample_ids()).expect("rows");
12239        writer.with_views(vec![sample_view("v")]).finish().expect("commit");
12240        let catalog = Catalog::open(&path).expect("reopen");
12241        assert_eq!(catalog.views().cloned().collect::<Vec<_>>(), vec![sample_view("v")]);
12242        // The tables are still there and are still read the same way, so the section on the end did
12243        // not move anything in front of it.
12244        assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
12245        fs::remove_file(&path).expect("clean up");
12246    }
12247
12248    /// A writer opened to append a table says nothing about views and must not lose them.
12249    #[test]
12250    fn appending_a_table_carries_the_views_forward() {
12251        let path = path("viewscarry");
12252        let mut writer =
12253            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
12254                .expect("new file");
12255        writer.append(&sample_ids()).expect("rows");
12256        writer.with_views(vec![sample_view("v")]).finish().expect("commit");
12257        let mut writer =
12258            Writer::open(&path, "other", vec![Field::required("id", LogicalType::Integer)])
12259                .expect("a second table");
12260        writer.append(&sample_ids()).expect("rows");
12261        writer.finish().expect("commit");
12262        let catalog = Catalog::open(&path).expect("reopen");
12263        assert_eq!(catalog.views().count(), 1);
12264        assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items", "other"]);
12265        fs::remove_file(&path).expect("clean up");
12266    }
12267
12268    /// The whole point of [`Writer::restate`]: the views change and the pages do not move.
12269    #[test]
12270    fn restating_the_views_leaves_every_table_where_it_was() {
12271        let path = path("restate");
12272        let mut writer =
12273            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
12274                .expect("new file");
12275        writer.append(&sample_ids()).expect("rows");
12276        writer.finish().expect("commit");
12277        let before = fs::metadata(&path).expect("the file is there").len();
12278        Writer::restate(&path, &[sample_view("v"), sample_view("w")]).expect("two views");
12279        let catalog = Catalog::open(&path).expect("reopen");
12280        assert_eq!(catalog.views().count(), 2);
12281        assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
12282        // A catalog on the end and nothing else, so what it grew by is the size of a catalog rather
12283        // than the size of the table.
12284        let after = fs::metadata(&path).expect("the file is there").len();
12285        assert!(after > before, "a generation was written");
12286        assert!(after - before < before, "the table was not written again");
12287        // The rows are still readable through the new generation, which is the part that would go
12288        // wrong if the catalog carried the wrong directory pointers forward.
12289        let reader = Catalog::open(&path).expect("reopen").table("items").expect("the table");
12290        assert_eq!(reader.table().rows, 3);
12291        // And a restate over a restate keeps working, because each one reads the slot that
12292        // checksummed rather than the highest number in the header.
12293        Writer::restate(&path, &[]).expect("no views at all");
12294        assert_eq!(Catalog::open(&path).expect("reopen").views().count(), 0);
12295        fs::remove_file(&path).expect("clean up");
12296    }
12297
12298    /// Two entries under one name is a catalog no lookup can answer, whichever two they are.
12299    #[test]
12300    fn a_view_named_after_a_table_is_refused_when_the_catalog_is_read() {
12301        let bytes = encode_catalog(
12302            &[Entry {
12303                name: "items".to_string(),
12304                fields: vec![Field::required("id", LogicalType::Integer)],
12305                rows: 1,
12306                directory: Page { offset: HEADER, length: 8, hash: 0 },
12307                nonzero: vec![None],
12308                aggregates: vec![None],
12309                distincts: vec![None],
12310                extremes: vec![None],
12311                frequencies: vec![None],
12312            }],
12313            &[sample_view("items")],
12314        )
12315        .expect("it encodes, because encoding does not look");
12316        let error = decode_catalog(&bytes, HEADER + 8).expect_err("and decoding does");
12317        assert!(error.to_string().contains("same name"), "{error}");
12318    }
12319
12320    /// A compressed text page read at some rows is those rows of the page read whole, nulls and
12321    /// all, and a row past the end or rows out of order are refused rather than guessed at.
12322    #[test]
12323    fn a_compressed_text_page_read_at_some_rows_is_those_rows_of_the_whole() {
12324        let rows: usize = 300;
12325        let text: Vec<String> =
12326            (0..rows).map(|row| format!("a street named after number {}", row * 7)).collect();
12327        let values: Vec<&[u8]> = text.iter().map(String::as_bytes).collect();
12328        let mut page = vec![6, 2];
12329        page.extend((0..rows.div_ceil(8)).map(|byte| {
12330            (0..8).filter(|bit| (byte * 8 + bit) % 5 != 3).fold(0_u8, |mask, bit| mask | 1 << bit)
12331        }));
12332        let compressed = string::encode_only(string::Kind::Fsst, &values)
12333            .expect("encoded")
12334            .expect("text this repetitive compresses");
12335        page.extend_from_slice(&compressed);
12336        let whole = decode(&LogicalType::Varchar, rows, &page, None).expect("the whole page");
12337        let positions = [0_u32, 3, 8, 13, 200, 299];
12338        let some =
12339            decode_at(&LogicalType::Varchar, rows, &page, None, &positions).expect("some rows");
12340        assert_eq!(some.len(), positions.len());
12341        for (at, &row) in positions.iter().enumerate() {
12342            assert_eq!(some.value_at(at), whole.value_at(row as usize), "row {row}");
12343        }
12344        assert_eq!(some.value_at(1), Value::Null, "row 3 is null");
12345        assert!(decode_at(&LogicalType::Varchar, rows, &page, None, &[300]).is_err());
12346        assert!(decode_at(&LogicalType::Varchar, rows, &page, None, &[8, 3]).is_err());
12347    }
12348
12349    /// Every column of a part read at some rows is the part read whole and gathered, whatever the
12350    /// page holds.
12351    #[test]
12352    fn a_part_read_at_some_rows_is_the_part_read_whole_and_gathered() {
12353        let path = path("rows");
12354        let mut writer = Writer::create(
12355            &path,
12356            "items",
12357            vec![
12358                Field::required("id", LogicalType::Integer),
12359                Field::new("text", LogicalType::Varchar),
12360            ],
12361        )
12362        .expect("new file");
12363        let rows = 2_000;
12364        let chunk = Chunk::new(vec![
12365            Vector::from_values(
12366                LogicalType::Integer,
12367                &(0..rows).map(Value::Integer).collect::<Vec<_>>(),
12368            )
12369            .expect("integers"),
12370            Vector::from_values(
12371                LogicalType::Varchar,
12372                &(0..rows)
12373                    .map(|row| {
12374                        if row % 7 == 2 {
12375                            Value::Null
12376                        } else {
12377                            Value::Varchar(format!("a comment about order {}", row * 13))
12378                        }
12379                    })
12380                    .collect::<Vec<_>>(),
12381            )
12382            .expect("strings"),
12383        ])
12384        .expect("matching rows");
12385        writer.append(&chunk).expect("one part");
12386        writer.finish().expect("commit");
12387        let reader = Reader::open(&path).expect("reopen from disk");
12388        let positions = [1_u32, 2, 9, 1_000, 1_999];
12389        for whole in [true, false] {
12390            let some = reader.read_rows(0, &[0, 1], &positions, whole).expect("some rows");
12391            let all = reader.read(0, &[0, 1]).expect("the whole part");
12392            assert_eq!(some.len(), positions.len());
12393            for column in 0..2 {
12394                for (at, &row) in positions.iter().enumerate() {
12395                    assert_eq!(some.value_at(at, column), all.value_at(row as usize, column));
12396                }
12397            }
12398        }
12399        assert!(reader.read_rows(0, &[1], &[2_000], true).is_err());
12400    }
12401
12402    #[test]
12403    fn committed_file_reopens_and_reads_only_requested_columns() {
12404        let path = path("reopen");
12405        let mut writer = Writer::create(
12406            &path,
12407            "items",
12408            vec![
12409                Field::required("id", LogicalType::Integer),
12410                Field::new("text", LogicalType::Varchar),
12411            ],
12412        )
12413        .expect("new file");
12414        writer.append(&sample()).expect("first part");
12415        writer.append(&sample()).expect("second part");
12416        writer.finish().expect("commit");
12417        let reader = Reader::open(&path).expect("reopen from disk");
12418        assert_eq!(reader.table().rows(), 6);
12419        // Two appends below the stripe bound are two parts of one stripe, which is the whole point
12420        // of the split: the directory describes the stripe and the scan still reads a part.
12421        assert_eq!(reader.table().stripes().len(), 1);
12422        assert_eq!(reader.parts(), 2);
12423        assert_eq!(reader.part_rows(0), 3);
12424        assert_eq!(reader.part_rows(1), 3);
12425        let text = reader.read(1, &[1]).expect("only text page");
12426        assert_eq!(text.width(), 1);
12427        assert_eq!(text.value_at(1, 0), Value::Null);
12428        assert_eq!(text.value_at(2, 0), Value::Varchar("long text after a slash".into()));
12429        let sparse = reader.read_sparse(1, &[1]).expect("one part without its whole page");
12430        assert_eq!(sparse.width(), 1);
12431        assert_eq!(sparse.value_at(1, 0), Value::Null);
12432        assert_eq!(sparse.value_at(2, 0), Value::Varchar("long text after a slash".into()));
12433        assert!(!reader.skips_codes(0, 1, &[0]).expect("alpha is in the stripe"));
12434        assert!(!reader.skips_codes(0, 1, &[2]).expect("long text is in the stripe"));
12435        assert!(reader.skips_codes(0, 1, &[3]).expect("unknown code is absent"));
12436        let count = reader.read(0, &[]).expect("no page is needed for count");
12437        assert_eq!(count.len(), 3);
12438        assert!(reader.skips(0, &[Probe { column: 0, op: Op::Greater, value: Bound::Int(100) }]));
12439        assert!(!reader.skips(0, &[Probe { column: 0, op: Op::Greater, value: Bound::Int(0) }]));
12440        let integers = reader.top_frequencies(0, 1).expect("valid integer synopsis").expect("kept");
12441        assert_eq!(
12442            integers,
12443            vec![(Value::Integer(-2), 2), (Value::Integer(4), 2), (Value::Integer(9), 2),]
12444        );
12445        let strings = reader.top_frequencies(1, 1).expect("valid string synopsis").expect("kept");
12446        assert_eq!(strings.len(), 3);
12447        assert!(strings.contains(&(Value::Null, 2)));
12448        assert!(strings.contains(&(Value::Varchar("alpha".into()), 2)));
12449        assert!(strings.contains(&(Value::Varchar("long text after a slash".into()), 2)));
12450        fs::remove_file(path).expect("remove scratch file");
12451    }
12452
12453    /// Two pipeline instances handing over whole runs, which is what makes the native sink safe to
12454    /// instance.
12455    ///
12456    /// The runs arrive in the order the instances finished reading them rather than in source
12457    /// order, and the second one to finish is the one that read the earlier rows. Each run is still
12458    /// a stripe of its own and the table still reads back in source order, which is the whole of
12459    /// what the writer promises about ordering.
12460    #[test]
12461    fn runs_handed_over_out_of_order_still_read_back_in_source_order() {
12462        let path = path("interleaved-runs");
12463        let mut writer =
12464            Writer::create(&path, "interleaved", vec![Field::new("v", LogicalType::BigInt)])
12465                .expect("new file");
12466        for morsel in [2_u64, 0, 3, 1] {
12467            let parts = (0..4_u64)
12468                .map(|chunk| {
12469                    let first = i64::try_from(morsel * 32 + chunk * 8).expect("small");
12470                    let values =
12471                        (0..8_i64).map(|row| Value::BigInt(first + row)).collect::<Vec<_>>();
12472                    let column =
12473                        Vector::from_values(LogicalType::BigInt, &values).expect("a column");
12474                    ((morsel, chunk), Chunk::new(vec![column]).expect("one column"))
12475                })
12476                .collect::<Vec<_>>();
12477            writer.append_stripe(parts).expect("a stripe");
12478        }
12479        writer.finish().expect("commit");
12480
12481        let reader = Reader::open(&path).expect("valid directory");
12482        assert_eq!(reader.table().stripes().len(), 4, "a run is a stripe of its own");
12483        assert_eq!(reader.table().rows(), 128);
12484        for part in 0..16_usize {
12485            let read = reader.read(part, &[0]).expect("a part back");
12486            for row in 0..8_usize {
12487                let want = i64::try_from(part * 8 + row).expect("small");
12488                assert_eq!(read.value_at(row, 0), Value::BigInt(want), "part {part} row {row}");
12489            }
12490        }
12491        fs::remove_file(path).expect("remove scratch file");
12492    }
12493
12494    /// Runs from different callers may interleave and may not overlap, and the commit is what
12495    /// catches an overlap.
12496    #[test]
12497    fn runs_that_overlap_each_other_are_refused_at_commit() {
12498        let path = path("overlapping-runs");
12499        let mut writer =
12500            Writer::create(&path, "overlapping", vec![Field::new("v", LogicalType::BigInt)])
12501                .expect("new file");
12502        let one = |order: (u64, u64)| {
12503            let column =
12504                Vector::from_values(LogicalType::BigInt, &[Value::BigInt(1)]).expect("a column");
12505            (order, Chunk::new(vec![column]).expect("one column"))
12506        };
12507        // The second run sits inside the first rather than after it, which is a thing no instance
12508        // holding its own contiguous run can produce and a thing the file cannot represent.
12509        writer.append_stripe(vec![one((0, 0)), one((0, 2))]).expect("a stripe");
12510        writer.append_stripe(vec![one((0, 1))]).expect("a stripe");
12511        let error = writer.finish().expect_err("the runs overlap");
12512        assert!(error.message().contains("source order"), "{error}");
12513        fs::remove_file(path).expect("remove scratch file");
12514    }
12515
12516    /// A stripe holds [`STRIPE_PARTS`] parts, so a run longer than that is a caller bug rather than
12517    /// something to split, and the writer says so at the door instead of quietly cutting it in two.
12518    #[test]
12519    fn a_run_longer_than_a_stripe_is_refused() {
12520        let path = path("overlong-run");
12521        let mut writer =
12522            Writer::create(&path, "overlong", vec![Field::new("v", LogicalType::BigInt)])
12523                .expect("new file");
12524        let parts = (0..=STRIPE_PARTS)
12525            .map(|at| {
12526                let column = Vector::from_values(LogicalType::BigInt, &[Value::BigInt(1)])
12527                    .expect("a column");
12528                let chunk = Chunk::new(vec![column]).expect("one column");
12529                ((0, u64::try_from(at).expect("small")), chunk)
12530            })
12531            .collect::<Vec<_>>();
12532        let error = writer.append_stripe(parts).expect_err("one part too many");
12533        assert!(error.message().contains("more parts than it holds"), "{error}");
12534        fs::remove_file(path).expect("remove scratch file");
12535    }
12536
12537    /// Parts past the stripe bound start a new stripe, and every part stays addressable on its own.
12538    ///
12539    /// This is the shape the format exists for, so both ends of the split are checked here. The
12540    /// directory holds three stripes rather than a hundred and thirty one, and a read of any one
12541    /// part still answers with that part's rows rather than with its whole stripe's.
12542    #[test]
12543    fn parts_past_the_stripe_bound_start_a_new_stripe() {
12544        let path = path("stripe-bound");
12545        let mut writer = Writer::create(
12546            &path,
12547            "items",
12548            vec![
12549                Field::required("id", LogicalType::Integer),
12550                Field::new("text", LogicalType::Varchar),
12551            ],
12552        )
12553        .expect("new file");
12554        let parts = STRIPE_PARTS * 2 + 3;
12555        for part in 0..parts {
12556            let id = part as i32;
12557            let chunk = Chunk::new(vec![
12558                Vector::from_values(
12559                    LogicalType::Integer,
12560                    &[Value::Integer(id), Value::Integer(-id)],
12561                )
12562                .expect("integers"),
12563                Vector::from_values(
12564                    LogicalType::Varchar,
12565                    &[Value::Varchar(format!("value {part}")), Value::Null],
12566                )
12567                .expect("strings"),
12568            ])
12569            .expect("matching rows");
12570            writer.append(&chunk).expect("one part");
12571        }
12572        writer.finish().expect("commit");
12573
12574        let reader = Reader::open(&path).expect("reopen from disk");
12575        assert_eq!(reader.parts(), parts);
12576        assert_eq!(reader.table().rows(), parts * 2);
12577        assert_eq!(reader.table().stripes().len(), parts.div_ceil(STRIPE_PARTS));
12578        assert_eq!(reader.table().stripes()[0].parts(), STRIPE_PARTS);
12579        assert_eq!(reader.table().stripes()[0].rows(), STRIPE_PARTS * 2);
12580        assert_eq!(reader.table().stripes()[2].parts(), 3);
12581        // Backwards on purpose. The reader keeps four stripes a column, so a scan that walks the
12582        // table the other way is what catches a cache that only ever holds what it just read.
12583        for part in (0..parts).rev() {
12584            let dense = reader.read(part, &[0, 1]).expect("a whole page read");
12585            let sparse = reader.read_sparse(part, &[0, 1]).expect("one part read");
12586            for chunk in [&dense, &sparse] {
12587                assert_eq!(chunk.len(), 2, "part {part} has its own row count");
12588                assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
12589                assert_eq!(chunk.value_at(1, 0), Value::Integer(-(part as i32)));
12590                assert_eq!(chunk.value_at(0, 1), Value::Varchar(format!("value {part}")));
12591                assert_eq!(chunk.value_at(1, 1), Value::Null);
12592            }
12593        }
12594        // The bounds are merged over the stripe, so they answer for the range the whole stripe
12595        // covers and not for the part that was asked about.
12596        let above = [Probe { column: 0, op: Op::Greater, value: Bound::Int(100) }];
12597        assert!(reader.skips(0, &above), "the first stripe stops at 63");
12598        assert!(!reader.skips(STRIPE_PARTS * 2, &above), "the third stripe reaches 130");
12599        fs::remove_file(path).expect("remove scratch file");
12600    }
12601
12602    /// A scattered value in the column that decides `WHERE UserID = ?`.
12603    fn scattered(n: i64) -> i64 {
12604        n.wrapping_mul(-7_046_029_254_386_353_131)
12605    }
12606
12607    /// A part whose sieve does not hold the constant is skipped, and a range would skip none of them.
12608    ///
12609    /// This is ClickBench query 19 in miniature. The values are spread over the whole of `BIGINT`, so
12610    /// every stripe's bounds cover nearly all of it and rule out nothing, and the part that really
12611    /// holds the value is the only one a scan has to read.
12612    #[test]
12613    fn a_part_is_skipped_when_its_sieve_does_not_hold_the_constant() {
12614        let path = path("sieve-skip");
12615        let mut writer =
12616            Writer::create(&path, "hits", vec![Field::required("id", LogicalType::BigInt)])
12617                .expect("new file");
12618        let parts = STRIPE_PARTS + 3;
12619        // Big enough that the filter is worth its bytes. A part of eight numbers packs to under a
12620        // hundred bytes and the smallest filter there is is sixty nine, so a filter over a part
12621        // that small costs about as much to read as the rows do and is no longer written.
12622        let per_part = 128;
12623        for part in 0..parts {
12624            let held: Vec<Value> = (0..per_part)
12625                .map(|row| Value::BigInt(scattered((part * per_part + row) as i64)))
12626                .collect();
12627            let chunk =
12628                Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
12629                    .expect("one column");
12630            writer.append(&chunk).expect("one part");
12631        }
12632        writer.finish().expect("commit");
12633
12634        let reader = Reader::open(&path).expect("reopen from disk");
12635        let probe = |value: i64| Probe {
12636            column: 0,
12637            op: Op::Equal,
12638            value: Bound::Int(i128::from(scattered(value))),
12639        };
12640        for wanted in [0_i64, (per_part + 1) as i64, (parts * per_part - 1) as i64] {
12641            let tests = [probe(wanted)];
12642            let kept: Vec<usize> = (0..parts).filter(|&part| !reader.skips(part, &tests)).collect();
12643            let home = wanted as usize / per_part;
12644            assert!(kept.contains(&home), "the part holding {wanted} is read");
12645            // A filter answers maybe, so a part it keeps need not hold the value. Sixty seven parts
12646            // of a hundred and twenty eight numbers each, at a dozen bits a value, is about one
12647            // stray part across the whole file and that is what this leaves room for.
12648            assert!(kept.len() <= 2, "{wanted} keeps {kept:?}, which is more than one stray part");
12649        }
12650        let absent = [probe((parts * per_part) as i64 + 1)];
12651        let kept = (0..parts).filter(|&part| !reader.skips(part, &absent)).count();
12652        assert!(kept <= 1, "{kept} parts of {parts} kept a value no part holds");
12653        // The same probes against the bounds alone, which is what this replaces. A column of
12654        // scattered numbers has a range per stripe that covers nearly the whole type.
12655        let tests = [probe(0)];
12656        assert!(
12657            reader.table().stripes().iter().all(|stripe| !stripe.zone.skips(&tests)),
12658            "the bounds rule out no stripe at all"
12659        );
12660        fs::remove_file(path).expect("remove scratch file");
12661    }
12662
12663    /// A part whose own bounds rule out an ordered comparison is skipped where the stripe's keep it.
12664    ///
12665    /// This is the shape of ClickBench 24. Each part covers a narrow stretch of the column and the
12666    /// stripe covers all sixty four of them at once, so a comparison that lands inside the stripe
12667    /// rules out none of it and rules out all but a few parts.
12668    #[test]
12669    fn a_part_is_skipped_when_its_own_bounds_rule_out_a_comparison_the_stripe_keeps() {
12670        let path = path("part-range-skip");
12671        let mut writer =
12672            Writer::create(&path, "hits", vec![Field::required("at", LogicalType::BigInt)])
12673                .expect("new file");
12674        let parts = STRIPE_PARTS + 3;
12675        let per_part = 128;
12676        for part in 0..parts {
12677            // Scattered inside the part's own band rather than a run, because a run of
12678            // consecutive numbers encodes to a stride of a few bytes and then the page of ranges
12679            // costs more than reading the column it indexes, which is the case the writer declines.
12680            let held: Vec<Value> = (0..per_part)
12681                .map(|row| {
12682                    Value::BigInt((part * 1_000) as i64 + (scattered(row as i64).rem_euclid(900)))
12683                })
12684                .collect();
12685            let chunk =
12686                Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
12687                    .expect("one column");
12688            writer.append(&chunk).expect("one part");
12689        }
12690        writer.finish().expect("commit");
12691
12692        let reader = Reader::open(&path).expect("reopen from disk");
12693        let under = [Probe { column: 0, op: Op::Less, value: Bound::Int(3_000) }];
12694        let kept: Vec<usize> = (0..parts).filter(|&part| !reader.skips(part, &under)).collect();
12695        assert_eq!(kept, vec![0, 1, 2], "only the three parts that start under three thousand");
12696        // The same question asked of the stripe alone, which is what this replaces.
12697        assert!(!reader.stripe_skips(0, &under), "the stripe reaches from zero and keeps itself");
12698        fs::remove_file(path).expect("remove scratch file");
12699    }
12700
12701    /// The other half of the same page. A part whose own bounds put every row of it inside the
12702    /// filter is waved through, so the comparison never runs on it, where the stripe's bounds reach
12703    /// across every part and can prove nothing.
12704    #[test]
12705    fn a_part_is_waved_through_when_its_own_bounds_pass_a_comparison_the_stripe_cannot() {
12706        let path = path("part-range-certain");
12707        let mut writer =
12708            Writer::create(&path, "hits", vec![Field::required("at", LogicalType::BigInt)])
12709                .expect("new file");
12710        let parts = STRIPE_PARTS + 3;
12711        let per_part = 128;
12712        for part in 0..parts {
12713            let held: Vec<Value> = (0..per_part)
12714                .map(|row| {
12715                    Value::BigInt((part * 1_000) as i64 + (scattered(row as i64).rem_euclid(900)))
12716                })
12717                .collect();
12718            let chunk =
12719                Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
12720                    .expect("one column");
12721            writer.append(&chunk).expect("one part");
12722        }
12723        writer.finish().expect("commit");
12724
12725        let reader = Reader::open(&path).expect("reopen from disk");
12726        let under = [Probe { column: 0, op: Op::Less, value: Bound::Int(3_000) }];
12727        let waved: Vec<usize> = (0..parts).filter(|&part| reader.certain(part, &under)).collect();
12728        assert_eq!(waved, vec![0, 1, 2], "the three parts that end under three thousand");
12729        // The first stripe reaches from zero to past sixty thousand, so it straddles three thousand
12730        // and settles nothing either way. The three yeses above are the parts' own ends talking.
12731        assert!(!reader.stripe_skips(0, &under), "the stripe straddles the comparison");
12732        fs::remove_file(path).expect("remove scratch file");
12733    }
12734
12735    /// The page is worth its bytes on a column with parts to tell apart and is not written on one
12736    /// that has a single part, where the stripe bounds already are the part's.
12737    #[test]
12738    fn a_stripe_of_one_part_writes_no_range_page_and_a_stripe_of_many_does() {
12739        for (parts, wanted) in [(1_usize, false), (STRIPE_PARTS, true)] {
12740            let path = path("part-range-page");
12741            let mut writer =
12742                Writer::create(&path, "hits", vec![Field::required("at", LogicalType::BigInt)])
12743                    .expect("new file");
12744            for part in 0..parts {
12745                let held: Vec<Value> = (0..128)
12746                    .map(|row| {
12747                        Value::BigInt((part * 1_000) as i64 + scattered(row as i64).rem_euclid(900))
12748                    })
12749                    .collect();
12750                let chunk = Chunk::new(vec![
12751                    Vector::from_values(LogicalType::BigInt, &held).expect("numbers"),
12752                ])
12753                .expect("one column");
12754                writer.append(&chunk).expect("one part");
12755            }
12756            writer.finish().expect("commit");
12757            let reader = Reader::open(&path).expect("reopen from disk");
12758            let bytes = reader.layout().columns[0].part_ranges;
12759            assert_eq!(bytes > 0, wanted, "{parts} parts wrote {bytes} bytes of ranges");
12760            fs::remove_file(path).expect("remove scratch file");
12761        }
12762    }
12763
12764    /// A cut down string end is still an end on the side it was, which is the only thing that keeps
12765    /// a shortened bound from turning a skip into a wrong answer.
12766    #[test]
12767    fn a_string_end_that_is_cut_down_still_covers_the_value_it_came_from() {
12768        let long = vec![b'a'; PART_BOUND_BYTES * 2];
12769        let low = shortened(Some(Bound::Bytes(long.clone())), false).expect("a low end");
12770        let high = shortened(Some(Bound::Bytes(long.clone())), true).expect("a high end");
12771        let Bound::Bytes(low) = low else { panic!("a string stays a string") };
12772        let Bound::Bytes(high) = high else { panic!("a string stays a string") };
12773        assert!(low.len() <= PART_BOUND_BYTES && high.len() <= PART_BOUND_BYTES);
12774        assert!(low.as_slice() <= long.as_slice(), "the low end is at or under the value");
12775        assert!(high.as_slice() >= long.as_slice(), "the high end is at or over the value");
12776    }
12777
12778    /// A string of nothing but the largest byte has no prefix that can be stepped up, so the high
12779    /// end is given up rather than claimed too small. No end keeps the part, which is always safe.
12780    #[test]
12781    fn a_string_end_with_no_room_to_step_up_gives_up_the_bound() {
12782        let long = vec![u8::MAX; PART_BOUND_BYTES * 2];
12783        assert_eq!(shortened(Some(Bound::Bytes(long.clone())), true), None);
12784        let low = shortened(Some(Bound::Bytes(long)), false).expect("a low end is still a prefix");
12785        assert_eq!(low, Bound::Bytes(vec![u8::MAX; PART_BOUND_BYTES]));
12786    }
12787
12788    /// What a column is stored as, asked of two files holding the same rows in a different order.
12789    ///
12790    /// This is the question the report exists to answer and it is the one the directory cannot. The
12791    /// two files have the same rows, the same schema and the same number of parts, and the column
12792    /// comes out four times smaller in one of them, because ascending keys delta encode to a few
12793    /// bits a row and shuffled ones do not. Nothing about the file's shape says so. The page header
12794    /// says so, and reading it is what this does.
12795    ///
12796    /// It is q18 on TPC-H in miniature: clustering lineitem by ship date leaves `l_orderkey`
12797    /// ascending inside a partition but sparse, its deltas go from six bits to twelve, and the scan
12798    /// pays for the wider ones.
12799    #[test]
12800    fn what_a_column_is_stored_as_follows_the_order_the_rows_were_written_in() {
12801        let parts = 4;
12802        let per_part = 1024;
12803        let rows = parts * per_part;
12804        let written = |name: &str, keys: &[i64]| {
12805            let path = path(name);
12806            let fields = vec![Field::required("key", LogicalType::BigInt)];
12807            let mut writer = Writer::create(&path, "keys", fields).expect("new file");
12808            for part in 0..parts {
12809                let values: Vec<Value> = keys[part * per_part..(part + 1) * per_part]
12810                    .iter()
12811                    .map(|key| Value::BigInt(*key))
12812                    .collect();
12813                let chunk = Chunk::new(vec![
12814                    Vector::from_values(LogicalType::BigInt, &values).expect("numbers"),
12815                ])
12816                .expect("one column");
12817                writer.append(&chunk).expect("one part");
12818            }
12819            writer.finish().expect("commit");
12820            path
12821        };
12822        // Ascending with a small irregular step, which is what a key column in arrival order looks
12823        // like: an order has one to seven line items, so the key repeats and then moves on by one.
12824        let climbing = |step: &dyn Fn(usize) -> i64| {
12825            let mut key = 0;
12826            (0..rows)
12827                .map(|row| {
12828                    key += step(row);
12829                    key
12830                })
12831                .collect::<Vec<i64>>()
12832        };
12833        let ascending = climbing(&|row| (row % 3) as i64);
12834        // The same rows in the same direction over a range a thousand times wider, which is what a
12835        // partition of a clustered table holds: still ascending, and far enough apart that the
12836        // deltas no longer fit in a handful of bits.
12837        let sparse = climbing(&|row| ((row * 2_654_435_761) % 4096) as i64);
12838        let near_path = written("stored-near", &ascending);
12839        let far_path = written("stored-far", &sparse);
12840
12841        let one = Reader::open(&near_path).expect("reopen from disk");
12842        let other = Reader::open(&far_path).expect("reopen from disk");
12843        let near = one.stored(0).expect("the column is stored");
12844        let far = other.stored(0).expect("the column is stored");
12845        assert_eq!(near.len(), parts, "one row per part");
12846        assert_eq!(far.len(), parts);
12847        // The bytes are the same bytes the directory totals, which is the check that this is
12848        // reading the pages the file really holds rather than some other pages.
12849        let total = |stored: &[StoredPart]| stored.iter().map(|part| part.bytes).sum::<u64>();
12850        assert_eq!(total(&near), one.layout().columns[0].pages);
12851        assert_eq!(total(&far), other.layout().columns[0].pages);
12852        assert!(
12853            total(&near) * 2 < total(&far),
12854            "the sparse keys cost more, {} against {}",
12855            total(&far),
12856            total(&near)
12857        );
12858        // Every part accounted for, in order, with the row it starts at following the one before.
12859        for (at, part) in near.iter().enumerate() {
12860            assert_eq!(part.part, at);
12861            assert_eq!(part.row, at * per_part);
12862            assert_eq!(part.rows, per_part);
12863            let held = &ascending[at * per_part..(at + 1) * per_part];
12864            assert_eq!(part.low, Some(Value::BigInt(held[0])));
12865            assert_eq!(part.high, Some(Value::BigInt(held[per_part - 1])));
12866            assert_eq!(part.nulls, Some(0));
12867        }
12868        // And the encoding is a line of text that names what the encoder chose, which is the whole
12869        // point. Both are a cascade over deltas and the widths inside them are what differ.
12870        assert!(near[0].encoding.contains("DELTA"), "{}", near[0].encoding);
12871        assert!(far[0].encoding.contains("DELTA"), "{}", far[0].encoding);
12872        assert_ne!(near[0].encoding, far[0].encoding);
12873        fs::remove_file(near_path).expect("remove scratch file");
12874        fs::remove_file(far_path).expect("remove scratch file");
12875    }
12876
12877    /// A sieve bigger than the part it indexes is not written, and one smaller than it still is.
12878    ///
12879    /// Both columns hold values spread over the whole of `BIGINT`, so neither gets a bitmap and both
12880    /// reach the filter. They differ in what the part costs to read. `spread` is a thousand distinct
12881    /// numbers and packs to eight kilobytes, so a filter of about thirteen hundred bytes is a good
12882    /// trade. `repeated` is the same thousand rows over four numbers in runs and encodes to
12883    /// almost nothing, but the filter is sized for the rows rather than the values it turns out to
12884    /// hold, so it comes out larger than the data. Reading it to decide whether to read the part spends more than
12885    /// the part, every time, and that is the case this drops.
12886    #[test]
12887    fn a_sieve_larger_than_the_part_it_indexes_is_not_written() {
12888        let path = path("sieve-pays");
12889        let fields = vec![
12890            Field::required("spread", LogicalType::BigInt),
12891            Field::required("repeated", LogicalType::BigInt),
12892        ];
12893        let mut writer = Writer::create(&path, "hits", fields).expect("new file");
12894        let parts = 3;
12895        let per_part = 1024;
12896        for part in 0..parts {
12897            let base = (part * per_part) as i64;
12898            let spread: Vec<Value> =
12899                (0..per_part).map(|row| Value::BigInt(scattered(base + row as i64))).collect();
12900            let repeated: Vec<Value> =
12901                (0..per_part).map(|row| Value::BigInt(scattered((row / 256) as i64))).collect();
12902            let chunk = Chunk::new(vec![
12903                Vector::from_values(LogicalType::BigInt, &spread).expect("numbers"),
12904                Vector::from_values(LogicalType::BigInt, &repeated).expect("numbers"),
12905            ])
12906            .expect("two columns");
12907            writer.append(&chunk).expect("one part");
12908        }
12909        writer.finish().expect("commit");
12910
12911        let reader = Reader::open(&path).expect("reopen from disk");
12912        let layout = reader.layout();
12913        let spread = &layout.columns[0];
12914        let repeated = &layout.columns[1];
12915        assert!(spread.sieves > 0, "a column whose parts are worth a filter keeps one");
12916        assert_eq!(
12917            repeated.sieves, 0,
12918            "a column whose filter costs more than its parts keeps none"
12919        );
12920        // Per part this is the rule itself, so it holds over the column as well: a part without a
12921        // sieve adds to one side of this and to nothing on the other.
12922        for column in &layout.columns {
12923            assert!(
12924                column.sieves < column.pages,
12925                "{} spends {} on sieves over {} of data",
12926                column.name,
12927                column.sieves,
12928                column.pages
12929            );
12930        }
12931        // The filter that was kept still does what it is for.
12932        let absent = [Probe {
12933            column: 0,
12934            op: Op::Equal,
12935            value: Bound::Int(i128::from(scattered((parts * per_part) as i64 + 1))),
12936        }];
12937        assert!((0..parts).all(|part| reader.skips(part, &absent)), "no part holds it");
12938        fs::remove_file(path).expect("remove scratch file");
12939    }
12940
12941    /// A damaged sieve page is a part that gets read, not a query that fails.
12942    ///
12943    /// A sieve is an index over rows that are still there and still correct, so losing one costs
12944    /// time and costs no answers. That is the opposite of the membership index beside it, which is
12945    /// the only thing standing between a string page and a wrong answer.
12946    #[test]
12947    fn a_damaged_sieve_page_is_read_through_rather_than_refused() {
12948        let path = path("sieve-damaged");
12949        let mut writer =
12950            Writer::create(&path, "hits", vec![Field::required("id", LogicalType::BigInt)])
12951                .expect("new file");
12952        let rows = 128;
12953        let held: Vec<Value> = (0..rows).map(|row| Value::BigInt(scattered(row))).collect();
12954        let chunk =
12955            Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
12956                .expect("one column");
12957        writer.append(&chunk).expect("one part");
12958        writer.finish().expect("commit");
12959
12960        let page = Reader::open(&path).expect("reopen").table.stripes[0]
12961            .sieves
12962            .get(0)
12963            .expect("a sieve page");
12964        let mut file = OpenOptions::new().write(true).open(&path).expect("open the sieve page");
12965        file.seek(SeekFrom::Start(page.offset + u64::from(page.length) - 1)).expect("seek");
12966        file.write_all(&[0xff]).expect("damage one byte");
12967        drop(file);
12968
12969        let reader = Reader::open(&path).expect("reopen the damaged file");
12970        let absent =
12971            [Probe { column: 0, op: Op::Equal, value: Bound::Int(i128::from(scattered(99))) }];
12972        assert!(!reader.skips(0, &absent), "a sieve that cannot be read skips nothing");
12973        assert_eq!(
12974            reader.read(0, &[0]).expect("the rows are untouched").len(),
12975            usize::try_from(rows).expect("a small count")
12976        );
12977        fs::remove_file(path).expect("remove scratch file");
12978    }
12979
12980    /// Eight workers over one stripe read it once between them.
12981    ///
12982    /// This is the shape a scan actually has. Parts are handed out in order, so every worker on a
12983    /// column crosses into a stripe within a few parts of the others, and before [`Reader::held`]
12984    /// started sharing the read every one of them read the whole page. On the full ClickBench file
12985    /// that was a `MIN(EventDate), MAX(EventDate)` moving 3.2 GB off the disk to look at 400 MB of
12986    /// column, which is most of what a first touch costs.
12987    ///
12988    /// The workers that lose the race still answer, out of the part reads they do instead, which is
12989    /// what the values below are checking.
12990    #[test]
12991    fn workers_that_want_the_same_stripe_read_it_once() {
12992        let path = path("single-flight");
12993        let mut writer =
12994            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
12995                .expect("new file");
12996        for part in 0..STRIPE_PARTS {
12997            let id = part as i32;
12998            let chunk = Chunk::new(vec![
12999                Vector::from_values(
13000                    LogicalType::Integer,
13001                    &[Value::Integer(id), Value::Integer(-id)],
13002                )
13003                .expect("integers"),
13004            ])
13005            .expect("matching rows");
13006            writer.append(&chunk).expect("one part");
13007        }
13008        writer.finish().expect("commit");
13009
13010        let reader = Reader::open(&path).expect("reopen from disk");
13011        assert_eq!(reader.table().stripes().len(), 1, "one stripe is the point of the test");
13012        let barrier = std::sync::Barrier::new(8);
13013        std::thread::scope(|scope| {
13014            for worker in 0..8 {
13015                let reader = &reader;
13016                let barrier = &barrier;
13017                scope.spawn(move || {
13018                    barrier.wait();
13019                    for part in (worker..STRIPE_PARTS).step_by(8) {
13020                        let chunk = reader.read(part, &[0]).expect("a whole page read");
13021                        assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
13022                        assert_eq!(chunk.value_at(1, 0), Value::Integer(-(part as i32)));
13023                    }
13024                });
13025            }
13026        });
13027        assert_eq!(reader.pages.load(Atomic::Relaxed), 1, "one stripe, one page read, whoever won");
13028        fs::remove_file(path).expect("remove scratch file");
13029    }
13030
13031    /// Opening a file reads the header and the directory, and nothing that depends on the rows.
13032    ///
13033    /// `spec/stats/04-in-memory.md` section 4.2. There are no statistics in the file yet, so this
13034    /// holds today by not having anything to load, and that is exactly why it is worth pinning now.
13035    /// The change that breaks it is the reasonable looking one: summaries are a few hundred bytes,
13036    /// the next query will want them, so read them on the way past. A process that opened the
13037    /// database to run one trivial query pays for all of it and gets nothing.
13038    ///
13039    /// Two files of the same shape and a thousand times the rows in one of them, opened, and the
13040    /// two openings cost the same. The stripe count is held equal so that the directory is the same
13041    /// size in both, which leaves the rows as the only thing that changed. Anything read out of the
13042    /// data would show up here.
13043    #[test]
13044    fn opening_costs_the_same_over_a_thousand_times_the_rows() {
13045        let opened = |label: &str, rows_per_part: i32| {
13046            let path = path(label);
13047            let mut writer =
13048                Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
13049                    .expect("new file");
13050            for part in 0..STRIPE_PARTS * 3 {
13051                // Scrambled rather than sequential, so that the fat file is actually fatter. A run
13052                // of consecutive integers encodes to almost nothing and would leave the two files
13053                // the same size, which would make this test pass for the wrong reason.
13054                let values = (0..rows_per_part)
13055                    .map(|row| {
13056                        Value::Integer((part as i32 * rows_per_part + row).wrapping_mul(2_654_435))
13057                    })
13058                    .collect::<Vec<_>>();
13059                let chunk = Chunk::new(vec![
13060                    Vector::from_values(LogicalType::Integer, &values).expect("integers"),
13061                ])
13062                .expect("matching rows");
13063                writer.append(&chunk).expect("one part");
13064            }
13065            writer.finish().expect("commit");
13066            let reader = Reader::open(&path).expect("reopen from disk");
13067            let size = fs::metadata(&path).expect("the file is there").len();
13068            let out = (reader.reads(), reader.table().stripes().len(), size);
13069            fs::remove_file(path).expect("remove scratch file");
13070            out
13071        };
13072
13073        let (thin, thin_stripes, thin_size) = opened("open-thin", 1);
13074        let (fat, fat_stripes, fat_size) = opened("open-fat", 1000);
13075        assert_eq!(
13076            thin_stripes, fat_stripes,
13077            "the same stripe count is what makes this a fair ask"
13078        );
13079        assert!(
13080            fat_size > thin_size * 50,
13081            "the fat file has to actually be larger, and it is {fat_size} against {thin_size}"
13082        );
13083
13084        assert_eq!(thin.opening.reads, fat.opening.reads, "the same reads either way");
13085        assert_eq!(thin.pages, 0, "opening read a page");
13086        assert_eq!(fat.pages, 0, "opening read a page");
13087        assert_eq!(thin.indexes, 0, "opening read an index");
13088        assert_eq!(fat.indexes, 0, "opening read an index");
13089        // Not exactly equal, because a directory holds offsets and a larger file has larger ones,
13090        // and a handful of bytes of varint is not somebody loading statistics. A factor is.
13091        assert!(
13092            fat.opening.bytes < thin.opening.bytes * 2,
13093            "opening the thin file read {} bytes and the fat one read {}",
13094            thin.opening.bytes,
13095            fat.opening.bytes
13096        );
13097    }
13098
13099    /// The reads a file costs to open are fixed by its shape and not by what ran before.
13100    ///
13101    /// `spec/stats/04-in-memory.md` section 4.3, which is the rule that keeps a plan reproducible:
13102    /// the plan is a function of the data, the generation and the settings, and never of what
13103    /// happened to be in cache. Opening the same file twice in the same process has to cost the
13104    /// same, because a second open that read less would be an open that was about to plan
13105    /// differently.
13106    #[test]
13107    fn two_opens_of_one_file_cost_the_same_and_the_second_is_not_cheaper() {
13108        let path = path("open-twice");
13109        let mut writer =
13110            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
13111                .expect("new file");
13112        for part in 0..STRIPE_PARTS * 3 {
13113            let chunk = Chunk::new(vec![
13114                Vector::from_values(LogicalType::Integer, &[Value::Integer(part as i32)])
13115                    .expect("integers"),
13116            ])
13117            .expect("matching rows");
13118            writer.append(&chunk).expect("one part");
13119        }
13120        writer.finish().expect("commit");
13121
13122        let first = Reader::open(&path).expect("open");
13123        // A whole scan in between, so the operating system's page cache is as warm as it gets and
13124        // anything that consulted it would show up in the second open.
13125        for part in 0..first.parts() {
13126            first.read(part, &[0]).expect("a part");
13127        }
13128        assert!(first.reads().pages > 0, "the scan has to have read something");
13129        let second = Reader::open(&path).expect("open again");
13130
13131        assert_eq!(first.reads().opening, second.reads().opening);
13132        assert_eq!(
13133            second.reads().pages,
13134            0,
13135            "the second open read a page off the back of the first"
13136        );
13137        assert_eq!(second.reads().indexes, 0, "the second open read an index it inherited");
13138        fs::remove_file(path).expect("remove scratch file");
13139    }
13140
13141    /// A scan reads a stripe's index once for the whole scan, not once per part that misses.
13142    ///
13143    /// The page cache holds four stripes and an index used to ride inside it, so a table with more
13144    /// stripes than that read the index again every time a stripe came back around. The index is a
13145    /// few hundred bytes and the page is a quarter of a megabyte, which is why they are now under
13146    /// different budgets. This is the test that keeps them there, since the saving is small enough
13147    /// that nothing in a benchmark would notice it going away again.
13148    #[test]
13149    fn an_index_is_read_once_per_stripe_however_often_the_page_is_evicted() {
13150        let path = path("index-cache");
13151        let mut writer =
13152            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
13153                .expect("new file");
13154        let parts = STRIPE_PARTS * (CACHED_STRIPES_PER_COLUMN + 2);
13155        for part in 0..parts {
13156            let id = part as i32;
13157            let chunk = Chunk::new(vec![
13158                Vector::from_values(LogicalType::Integer, &[Value::Integer(id)]).expect("integers"),
13159            ])
13160            .expect("matching rows");
13161            writer.append(&chunk).expect("one part");
13162        }
13163        writer.finish().expect("commit");
13164
13165        let reader = Reader::open(&path).expect("reopen from disk");
13166        let stripes = reader.table().stripes().len();
13167        assert!(stripes > CACHED_STRIPES_PER_COLUMN, "the page cache has to be too small for this");
13168        // Twice over, so that the second pass finds every page evicted and every index kept.
13169        for _ in 0..2 {
13170            for part in 0..parts {
13171                let chunk = reader.read(part, &[0]).expect("a part");
13172                assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
13173            }
13174        }
13175        assert_eq!(reader.indexes.load(Atomic::Relaxed), stripes, "one index read per stripe");
13176        assert!(
13177            reader.pages.load(Atomic::Relaxed) > stripes,
13178            "the pages are the ones that get read again, which is what makes the index count mean \
13179             something"
13180        );
13181        fs::remove_file(path).expect("remove scratch file");
13182    }
13183
13184    /// A page stays in memory from one scan to the next while the pool has room for it, and a
13185    /// table that is being read takes room from one that is not, down to the floor and no further.
13186    ///
13187    /// This is what the pool is for. Each reader lives as long as its database, so a second query
13188    /// over the same table should find every page it read the first time, and before the pool it
13189    /// found four stripes a column and read the rest off the file again.
13190    #[test]
13191    fn a_pool_keeps_pages_between_scans_and_gives_them_to_the_table_being_read() {
13192        let path = path("page-pool");
13193        let parts = STRIPE_PARTS * (CACHED_STRIPES_PER_COLUMN + 2);
13194        let fields = || vec![Field::required("id", LogicalType::Integer)];
13195        let mut writer = Writer::create(&path, "a", fields()).expect("new file");
13196        for table in ["a", "b"] {
13197            if table == "b" {
13198                writer = writer.next("b".to_string(), fields()).expect("a second table");
13199            }
13200            for part in 0..parts {
13201                let chunk = Chunk::new(vec![
13202                    Vector::from_values(LogicalType::Integer, &[Value::Integer(part as i32)])
13203                        .expect("integers"),
13204                ])
13205                .expect("matching rows");
13206                writer.append(&chunk).expect("one part");
13207            }
13208        }
13209        writer.finish().expect("commit");
13210
13211        let pool = PagePool::new(usize::MAX);
13212        let catalog = Catalog::open_in(&path, &pool).expect("the file opens");
13213        let (a, b) = (catalog.table("a").expect("a"), catalog.table("b").expect("b"));
13214        let stripes = a.table().stripes().len();
13215        assert!(stripes > CACHED_STRIPES_PER_COLUMN, "the floor has to be smaller than a table");
13216        let scan = |reader: &Reader| {
13217            for part in 0..parts {
13218                let chunk = reader.read(part, &[0]).expect("a part");
13219                assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
13220            }
13221        };
13222        scan(&a);
13223        scan(&a);
13224        assert_eq!(a.pages.load(Atomic::Relaxed), stripes, "the second scan reads nothing");
13225        let one = pool.bytes();
13226        assert!(one > 0, "the pool counts what the reader holds");
13227
13228        // Room for one table. Reading the other takes the first one's pages down to its floor.
13229        pool.budget.store(one, Atomic::Relaxed);
13230        scan(&b);
13231        assert_eq!(b.pages.load(Atomic::Relaxed), stripes, "a page is never let go while in use");
13232        assert_eq!(a.cache.held[0].load(Atomic::Relaxed), CACHED_STRIPES_PER_COLUMN);
13233        let held = a.cache.columns[0].lock().expect("the column").pages.iter().flatten().count();
13234        assert_eq!(held, CACHED_STRIPES_PER_COLUMN, "the count and the slots agree");
13235
13236        // A reader that goes takes its pages out of the count with it.
13237        drop((a, b, catalog));
13238        let c = Catalog::open_in(&path, &pool).expect("again").table("a").expect("a");
13239        scan(&c);
13240        assert!(pool.bytes() <= one, "only what the live reader holds is counted");
13241        fs::remove_file(path).expect("remove scratch file");
13242    }
13243
13244    /// A worker per stripe reads its stripe once, once the cache has been told how many there are.
13245    ///
13246    /// This is the shape a scan has when it hands out a whole stripe per morsel rather than a part.
13247    /// Nobody races for a page any more, but every worker holds a different one for the length of a
13248    /// stripe, so a cache that keeps four pages while eight workers are in eight stripes evicts
13249    /// every one of them before its owner has finished with it, and the owner reads a quarter of a
13250    /// megabyte again for the next part. The barrier is what makes that certain rather than likely:
13251    /// without it a worker can run a whole stripe before the next one starts and never collide.
13252    #[test]
13253    fn a_worker_per_stripe_reads_its_page_once_when_the_cache_was_told_to_expect_it() {
13254        let workers = CACHED_STRIPES_PER_COLUMN + 4;
13255        let path = path("stripe-per-worker");
13256        let mut writer =
13257            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
13258                .expect("new file");
13259        for part in 0..STRIPE_PARTS * workers {
13260            let chunk = Chunk::new(vec![
13261                Vector::from_values(LogicalType::Integer, &[Value::Integer(part as i32)])
13262                    .expect("integers"),
13263            ])
13264            .expect("matching rows");
13265            writer.append(&chunk).expect("one part");
13266        }
13267        writer.finish().expect("commit");
13268
13269        let read = |told: bool| {
13270            let reader = Reader::open(&path).expect("reopen from disk");
13271            assert_eq!(reader.table().stripes().len(), workers, "a stripe per worker");
13272            if told {
13273                reader.keep_stripes(workers);
13274            }
13275            let barrier = std::sync::Barrier::new(workers);
13276            std::thread::scope(|scope| {
13277                for (worker, run) in reader.stripe_parts().into_iter().enumerate() {
13278                    let reader = &reader;
13279                    let barrier = &barrier;
13280                    scope.spawn(move || {
13281                        for part in run {
13282                            barrier.wait();
13283                            let chunk = reader.read(part, &[0]).expect("a part of my own stripe");
13284                            assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
13285                        }
13286                        assert!(worker < workers);
13287                    });
13288                }
13289            });
13290            reader.pages.load(Atomic::Relaxed)
13291        };
13292
13293        assert_eq!(read(true), workers, "one page read per stripe and no more");
13294        assert!(read(false) > workers, "a cache that small is read again on every part");
13295        fs::remove_file(path).expect("remove scratch file");
13296    }
13297
13298    /// A damaged index page is caught before anything decodes a part out of it.
13299    ///
13300    /// The index is the one structure a reader trusts to find bytes with, so it carries a checksum
13301    /// per column section rather than one for the page, and this is what says that check runs.
13302    #[test]
13303    fn a_damaged_index_page_is_an_error() {
13304        let path = path("damaged-index");
13305        let mut writer =
13306            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
13307                .expect("new file");
13308        writer.append(&sample_ids()).expect("first part");
13309        writer.append(&sample_ids()).expect("second part");
13310        writer.finish().expect("commit");
13311
13312        let reader = Reader::open(&path).expect("valid directory");
13313        let index = reader.table.stripes[0].index;
13314        let mut byte = [0; 1];
13315        read_at(&reader.file, index.offset, &mut byte).expect("the first part length");
13316        let mut file = OpenOptions::new().write(true).open(&path).expect("open index page");
13317        file.seek(SeekFrom::Start(index.offset)).expect("index start");
13318        file.write_all(&[!byte[0]]).expect("damage the first part length");
13319        let error = reader.read(1, &[0]).expect_err("a damaged index must not be used");
13320        assert!(error.message().contains("index page section checksum differs"), "{error}");
13321        fs::remove_file(path).expect("remove scratch file");
13322    }
13323
13324    /// Every integer width the format knows about, written and read back.
13325    ///
13326    /// The unsigned ones are the reason ClickBench can be stored at all: `hits` types `EventDate`
13327    /// as `USMALLINT`, and one unsupported column meant the whole table was refused. The extremes
13328    /// are in here on purpose, because a width that round trips through the wrong signedness only
13329    /// goes wrong at the end of its range.
13330    #[test]
13331    fn every_integer_width_round_trips_through_a_page() {
13332        let path = path("integer-widths");
13333        let columns = [
13334            (LogicalType::TinyInt, vec![Value::TinyInt(i8::MIN), Value::TinyInt(i8::MAX)]),
13335            (LogicalType::UTinyInt, vec![Value::UTinyInt(0), Value::UTinyInt(u8::MAX)]),
13336            (LogicalType::SmallInt, vec![Value::SmallInt(i16::MIN), Value::SmallInt(i16::MAX)]),
13337            (LogicalType::USmallInt, vec![Value::USmallInt(0), Value::USmallInt(u16::MAX)]),
13338            (LogicalType::Integer, vec![Value::Integer(i32::MIN), Value::Integer(i32::MAX)]),
13339            (LogicalType::UInteger, vec![Value::UInteger(0), Value::UInteger(u32::MAX)]),
13340            (LogicalType::BigInt, vec![Value::BigInt(i64::MIN), Value::BigInt(i64::MAX)]),
13341            (LogicalType::UBigInt, vec![Value::UBigInt(0), Value::UBigInt(u64::MAX)]),
13342        ];
13343        let fields = columns
13344            .iter()
13345            .enumerate()
13346            .map(|(at, (ty, _))| Field::required(format!("c{at}"), ty.clone()))
13347            .collect::<Vec<_>>();
13348        let vectors = columns
13349            .iter()
13350            .map(|(ty, values)| Vector::from_values(ty.clone(), values).expect("a vector"))
13351            .collect::<Vec<_>>();
13352        let mut writer = Writer::create(&path, "widths", fields).expect("new file");
13353        writer.append(&Chunk::new(vectors).expect("matching rows")).expect("one stripe");
13354        writer.finish().expect("commit");
13355
13356        let reader = Reader::open(&path).expect("reopen from disk");
13357        let wanted = (0..columns.len()).collect::<Vec<_>>();
13358        let read = reader.read(0, &wanted).expect("every column");
13359        assert_eq!(read.len(), 2);
13360        // row at a time: each column has its own type and its own pair of extremes.
13361        for (at, (ty, values)) in columns.iter().enumerate() {
13362            assert_eq!(read.value_at(0, at), values[0], "the low end of {ty}");
13363            assert_eq!(read.value_at(1, at), values[1], "the high end of {ty}");
13364        }
13365        fs::remove_file(path).expect("remove scratch file");
13366    }
13367
13368    /// The rest of the fixed width types, and the byte strings, written and read back.
13369    ///
13370    /// The extremes again, and for a float that means more than the ends of the range. Negative
13371    /// zero and a NaN are the two values that go through an encoder unnoticed and come back
13372    /// different, so they are here on purpose, and the NaN is compared by its bits rather than by
13373    /// `==`, which a NaN fails against itself.
13374    ///
13375    /// A blob is here beside them because it is the same round trip asked of bytes that are not
13376    /// text. The value in it is not UTF-8, so a path that reads a payload as a string on the way
13377    /// past turns this test red rather than turning a user's column into nulls.
13378    #[test]
13379    fn every_other_type_the_format_knows_round_trips_through_a_page() {
13380        let path = path("other-types");
13381        let columns = [
13382            (LogicalType::Float, vec![Value::Float(f32::MIN), Value::Float(-0.0)]),
13383            (LogicalType::Double, vec![Value::Double(f64::MIN), Value::Double(f64::MAX)]),
13384            (LogicalType::HugeInt, vec![Value::HugeInt(i128::MIN), Value::HugeInt(i128::MAX)]),
13385            (LogicalType::UHugeInt, vec![Value::UHugeInt(0), Value::UHugeInt(u128::MAX)]),
13386            (LogicalType::Time, vec![Value::Time(0), Value::Time(86_399_999_999)]),
13387            (LogicalType::TimeTz, vec![Value::TimeTz(-50_400_000_000), Value::TimeTz(0)]),
13388            (
13389                LogicalType::TimestampTz,
13390                vec![Value::TimestampTz(i64::MIN + 1), Value::TimestampTz(i64::MAX)],
13391            ),
13392            (
13393                LogicalType::Interval,
13394                vec![
13395                    Value::Interval { months: i32::MIN, days: i32::MAX, micros: i64::MIN },
13396                    Value::Interval { months: 13, days: -1, micros: 1 },
13397                ],
13398            ),
13399            (
13400                LogicalType::Blob,
13401                vec![Value::Blob(vec![0, 0xff, 0x80, 0xfe]), Value::Blob(Vec::new())],
13402            ),
13403        ];
13404        let fields = columns
13405            .iter()
13406            .enumerate()
13407            .map(|(at, (ty, _))| Field::required(format!("c{at}"), ty.clone()))
13408            .collect::<Vec<_>>();
13409        let vectors = columns
13410            .iter()
13411            .map(|(ty, values)| Vector::from_values(ty.clone(), values).expect("a vector"))
13412            .collect::<Vec<_>>();
13413        let mut writer = Writer::create(&path, "others", fields).expect("new file");
13414        writer.append(&Chunk::new(vectors).expect("matching rows")).expect("one stripe");
13415        writer.finish().expect("commit");
13416
13417        let reader = Reader::open(&path).expect("reopen from disk");
13418        let wanted = (0..columns.len()).collect::<Vec<_>>();
13419        let read = reader.read(0, &wanted).expect("every column");
13420        assert_eq!(read.len(), 2);
13421        for (at, (ty, values)) in columns.iter().enumerate() {
13422            assert_eq!(read.value_at(0, at), values[0], "the low end of {ty}");
13423            assert_eq!(read.value_at(1, at), values[1], "the high end of {ty}");
13424        }
13425        // A float keeps its sign through a zero, which `==` says nothing about because negative
13426        // zero and zero compare equal.
13427        let Value::Float(zero) = read.value_at(1, 0) else { panic!("a float stays a float") };
13428        assert!(zero.is_sign_negative(), "a negative zero came back as {zero}");
13429
13430        fs::remove_file(path).expect("remove scratch file");
13431    }
13432
13433    /// A NaN is still a NaN after a trip through a page.
13434    ///
13435    /// Apart from the other floats because it cannot be asserted the same way. A NaN is not equal
13436    /// to itself, so a comparison against the value that was written passes for every NaN and for
13437    /// nothing else, which is the one assertion that would not catch a page that lost it.
13438    #[test]
13439    fn a_nan_survives_being_written_down() {
13440        let path = path("nan");
13441        let nan = Vector::from_values(LogicalType::Double, &[Value::Double(f64::NAN)])
13442            .expect("a NaN vector");
13443        let mut writer =
13444            Writer::create(&path, "nan", vec![Field::required("d", LogicalType::Double)])
13445                .expect("new file");
13446        writer.append(&Chunk::new(vec![nan]).expect("one column")).expect("one stripe");
13447        writer.finish().expect("commit");
13448        let read = Reader::open(&path).expect("reopen").read(0, &[0]).expect("the column");
13449        let Value::Double(back) = read.value_at(0, 0) else { panic!("a double stays a double") };
13450        assert!(back.is_nan(), "a NaN came back as {back}");
13451        fs::remove_file(path).expect("remove scratch file");
13452    }
13453
13454    /// A uuid and a bit string, which have no `Value` arm of their own and are checked as bits.
13455    ///
13456    /// A uuid is the 128 bit lane and a bit string is bytes, and neither of them reads back as
13457    /// anything in `Value` today, so asking for a value here would compare two nulls and pass
13458    /// whatever the file held. The data underneath is what the storage promise is about, so that is
13459    /// what this reads.
13460    #[test]
13461    fn a_uuid_and_a_bit_string_come_back_as_the_bits_that_went_in() {
13462        let path = path("uuid-and-bit");
13463        let uuids = vec![0_i128, i128::MIN, -1];
13464        let mut bits = StringColumn::new();
13465        for value in [&b"\x02\xff"[..], &b""[..], &b"\x00\x01\x02\x03\x04\x05"[..]] {
13466            bits.push_bytes(value);
13467        }
13468        let expected = bits.clone();
13469        let fields =
13470            vec![Field::required("u", LogicalType::Uuid), Field::required("b", LogicalType::Bit)];
13471        let vectors = vec![
13472            Vector::flat(LogicalType::Uuid, Data::Int128(uuids.clone().into())).expect("uuids"),
13473            Vector::flat(LogicalType::Bit, Data::Varlen(bits)).expect("bit strings"),
13474        ];
13475        let mut writer = Writer::create(&path, "ids", fields).expect("new file");
13476        writer.append(&Chunk::new(vectors).expect("matching rows")).expect("one stripe");
13477        writer.finish().expect("commit");
13478
13479        let reader = Reader::open(&path).expect("reopen from disk");
13480        let read = reader.read(0, &[0, 1]).expect("both columns").flatten().expect("flat");
13481        let Some(Data::Int128(back)) = read.column(0).expect("the uuids").data() else {
13482            panic!("a uuid column is the 128 bit lane")
13483        };
13484        assert_eq!(back.as_slice(), uuids.as_slice());
13485        let Some(Data::Varlen(back)) = read.column(1).expect("the bits").data() else {
13486            panic!("a bit column is bytes")
13487        };
13488        for row in 0..expected.len() {
13489            assert_eq!(back.bytes(row), expected.bytes(row), "row {row} of the bit column");
13490        }
13491        fs::remove_file(path).expect("remove scratch file");
13492    }
13493
13494    /// Counting a run at once has to leave the candidate table exactly where counting its rows one
13495    /// at a time would, including once the table is full and a run is turned away row by row.
13496    #[test]
13497    fn a_run_counted_at_once_leaves_the_candidates_a_row_at_a_time_would() {
13498        let mut rows: Vec<Option<u64>> = Vec::new();
13499        let mut state = 0x2545_f491_4f6c_dd1d_u64;
13500        for index in 0..400_000_u64 {
13501            state ^= state << 13;
13502            state ^= state >> 7;
13503            state ^= state << 17;
13504            let times = 1 + (state % 7) as usize;
13505            let bits = match state % 11 {
13506                0 => None,
13507                1..=3 => Some(state % 16),
13508                _ => Some(index.wrapping_mul(0x9e37_79b9_7f4a_7c15)),
13509            };
13510            rows.extend(std::iter::repeat_n(bits, times));
13511        }
13512        let mut by_row = Candidates::default();
13513        for &bits in &rows {
13514            by_row.add(bits, 1);
13515        }
13516        let mut by_run = Candidates::default();
13517        let mut run = Run::default();
13518        let mut runs = 0_usize;
13519        for &bits in &rows {
13520            if let Some((bits, times)) = run.push(bits) {
13521                by_run.add(bits, times);
13522                runs += 1;
13523            }
13524        }
13525        if let Some((bits, times)) = run.take() {
13526            by_run.add(bits, times);
13527        }
13528        assert!(runs < rows.len() / 2, "the rows came in runs");
13529        assert!(by_row.decrements > 0, "the table filled and turned values away");
13530        assert_eq!(by_run.counts, by_row.counts);
13531        assert_eq!(by_run.nulls, by_row.nulls);
13532        assert_eq!(by_run.decrements, by_row.decrements);
13533    }
13534
13535    #[test]
13536    fn numeric_frequency_candidates_keep_bounded_row_ordinals() {
13537        let path = path("frequency-ordinals");
13538        let mut writer =
13539            Writer::create(&path, "items", vec![Field::required("id", LogicalType::BigInt)])
13540                .expect("new file");
13541        let mut values = Vec::new();
13542        for leader in 0..10_i64 {
13543            values.extend(std::iter::repeat_n(leader, 100));
13544        }
13545        values.extend(1_000_i64..41_000);
13546        for part in values.chunks(1_024) {
13547            let vector = Vector::flat(LogicalType::BigInt, Data::Int64(part.to_vec().into()))
13548                .expect("big integers");
13549            writer.append(&Chunk::new(vec![vector]).expect("one column")).expect("one stripe");
13550        }
13551        writer.finish().expect("commit");
13552
13553        let reader = Reader::open(&path).expect("reopen from disk");
13554        let occurrences =
13555            reader.frequency_occurrences(0).expect("valid metadata").expect("bounded ordinals");
13556        assert!(occurrences.omitted_max < 100);
13557        assert!(occurrences.ordinals.len() <= FREQUENCY_ORDINALS);
13558        assert_eq!(occurrences.anchor_indices.len(), occurrences.ordinals.len());
13559        assert!(occurrences.ordinals.windows(2).all(|pair| pair[0] < pair[1]));
13560        assert_eq!(&occurrences.ordinals[..1_000], &(0_u64..1_000).collect::<Vec<_>>());
13561        assert_eq!(
13562            &occurrences.anchor_indices[..1_000]
13563                .iter()
13564                .map(|&entry| occurrences.anchors[entry as usize].clone())
13565                .collect::<Vec<_>>(),
13566            &(0_i64..10)
13567                .flat_map(|leader| std::iter::repeat_n(Value::BigInt(leader), 100))
13568                .collect::<Vec<_>>()
13569        );
13570        fs::remove_file(path).expect("remove scratch file");
13571    }
13572
13573    #[test]
13574    fn numeric_frequencies_count_nulls_and_values_past_the_top_of_bigint() {
13575        // Ten leaders, then more unique values than the candidate table holds, so the first pass
13576        // has to decrement and the counts come from the recount. The unsigned leaders sit above
13577        // `i64::MAX`, where reading the bits as signed would give a different value, and the signed
13578        // ones are negative, where reading them as unsigned would.
13579        let path = path("frequency-bits");
13580        let mut writer = Writer::create(
13581            &path,
13582            "items",
13583            vec![Field::new("u", LogicalType::UBigInt), Field::new("s", LogicalType::BigInt)],
13584        )
13585        .expect("new file");
13586        let mut rows = Vec::new();
13587        let mut leaders = Vec::new();
13588        for leader in 0..10_u64 {
13589            let count = 300 - leader * 10;
13590            let (unsigned, signed) = if leader == 0 {
13591                (Value::Null, Value::Null)
13592            } else {
13593                (Value::UBigInt(u64::MAX - leader), Value::BigInt(-(leader as i64)))
13594            };
13595            rows.extend(std::iter::repeat_n((unsigned.clone(), signed.clone()), count as usize));
13596            leaders.push(((unsigned, count), (signed, count)));
13597        }
13598        rows.extend((1_000..41_000_u64).map(|id| (Value::UBigInt(id), Value::BigInt(id as i64))));
13599        for part in rows.chunks(1_024) {
13600            let unsigned = part.iter().map(|(value, _)| value.clone()).collect::<Vec<_>>();
13601            let signed = part.iter().map(|(_, value)| value.clone()).collect::<Vec<_>>();
13602            let chunk = Chunk::new(vec![
13603                Vector::from_values(LogicalType::UBigInt, &unsigned).expect("unsigned"),
13604                Vector::from_values(LogicalType::BigInt, &signed).expect("signed"),
13605            ])
13606            .expect("matching columns");
13607            writer.append(&chunk).expect("rows");
13608        }
13609        writer.finish().expect("commit");
13610
13611        let reader = Reader::open(&path).expect("reopen from disk");
13612        for column in 0..2 {
13613            let prefix =
13614                reader.frequency_prefix(column).expect("valid metadata").expect("a synopsis");
13615            let wanted = leaders
13616                .iter()
13617                .map(|(unsigned, signed)| if column == 0 { unsigned } else { signed })
13618                .cloned()
13619                .collect::<Vec<_>>();
13620            assert_eq!(&prefix.entries[..10], &wanted[..], "column {column}");
13621            assert!(prefix.omitted_max < 210, "column {column}");
13622            assert_eq!(
13623                reader.distinct_values(column).expect("valid metadata"),
13624                Some(9 + 40_000),
13625                "column {column}"
13626            );
13627        }
13628        fs::remove_file(path).expect("remove scratch file");
13629    }
13630
13631    #[test]
13632    fn narrow_nonzero_count_matches_the_full_reader_across_stripes() {
13633        let path = path("quick-nonzero");
13634        let mut writer = Writer::create(
13635            &path,
13636            "items",
13637            vec![Field::new("label", LogicalType::Varchar), Field::new("id", LogicalType::Integer)],
13638        )
13639        .expect("create");
13640        for ids in [
13641            &[Value::Integer(0), Value::Null, Value::Integer(3)][..],
13642            &[Value::Integer(0), Value::Integer(7), Value::Null][..],
13643        ] {
13644            let labels = vec![Value::Varchar("same".into()); ids.len()];
13645            writer
13646                .append(
13647                    &Chunk::new(vec![
13648                        Vector::from_values(LogicalType::Varchar, &labels).expect("labels"),
13649                        Vector::from_values(LogicalType::Integer, ids).expect("ids"),
13650                    ])
13651                    .expect("chunk"),
13652                )
13653                .expect("append");
13654        }
13655        writer.finish().expect("finish");
13656        let catalog = Catalog::open(&path).expect("catalog");
13657        assert_eq!(catalog.entries[0].nonzero, vec![None, Some(2)]);
13658        assert_eq!(catalog.entries[0].aggregates, vec![None, Some((10, 4))]);
13659        assert_eq!(catalog.entries[0].distincts, vec![Some(1), Some(3)]);
13660        let frequencies =
13661            catalog.exact_numeric_frequencies("items", 1).expect("frequencies").expect("complete");
13662        assert_eq!(frequencies.len(), 4);
13663        for pair in [(Some(0), 2), (Some(3), 1), (Some(7), 1), (None, 2)] {
13664            assert!(frequencies.contains(&pair), "missing {pair:?}");
13665        }
13666        assert_eq!(catalog.distinct_count("items", 1).expect("distinct count"), Some(3));
13667        assert_eq!(
13668            catalog.integer_extremes("items", 1).expect("extremes"),
13669            Some(IntegerExtremes::Values { low: 0, high: 7 })
13670        );
13671        assert_eq!(
13672            catalog.aggregate_sums("items", &[1]).expect("catalog sums"),
13673            Some(CertifiedSums { columns: vec![(10, 4)], rows: 6 })
13674        );
13675        assert_eq!(catalog.nonzero_count("items", 1).expect("quick count"), Some(2));
13676        assert_eq!(
13677            reader_nonzero_counts(&catalog.table("items").expect("reader")).expect("counts"),
13678            vec![None, Some(2)]
13679        );
13680        Writer::certify_counts(&path).expect("recertify");
13681        assert_eq!(
13682            Catalog::open(&path).expect("reopen").nonzero_count("items", 1).expect("count"),
13683            Some(2)
13684        );
13685        assert_eq!(
13686            Catalog::open(&path).expect("reopen").aggregate_sums("items", &[1]).expect("sums"),
13687            Some(CertifiedSums { columns: vec![(10, 4)], rows: 6 })
13688        );
13689        assert_eq!(
13690            Catalog::open(&path).expect("reopen").distinct_count("items", 1).expect("distinct"),
13691            Some(3)
13692        );
13693        assert_eq!(
13694            Catalog::open(&path).expect("reopen").integer_extremes("items", 1).expect("ends"),
13695            Some(IntegerExtremes::Values { low: 0, high: 7 })
13696        );
13697        assert_eq!(
13698            Catalog::open(&path)
13699                .expect("reopen")
13700                .exact_numeric_frequencies("items", 1)
13701                .expect("frequencies"),
13702            Some(frequencies)
13703        );
13704        assert_eq!(catalog.table("items").expect("reader").null_count(1).expect("nulls"), 2);
13705        fs::remove_file(path).expect("remove scratch file");
13706    }
13707
13708    #[test]
13709    fn numeric_string_pair_leaders_are_certified_in_the_directory() {
13710        let path = path("pair-frequencies");
13711        let mut pairs = Vec::new();
13712        pairs.extend(std::iter::repeat_n((1_i64, "alpha".to_string()), 100));
13713        pairs.extend(std::iter::repeat_n((1_i64, "beta".to_string()), 50));
13714        pairs.extend(std::iter::repeat_n((2_i64, "gamma".to_string()), 40));
13715        pairs.extend((1_000_i64..1_600).map(|id| (id, format!("tail {id}"))));
13716        let mut writer = Writer::create(
13717            &path,
13718            "items",
13719            vec![
13720                Field::required("id", LogicalType::BigInt),
13721                Field::required("phrase", LogicalType::Varchar),
13722            ],
13723        )
13724        .expect("new file");
13725        for part in pairs.chunks(1_024) {
13726            let ids = part.iter().map(|(id, _)| Value::BigInt(*id)).collect::<Vec<_>>();
13727            let phrases =
13728                part.iter().map(|(_, phrase)| Value::Varchar(phrase.clone())).collect::<Vec<_>>();
13729            writer
13730                .append(
13731                    &Chunk::new(vec![
13732                        Vector::from_values(LogicalType::BigInt, &ids).expect("ids"),
13733                        Vector::from_values(LogicalType::Varchar, &phrases).expect("phrases"),
13734                    ])
13735                    .expect("matching columns"),
13736                )
13737                .expect("rows");
13738        }
13739        writer.finish().expect("commit");
13740
13741        let reader = Reader::open(&path).expect("reopen from disk");
13742        assert!(
13743            reader.table.pair_frequencies.is_empty(),
13744            "no query-specific pair result is stored"
13745        );
13746        fs::remove_file(path).expect("remove scratch file");
13747    }
13748
13749    /// The bug this is here for cost a 43 GB ClickBench table and an hour of reloading it. The
13750    /// format went from 11 to 12, every binary built after that said "magic or major version is
13751    /// unsupported" about the file, and there was no way to tell from the message whether the path
13752    /// was wrong, the file was truncated, or it was ours and simply older. The number this build
13753    /// wants is the whole answer and it was the one thing the message did not carry.
13754    #[test]
13755    fn a_file_from_another_format_says_which_format_it_is() {
13756        let older = path("older-format");
13757        let mut writer =
13758            Writer::create(&older, "items", vec![Field::new("id", LogicalType::Integer)])
13759                .expect("new file");
13760        let chunk = Chunk::new(vec![
13761            Vector::flat(LogicalType::Integer, Data::Int32(vec![1, 2, 3].into()))
13762                .expect("integers"),
13763        ])
13764        .expect("chunk");
13765        writer.append(&chunk).expect("page written");
13766        writer.finish().expect("commit");
13767
13768        // A format below the whole readable set, rather than `FORMAT - 1`, because the set has
13769        // more than one member now: format 22 is deliberately still readable, so the version that
13770        // has to be refused is the one under the oldest one accepted.
13771        let unreadable =
13772            READABLE.iter().copied().min().expect("at least one format is readable") - 1;
13773        let mut file = OpenOptions::new().write(true).open(&older).expect("open for the header");
13774        file.seek(SeekFrom::Start(8)).expect("the version follows the magic");
13775        file.write_all(&unreadable.to_le_bytes()).expect("write an older version");
13776        drop(file);
13777        let complaint = Reader::open(&older).expect_err("an older format is refused").to_string();
13778        assert!(complaint.contains(&format!("format {unreadable}")), "{complaint}");
13779        assert!(complaint.contains(&format!("format {FORMAT}")), "{complaint}");
13780
13781        let mut file = OpenOptions::new().write(true).open(&older).expect("open for the header");
13782        file.seek(SeekFrom::Start(0)).expect("the magic is first");
13783        file.write_all(b"NOTRUDB!").expect("write another engine's magic");
13784        drop(file);
13785        let complaint = Reader::open(&older).expect_err("a foreign file is refused").to_string();
13786        assert!(complaint.contains("magic"), "{complaint}");
13787        assert!(!complaint.contains("format"), "a version has nothing to do with it: {complaint}");
13788        fs::remove_file(older).expect("remove scratch file");
13789    }
13790
13791    #[test]
13792    fn an_unfinished_or_damaged_file_does_not_answer_with_partial_rows() {
13793        let unfinished = path("unfinished");
13794        let mut writer =
13795            Writer::create(&unfinished, "items", vec![Field::new("id", LogicalType::Integer)])
13796                .expect("new file");
13797        let chunk = Chunk::new(vec![
13798            Vector::flat(LogicalType::Integer, Data::Int32(vec![1, 2, 3].into()))
13799                .expect("integers"),
13800        ])
13801        .expect("chunk");
13802        writer.append(&chunk).expect("page written");
13803        drop(writer);
13804        assert!(Reader::open(&unfinished).is_err(), "no directory was committed");
13805        fs::remove_file(unfinished).expect("remove scratch file");
13806
13807        let damaged = path("damaged");
13808        let mut writer =
13809            Writer::create(&damaged, "items", vec![Field::new("id", LogicalType::Integer)])
13810                .expect("new file");
13811        writer.append(&chunk).expect("page written");
13812        writer.finish().expect("commit");
13813        let reader = Reader::open(&damaged).expect("valid directory");
13814        let mut file =
13815            OpenOptions::new().write(true).open(&damaged).expect("open for a damaged page");
13816        file.seek(SeekFrom::Start(HEADER + 1)).expect("inside first page");
13817        file.write_all(&[255]).expect("damage one byte");
13818        assert!(reader.read(0, &[0]).is_err(), "page checksum rejects corruption");
13819        fs::remove_file(damaged).expect("remove scratch file");
13820    }
13821
13822    #[test]
13823    fn damaged_lazy_dictionary_payload_is_an_error() {
13824        let path = path("damaged-dictionary");
13825        let mut writer = Writer::create(
13826            &path,
13827            "items",
13828            vec![
13829                Field::required("id", LogicalType::Integer),
13830                Field::new("text", LogicalType::Varchar),
13831            ],
13832        )
13833        .expect("new file");
13834        writer.append(&sample()).expect("stripe written");
13835        writer.finish().expect("commit");
13836
13837        let reader = Reader::open(&path).expect("valid directory");
13838        let dictionary = reader.table.dictionaries[1].expect("string dictionary page");
13839        // Read the count out of the page rather than writing it here, so that adding something
13840        // else to the index does not silently turn this into a test that damages the index.
13841        let mut header = [0; DICTIONARY_HEADER];
13842        read_at(&reader.file, dictionary.offset, &mut header).expect("dictionary header");
13843        // The first block's start is the first word after the offsets, since the blocks are written
13844        // during the load and are wherever the writer was when each was encoded.
13845        let count = u32::from_le_bytes(header[0..4].try_into().expect("four bytes")) as usize;
13846        let width = u32::from_le_bytes(header[12..16].try_into().expect("four bytes"));
13847        assert_ne!(width & DICTIONARY_SCATTERED, 0, "the blocks say where they are");
13848        let bits = (width & !DICTIONARY_FLAGS) as usize;
13849        let mut start = [0; 8];
13850        let at = dictionary.offset + (DICTIONARY_HEADER + offset_bytes(count, bits)) as u64;
13851        read_at(&reader.file, at, &mut start).expect("the first block's start");
13852        let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
13853        file.seek(SeekFrom::Start(u64::from_le_bytes(start))).expect("inside dictionary payload");
13854        file.write_all(&[255]).expect("damage dictionary payload");
13855
13856        let chunk = reader.read(0, &[1]).expect("code page and dictionary index remain valid");
13857        let error =
13858            chunk.validate_external().expect_err("payload corruption must reach the caller");
13859        assert!(error.message().contains("payload checksum differs"), "{error}");
13860        fs::remove_file(path).expect("remove scratch file");
13861    }
13862
13863    /// A column whose values are all different is written without a dictionary, and one whose
13864    /// values repeat keeps it.
13865    ///
13866    /// The two columns go in the same table and hold the same number of rows, so the only thing
13867    /// separating them is how much of the first stripe was a value it had not seen before. Both have
13868    /// to read back the values that were written, because the decision is about cost and nothing
13869    /// else. The file size is the other half of it: a column written without a dictionary goes
13870    /// through the string cascade instead, so dropping the dictionary must not turn into storing the
13871    /// column raw.
13872    #[test]
13873    fn a_column_of_all_different_values_is_written_without_a_dictionary() {
13874        let path = path("dictionary-decide");
13875        let rows = 20_000;
13876        // Long enough that storing it raw would show, and different in every row.
13877        let unique =
13878            |row: usize| format!("{row:09} a value that appears exactly once in the table");
13879        // The same values in the same shape, each one used forty times over.
13880        let repeated = |row: usize| unique(row / 40);
13881        let mut writer = Writer::create(
13882            &path,
13883            "items",
13884            vec![
13885                Field::required("unique", LogicalType::Varchar),
13886                Field::required("repeated", LogicalType::Varchar),
13887            ],
13888        )
13889        .expect("new file");
13890        for part in (0..rows).step_by(1_000) {
13891            let span = part..(part + 1_000).min(rows);
13892            let left = span.clone().map(|row| Value::Varchar(unique(row))).collect::<Vec<_>>();
13893            let right = span.map(|row| Value::Varchar(repeated(row))).collect::<Vec<_>>();
13894            writer
13895                .append(
13896                    &Chunk::new(vec![
13897                        Vector::from_values(LogicalType::Varchar, &left).expect("strings"),
13898                        Vector::from_values(LogicalType::Varchar, &right).expect("strings"),
13899                    ])
13900                    .expect("two columns"),
13901                )
13902                .expect("a part");
13903        }
13904        writer.finish().expect("commit");
13905
13906        let reader = Reader::open(&path).expect("reopen from disk");
13907        assert!(
13908            reader.table.dictionaries[0].is_none(),
13909            "a column with no repeats has nothing to say twice"
13910        );
13911        assert!(
13912            reader.table.dictionaries[1].is_some(),
13913            "a column whose values come round again keeps its dictionary"
13914        );
13915        let mut first = 0;
13916        for part in 0..reader.parts() {
13917            let chunk = reader.read(part, &[0, 1]).expect("a part");
13918            for row in 0..chunk.len() {
13919                assert_eq!(chunk.value_at(row, 0), Value::Varchar(unique(first + row)));
13920                assert_eq!(chunk.value_at(row, 1), Value::Varchar(repeated(first + row)));
13921            }
13922            first += chunk.len();
13923        }
13924        assert_eq!(first, rows, "every row was read back");
13925        let raw = (0..rows).map(|row| unique(row).len()).sum::<usize>();
13926        let size = fs::metadata(&path).expect("the file is there").len() as usize;
13927        assert!(size < raw, "a column without a dictionary is still encoded: {size} against {raw}");
13928        fs::remove_file(path).expect("remove scratch file");
13929    }
13930
13931    /// A payload of many blocks reads and checks every block of it.
13932    ///
13933    /// The test above has a dictionary of three values, which is one block, so it says nothing
13934    /// about a reader finding the right block among many. This one has thirty two thousand values,
13935    /// which is thirty two blocks, and it reads a value out of the first block and a value out of
13936    /// the last and then damages the last and asks for it again.
13937    ///
13938    /// Forty thousand rows over those thirty two thousand values, because a column the writer finds
13939    /// to be all distinct does not get a dictionary at all and there would be nothing here to test.
13940    /// Four rows in five holding a value the stripe has not seen before is a column that keeps one.
13941    /// The repeats are put at the front so that the values still arrive in order after them, which
13942    /// is what keeps the last part of the table on the last block of the payload.
13943    #[test]
13944    fn a_dictionary_over_many_blocks_checks_every_block_of_it() {
13945        let path = path("dictionary-blocks");
13946        let value = |row: usize| {
13947            let row = row.saturating_sub(8_000);
13948            format!("{row:07} a value long enough to be worth a payload block")
13949        };
13950        let parts = 40;
13951        let per_part = 1000;
13952        let mut writer =
13953            Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
13954                .expect("new file");
13955        for part in 0..parts {
13956            let values = (0..per_part)
13957                .map(|row| Value::Varchar(value(part * per_part + row)))
13958                .collect::<Vec<_>>();
13959            let chunk = Chunk::new(vec![
13960                Vector::from_values(LogicalType::Varchar, &values).expect("strings"),
13961            ])
13962            .expect("matching rows");
13963            writer.append(&chunk).expect("a part");
13964        }
13965        writer.finish().expect("commit");
13966
13967        let reader = Reader::open(&path).expect("reopen from disk");
13968        let dictionary = reader.table.dictionaries[0].expect("string dictionary page");
13969        assert!(
13970            parts * per_part > TEXT_PAYLOAD_VALUES * 4,
13971            "the dictionary has to be several blocks for this to be testing anything"
13972        );
13973        for part in [0, parts - 1] {
13974            let chunk = reader.read(part, &[0]).expect("a part");
13975            chunk.validate_external().expect("every payload block checks out");
13976            assert_eq!(chunk.value_at(0, 0), Value::Varchar(value(part * per_part)));
13977        }
13978
13979        // The last block is wherever the writer was when it was encoded, which the index says.
13980        let mut header = [0; DICTIONARY_HEADER];
13981        read_at(&reader.file, dictionary.offset, &mut header).expect("dictionary header");
13982        let count = u32::from_le_bytes(header[0..4].try_into().expect("four bytes")) as usize;
13983        let blocks = u32::from_le_bytes(header[8..12].try_into().expect("four bytes")) as usize;
13984        let width = u32::from_le_bytes(header[12..16].try_into().expect("four bytes"));
13985        let bits = (width & !DICTIONARY_FLAGS) as usize;
13986        let mut place = [0; 16];
13987        let at = DICTIONARY_HEADER + offset_bytes(count, bits) + (blocks - 1) * 16;
13988        read_at(&reader.file, dictionary.offset + at as u64, &mut place).expect("its place");
13989        let start = u64::from_le_bytes(place[..8].try_into().expect("eight bytes"));
13990        let length = u64::from_le_bytes(place[8..].try_into().expect("eight bytes"));
13991        let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
13992        file.seek(SeekFrom::Start(start + length - 4)).expect("the last bytes of the last block");
13993        file.write_all(&[255]).expect("damage the last payload block");
13994        let reader = Reader::open(&path).expect("the directory and the index are untouched");
13995        let chunk = reader.read(parts - 1, &[0]).expect("the code page remains valid");
13996        let error = chunk.validate_external().expect_err("the damage must reach the caller");
13997        assert!(error.message().contains("payload checksum differs"), "{error}");
13998        fs::remove_file(path).expect("remove scratch file");
13999    }
14000
14001    /// Values of different lengths read back where the offsets say they do.
14002    ///
14003    /// The offsets are packed at one width for the column, they are relative to the payload block a
14004    /// value lands in, and they go in runs of half a block, so there are two boundaries where the
14005    /// arithmetic could be off by one and neither shows up on values that are all the same length.
14006    /// This writes 5,000 values whose lengths cycle through a wide range and reads every one back,
14007    /// so the first value of a block, the last value of a run and the last value of a block are all
14008    /// covered several times over. An empty value is in the cycle because a zero length span is the
14009    /// case the reader short circuits.
14010    ///
14011    /// Six thousand rows over those 5,000 values, because a column the writer finds to be all
14012    /// distinct is written without a dictionary and then there are no packed offsets to be off by
14013    /// one in.
14014    #[test]
14015    fn values_of_different_lengths_read_back_out_of_packed_offsets() {
14016        let path = path("dictionary-offsets");
14017        let value = |row: usize| {
14018            let row = row % 5_000;
14019            if row % 511 == 3 { String::new() } else { "x".repeat(row % 97) + &format!("{row:05}") }
14020        };
14021        let rows = 6_000;
14022        let mut writer =
14023            Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
14024                .expect("new file");
14025        let values = (0..rows).map(|row| Value::Varchar(value(row))).collect::<Vec<_>>();
14026        for part in values.chunks(1_000) {
14027            let chunk =
14028                Chunk::new(vec![Vector::from_values(LogicalType::Varchar, part).expect("strings")])
14029                    .expect("matching rows");
14030            writer.append(&chunk).expect("a part");
14031        }
14032        writer.finish().expect("commit");
14033
14034        let reader = Reader::open(&path).expect("reopen from disk");
14035        assert!(
14036            rows > TEXT_PAYLOAD_VALUES * 4,
14037            "the dictionary has to be several blocks for this to be testing anything"
14038        );
14039        for part in 0..rows / 1_000 {
14040            let chunk = reader.read(part, &[0]).expect("a part");
14041            for row in 0..1_000 {
14042                let row = part * 1_000 + row;
14043                assert_eq!(
14044                    chunk.value_at(row % 1_000, 0),
14045                    Value::Varchar(value(row)),
14046                    "value {row}"
14047                );
14048            }
14049        }
14050        // The lengths a vector at a time, twice over, because the first pass is what makes the
14051        // table of ends worth building and the second is read out of the lengths worked out of it.
14052        for _ in 0..2 {
14053            for part in 0..rows / 1_000 {
14054                let chunk = reader.read(part, &[0]).expect("a part");
14055                let mut lens = vec![0_i64; 1_000];
14056                let column = chunk.column(0).expect("one column");
14057                assert!(column.try_bytes_lens(&mut lens).expect("lengths"), "a stored column");
14058                for (row, &len) in lens.iter().enumerate() {
14059                    let row = part * 1_000 + row;
14060                    assert_eq!(len as usize, value(row).len(), "the length of value {row}");
14061                }
14062            }
14063        }
14064        fs::remove_file(path).expect("remove scratch file");
14065    }
14066
14067    /// Lengths start again at every block, and ends that go backwards inside one give no table.
14068    #[test]
14069    fn lengths_restart_at_each_block_and_refuse_ends_that_go_backwards() {
14070        let mut ends: Vec<u32> = (1..=TEXT_PAYLOAD_VALUES as u32).map(|at| at * 2).collect();
14071        ends.extend([3, 3, 10]);
14072        let lens = lengths_of(&ends).expect("ordered ends");
14073        assert!(lens[..TEXT_PAYLOAD_VALUES].iter().all(|&len| len == 2));
14074        assert_eq!(&lens[TEXT_PAYLOAD_VALUES..], &[3, 0, 7]);
14075        ends.push(9);
14076        assert_eq!(lengths_of(&ends), None);
14077    }
14078
14079    /// Every worker of a scan wants the dictionary at the same moment and one of them fetches it.
14080    ///
14081    /// Asking a `OnceLock` whether it holds something answers the question a worker that already has
14082    /// the dictionary is asking and not the one a worker without it is asking, which is whether
14083    /// somebody is already on their way with it. Sixteen workers that all miss will all read the
14084    /// page, all verify it and all decode it, and fifteen will drop the result. Nothing about that
14085    /// is incorrect, which is why it went unnoticed, and it showed up as ClickBench 38 getting
14086    /// slower when the scan in front of it got faster and stopped staggering the arrivals.
14087    ///
14088    /// The barrier is what makes the test about that rather than about luck. Without it the first
14089    /// thread is usually finished before the last one starts and the count is one either way.
14090    #[test]
14091    fn a_global_dictionary_is_opened_once_however_many_workers_ask_at_once() {
14092        let path = path("dictionary-once");
14093        let parts = 8;
14094        let per_part = 500;
14095        let value =
14096            |row: usize| format!("{row:07} a value long enough to be worth a payload block");
14097        let mut writer =
14098            Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
14099                .expect("new file");
14100        for part in 0..parts {
14101            let values = (0..per_part)
14102                .map(|row| Value::Varchar(value(part * per_part + row)))
14103                .collect::<Vec<_>>();
14104            let chunk = Chunk::new(vec![
14105                Vector::from_values(LogicalType::Varchar, &values).expect("strings"),
14106            ])
14107            .expect("matching rows");
14108            writer.append(&chunk).expect("a part");
14109        }
14110        writer.finish().expect("commit");
14111
14112        let reader = Reader::open(&path).expect("reopen from disk");
14113        assert!(reader.table.dictionaries[0].is_some(), "the column has to have one to share");
14114        assert_eq!(reader.reads().dictionaries, 0, "opening the file does not open a dictionary");
14115
14116        let workers = 16;
14117        let gate = std::sync::Barrier::new(workers);
14118        std::thread::scope(|scope| {
14119            for worker in 0..workers {
14120                let reader = reader.clone();
14121                let gate = &gate;
14122                scope.spawn(move || {
14123                    gate.wait();
14124                    let chunk = reader.read(worker % parts, &[0]).expect("a part");
14125                    assert_eq!(
14126                        chunk.value_at(0, 0),
14127                        Value::Varchar(value((worker % parts) * per_part))
14128                    );
14129                });
14130            }
14131        });
14132
14133        assert_eq!(reader.reads().dictionaries, 1, "sixteen workers, one dictionary, one open");
14134        fs::remove_file(path).expect("remove scratch file");
14135    }
14136
14137    /// The sorted order sits outside the index the page checksum covers, because a query that
14138    /// never searches a dictionary should not read it, so it carries its own checksums and this is
14139    /// what says they are checked. A search that trusted a damaged order would give a wrong answer
14140    /// rather than a slow one.
14141    #[test]
14142    fn a_damaged_sorted_order_is_an_error() {
14143        let path = path("damaged-order");
14144        let mut writer = Writer::create(
14145            &path,
14146            "items",
14147            vec![
14148                Field::required("id", LogicalType::Integer),
14149                Field::new("text", LogicalType::Varchar),
14150            ],
14151        )
14152        .expect("new file");
14153        writer.append(&sample()).expect("stripe written");
14154        writer.finish().expect("commit");
14155
14156        let reader = Reader::open(&path).expect("valid directory");
14157        let page = reader.table.dictionaries[1].expect("string dictionary page");
14158        let mut header = [0; DICTIONARY_HEADER];
14159        read_at(&reader.file, page.offset, &mut header).expect("dictionary header");
14160        let index_len = dictionary_index_len(&header);
14161        let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
14162        file.seek(SeekFrom::Start(page.offset + index_len)).expect("the first head");
14163        file.write_all(&[255]).expect("damage the order");
14164
14165        let dictionary = reader.dictionary(1).expect("read").expect("a string column has one");
14166        let error = dictionary.compare_rank(0, b"anything").expect_err("a damaged order is caught");
14167        assert!(error.message().contains("rank checksum differs"), "{error}");
14168        fs::remove_file(path).expect("remove scratch file");
14169    }
14170
14171    /// Codes stay in first appearance order and the sorted order is written beside them, so a
14172    /// reader can put the values back in order without the writer having had to know them all
14173    /// before it handed out the first code.
14174    #[test]
14175    fn a_global_dictionary_carries_the_sorted_order_of_its_values() {
14176        // Chosen so the sort cannot be decided on the first eight bytes alone. Three values share
14177        // a nine byte prefix, one is a prefix of another, and one is empty.
14178        let spellings = ["overlong1z", "b", "", "overlong1a", "overlong", "ab", "a", "overlong1"];
14179        let path = path("dictionary-order");
14180        let mut writer =
14181            Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
14182                .expect("new file");
14183        writer
14184            .append(
14185                &Chunk::new(vec![
14186                    Vector::from_values(
14187                        LogicalType::Varchar,
14188                        &spellings.map(|text| Value::Varchar(text.into())),
14189                    )
14190                    .expect("strings"),
14191                ])
14192                .expect("one column"),
14193            )
14194            .expect("stripe written");
14195        writer.finish().expect("commit");
14196
14197        let reader = Reader::open(&path).expect("valid directory");
14198        let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
14199        let count = dictionary.ranks().expect("a v10 file stores one");
14200        assert_eq!(count, spellings.len(), "every distinct value has a rank");
14201        let order = (0..count)
14202            .map(|rank| dictionary.code_at_rank(rank).expect("a code"))
14203            .collect::<Vec<_>>();
14204        let mut seen = order.clone();
14205        seen.sort_unstable();
14206        assert_eq!(seen, (0..spellings.len() as u32).collect::<Vec<_>>(), "a permutation of codes");
14207
14208        let ranked = order
14209            .iter()
14210            .map(|&code| {
14211                dictionary.try_bytes_at(code as usize).expect("read").expect("a value").to_vec()
14212            })
14213            .collect::<Vec<_>>();
14214        let mut expected = spellings.map(|text| text.as_bytes().to_vec()).to_vec();
14215        expected.sort();
14216        assert_eq!(ranked, expected, "rank order is value order");
14217
14218        // What a search asks, on the values themselves rather than through a kernel, so that a
14219        // file whose heads disagree with its bytes is caught here rather than as a wrong answer.
14220        for (rank, value) in expected.iter().enumerate() {
14221            assert_eq!(
14222                dictionary.compare_rank(rank, value).expect("compare"),
14223                Ordering::Equal,
14224                "rank {rank} is its own value"
14225            );
14226            if rank > 0 {
14227                assert_eq!(
14228                    dictionary.compare_rank(rank - 1, value).expect("compare"),
14229                    Ordering::Less,
14230                    "rank {rank} follows the one before it"
14231                );
14232            }
14233        }
14234        fs::remove_file(path).expect("remove scratch file");
14235    }
14236
14237    /// Five text columns of different sizes close at the same time, and each comes back with its
14238    /// own values in its own order.
14239    ///
14240    /// The sizes differ so that the columns are taken in an order that is not the column order, and
14241    /// the values of each column are spelled with its number so that one column's page written in
14242    /// another's place would read back as the wrong strings rather than the right ones by chance.
14243    #[test]
14244    fn text_columns_closed_at_once_each_keep_their_own_dictionary() {
14245        let sizes = [300_usize, 5_000, 40, 2_000, 1_200];
14246        let path = path("dictionaries-at-once");
14247        let fields = (0..sizes.len())
14248            .map(|column| Field::new(format!("text{column}"), LogicalType::Varchar))
14249            .collect::<Vec<_>>();
14250        let mut writer = Writer::create(&path, "items", fields).expect("new file");
14251        let rows = 10_000_usize;
14252        for start in (0..rows).step_by(1_024) {
14253            let columns = sizes
14254                .iter()
14255                .enumerate()
14256                .map(|(column, &size)| {
14257                    let values = (start..(start + 1_024).min(rows))
14258                        .map(|row| Value::Varchar(format!("c{column}-{:05}", (row * 7919) % size)))
14259                        .collect::<Vec<_>>();
14260                    Vector::from_values(LogicalType::Varchar, &values).expect("strings")
14261                })
14262                .collect::<Vec<_>>();
14263            writer.append(&Chunk::new(columns).expect("five columns")).expect("stripe written");
14264        }
14265        writer.finish().expect("commit");
14266
14267        let reader = Reader::open(&path).expect("valid directory");
14268        for (column, &size) in sizes.iter().enumerate() {
14269            let dictionary =
14270                reader.dictionary(column).expect("read").expect("a string column has one");
14271            let count = dictionary.ranks().expect("a v10 file stores one");
14272            assert_eq!(count, size, "column {column} has its own distinct count");
14273            let ranked = (0..count)
14274                .map(|rank| {
14275                    let code = dictionary.code_at_rank(rank).expect("a code");
14276                    dictionary.try_bytes_at(code as usize).expect("read").expect("a value").to_vec()
14277                })
14278                .collect::<Vec<_>>();
14279            let expected = (0..size)
14280                .map(|value| format!("c{column}-{value:05}").into_bytes())
14281                .collect::<Vec<_>>();
14282            assert_eq!(ranked, expected, "column {column} ranks its own values in order");
14283        }
14284        fs::remove_file(path).expect("remove scratch file");
14285    }
14286
14287    /// A dictionary large enough to be decoded and sorted on several threads ranks the way one small
14288    /// enough for one thread does.
14289    ///
14290    /// Seventy thousand values over sixty nine blocks, in no order and each four times over so the
14291    /// column is worth a dictionary, written and ranked in the close.
14292    /// Some share a long prefix and some differ only in the last byte, so the buckets of the sort cut
14293    /// through runs of values that agree for a long way.
14294    #[test]
14295    fn a_large_dictionary_ranks_in_value_order() {
14296        let path = path("dictionary-large-rank");
14297        let value = |row: u64| {
14298            let mixed = row.wrapping_mul(0x9e37_79b9_7f4a_7c15) >> 40;
14299            match row % 3 {
14300                0 => format!("https://example.com/a/long/shared/path/{mixed:08}"),
14301                1 => format!("{mixed}"),
14302                _ => format!("x{}", row % 1000).repeat(1 + (row % 4) as usize) + &row.to_string(),
14303            }
14304        };
14305        let distinct = 70_000;
14306        let parts = 4 * distinct / 1000;
14307        let per_part = 1000;
14308        let mut writer =
14309            Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
14310                .expect("new file");
14311        for part in 0..parts {
14312            let values = (0..per_part)
14313                .map(|row| Value::Varchar(value((part * per_part + row) / 4)))
14314                .collect::<Vec<_>>();
14315            let chunk = Chunk::new(vec![
14316                Vector::from_values(LogicalType::Varchar, &values).expect("strings"),
14317            ])
14318            .expect("matching rows");
14319            writer.append(&chunk).expect("a part");
14320        }
14321        writer.finish().expect("commit");
14322
14323        let reader = Reader::open(&path).expect("reopen from disk");
14324        let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
14325        let count = dictionary.ranks().expect("a ranked dictionary");
14326        assert_eq!(count, distinct as usize, "every distinct value has a rank");
14327        assert!(count >= PARALLEL_SORT_MIN, "too few values to be sorted on more than one thread");
14328        let ranked = (0..count)
14329            .map(|rank| {
14330                let code = dictionary.code_at_rank(rank).expect("a code");
14331                dictionary.try_bytes_at(code as usize).expect("read").expect("a value").to_vec()
14332            })
14333            .collect::<Vec<_>>();
14334        let mut expected = (0..distinct).map(|row| value(row).into_bytes()).collect::<Vec<_>>();
14335        expected.sort();
14336        assert_eq!(ranked, expected, "rank order is value order");
14337        fs::remove_file(path).expect("remove scratch file");
14338    }
14339
14340    /// A string column's synopsis is turned into values without keeping the blocks it went through.
14341    ///
14342    /// Three thousand values, every fifth of them four times over, so the synopsis is a prefix of
14343    /// five hundred and twelve codes spread over all three payload blocks. Reading it used to leave
14344    /// all three decoded for as long as the reader lived. It leaves none of them now, and the second
14345    /// read answers out of what the first remembered.
14346    /// A directory read out of the file a window at a time is the directory read whole.
14347    ///
14348    /// The windows here are far smaller than any field is long, so every kind of field is split
14349    /// across a refill somewhere, and a bound is offered to its codec short more than once. The
14350    /// synopses are left in the file, and each one read back from where it was left is the one the
14351    /// whole read decoded.
14352    #[test]
14353    fn a_directory_read_a_window_at_a_time_is_the_directory_read_whole() {
14354        let path = path("windowed-directory");
14355        let fields = vec![
14356            Field::required("id", LogicalType::BigInt),
14357            Field::required("word", LogicalType::Varchar),
14358            Field::new("score", LogicalType::Double),
14359        ];
14360        let mut writer = Writer::create(&path, "items", fields).expect("new file");
14361        for part in 0..70_i64 {
14362            let ids = (0..100).map(|row| Value::BigInt(part * 100 + row % 7)).collect::<Vec<_>>();
14363            let words = (0..100)
14364                .map(|row| Value::Varchar(format!("word {}", row % 13)))
14365                .collect::<Vec<_>>();
14366            let scores = (0..100)
14367                .map(|row| if row % 4 == 0 { Value::Null } else { Value::Double(row as f64) })
14368                .collect::<Vec<_>>();
14369            let chunk = Chunk::new(vec![
14370                Vector::from_values(LogicalType::BigInt, &ids).expect("integers"),
14371                Vector::from_values(LogicalType::Varchar, &words).expect("strings"),
14372                Vector::from_values(LogicalType::Double, &scores).expect("doubles"),
14373            ])
14374            .expect("three columns");
14375            writer.append(&chunk).expect("a part");
14376        }
14377        writer.finish().expect("commit");
14378
14379        let catalog = Catalog::open(&path).expect("reopen");
14380        let entry = catalog.entries.first().expect("one table").directory;
14381        let (offset, length) = (entry.offset, entry.length as usize);
14382        let mut bytes = vec![0; length];
14383        read_at(&catalog.file, offset, &mut bytes).expect("the directory");
14384        assert_eq!(file_checksum(&catalog.file, offset, length).expect("checksum"), entry.hash);
14385        let whole = decode_directory(&bytes, catalog.size).expect("whole");
14386        assert!(whole.stripes.len() > 1, "the table should span stripes");
14387        for size in [1, 7, 33, 4_096] {
14388            let mut cursor = Cursor::over(&catalog.file, offset, length);
14389            cursor.window.as_mut().expect("a window").size = size;
14390            let windowed = read_directory(cursor, catalog.size, Some(offset)).expect("windowed");
14391            assert_eq!(format!("{:?}", windowed.stripes), format!("{:?}", whole.stripes));
14392            assert_eq!(format!("{:?}", windowed.fields), format!("{:?}", whole.fields));
14393            let mut stored = 0;
14394            for (column, (left, held)) in
14395                windowed.frequencies.iter().zip(&whole.frequencies).enumerate()
14396            {
14397                match (left, held) {
14398                    (None, None) => {}
14399                    (
14400                        Some(super::Frequencies::Stored { span, values }),
14401                        Some(super::Frequencies::Held(summary)),
14402                    ) => {
14403                        let mut one = vec![0; span.length as usize];
14404                        read_at(&catalog.file, span.offset, &mut one).expect("a synopsis");
14405                        let read = decode_summary(
14406                            &mut Cursor::new(&one),
14407                            &whole.fields[column],
14408                            whole.rows,
14409                            *values,
14410                        )
14411                        .expect("a valid synopsis")
14412                        .expect("one is there");
14413                        assert_eq!(format!("{read:?}"), format!("{summary:?}"));
14414                        stored += 1;
14415                    }
14416                    other => panic!("column {column} came back as {other:?}"),
14417                }
14418            }
14419            assert!(stored >= 2, "only {stored} synopses were left in the file");
14420        }
14421        let reader = catalog.table("items").expect("the table");
14422        assert!(reader.frequency_summaries[1].get().is_none());
14423        assert!(reader.top_frequencies(1, 1).expect("a readable synopsis").is_some());
14424        let first = reader.frequency_summaries[1].get().expect("decoded synopsis");
14425        let clone = reader.clone();
14426        assert!(clone.top_frequencies(1, 1).expect("cached synopsis").is_some());
14427        assert!(Arc::ptr_eq(first, clone.frequency_summaries[1].get().expect("same synopsis")));
14428        fs::remove_file(path).expect("remove scratch file");
14429    }
14430
14431    #[test]
14432    fn a_checksum_carried_across_reads_is_the_checksum_of_the_whole() {
14433        let path = path("file-checksum");
14434        let bytes = (0..200_000_u32)
14435            .map(|at| (at.wrapping_mul(2_654_435_761) >> 13) as u8)
14436            .collect::<Vec<_>>();
14437        fs::write(&path, &bytes).expect("scratch file");
14438        let file = File::open(&path).expect("open");
14439        for (offset, length) in [
14440            (0, 0),
14441            (3, 1),
14442            (5, 31),
14443            (0, 32),
14444            (9, 33),
14445            (1, 65_536),
14446            (7, 65_567),
14447            (0, 200_000),
14448            (11, 131_101),
14449        ] {
14450            let whole = checksum(&bytes[offset..offset + length]);
14451            assert_eq!(
14452                file_checksum(&file, offset as u64, length).expect("read"),
14453                whole,
14454                "{offset} {length}"
14455            );
14456        }
14457        fs::remove_file(path).expect("remove scratch file");
14458    }
14459
14460    #[test]
14461    fn a_string_synopsis_is_read_without_keeping_the_dictionary_blocks() {
14462        let path = path("synopsis-keeps-no-block");
14463        let spelled = |index: usize| Value::Varchar(format!("phrase {index:05}"));
14464        let mut values = (0..3_000).map(spelled).collect::<Vec<_>>();
14465        for _ in 0..3 {
14466            values.extend((0..3_000).step_by(5).map(spelled));
14467        }
14468        let mut writer =
14469            Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
14470                .expect("new file");
14471        for part in values.chunks(1_024) {
14472            writer
14473                .append(
14474                    &Chunk::new(vec![
14475                        Vector::from_values(LogicalType::Varchar, part).expect("strings"),
14476                    ])
14477                    .expect("one column"),
14478                )
14479                .expect("a part");
14480        }
14481        writer.finish().expect("commit");
14482
14483        let reader = Reader::open(&path).expect("reopen from disk");
14484        let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
14485        let resting = dictionary.footprint();
14486        let prefix = reader.frequency_prefix(0).expect("a readable synopsis").expect("one");
14487        assert_eq!(prefix.entries.len(), 512);
14488        for (value, count) in &prefix.entries {
14489            let Value::Varchar(text) = value else { panic!("a string column gave {value:?}") };
14490            let index = text["phrase ".len()..].parse::<usize>().expect("a spelled number");
14491            assert_eq!((index % 5, *count), (0, 4), "{text} came back with {count}");
14492        }
14493        assert_eq!(dictionary.footprint(), resting, "reading the synopsis kept a decoded block");
14494        let again = reader.frequency_prefix(0).expect("a readable synopsis").expect("one");
14495        assert_eq!(again.entries, prefix.entries);
14496        fs::remove_file(path).expect("remove scratch file");
14497    }
14498
14499    /// A sweep of the dictionary reads every value, and the second sweep keeps what it read, up to
14500    /// the budget.
14501    ///
14502    /// The point of the sweep is the resident size rather than the answer, so both are checked
14503    /// here. The first sweep keeps nothing, because a process that runs one statement never reads
14504    /// a block twice. A dictionary this small is well under [`TEXT_KEEP_BUDGET`], so the second
14505    /// sweep keeps everything and a third decodes nothing, which is what makes a session asking the
14506    /// same question again cost what it should. The ceiling is the other half of it and it has its own
14507    /// test below, because a ceiling that never binds is not a ceiling anybody checked.
14508    #[test]
14509    fn a_dictionary_sweep_reads_every_value_and_keeps_it_under_the_budget() {
14510        let path = path("dictionary-sweep");
14511        // Two thousand five hundred distinct values is two whole payload blocks and a part of a
14512        // third, so the sweep has to be called more than once and the last call has to stop short.
14513        let spellings = (0..2_500)
14514            .map(|index| Value::Varchar(format!("value {index:08} {}", "x".repeat(index % 40))))
14515            .collect::<Vec<_>>();
14516        let mut writer =
14517            Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
14518                .expect("new file");
14519        // A chunk is a part and a part is at most 1,024 rows, so the values go in three of them.
14520        // The dictionary is table wide and does not care where a value was written.
14521        for part in spellings.chunks(1_024) {
14522            writer
14523                .append(
14524                    &Chunk::new(vec![
14525                        Vector::from_values(LogicalType::Varchar, part).expect("strings"),
14526                    ])
14527                    .expect("one column"),
14528                )
14529                .expect("stripe written");
14530        }
14531        writer.finish().expect("commit");
14532
14533        let reader = Reader::open(&path).expect("valid directory");
14534        let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
14535        assert_eq!(dictionary.len(), spellings.len(), "every value is distinct");
14536        for first in [0, TEXT_PAYLOAD_VALUES, TEXT_PAYLOAD_VALUES * 2] {
14537            assert!(dictionary.text_block_might_contain(first, b"value").expect("signature"));
14538            assert!(!dictionary.text_block_might_contain(first, b"google").expect("signature"));
14539        }
14540
14541        let resting = dictionary.footprint();
14542        let sweep = || {
14543            let mut swept: Vec<Vec<u8>> = Vec::new();
14544            let mut at = 0;
14545            let mut calls = 0;
14546            while at < dictionary.len() {
14547                let stopped = dictionary
14548                    .sweep_text(at, dictionary.len(), &mut |index: usize, text: &[u8]| {
14549                        assert_eq!(index, swept.len(), "a sweep hands its values over in order");
14550                        swept.push(text.to_vec());
14551                        Ok(())
14552                    })
14553                    .expect("a sweep reads");
14554                assert!(stopped > at, "a sweep moves");
14555                at = stopped;
14556                calls += 1;
14557            }
14558            assert_eq!(calls, 3, "a sweep hands over one block at a time");
14559            swept
14560        };
14561        let swept = sweep();
14562        assert_eq!(dictionary.footprint(), resting, "a first sweep keeps nothing it decoded");
14563        assert_eq!(sweep(), swept, "a second sweep reads what the first did");
14564        let after = dictionary.footprint();
14565        assert!(after > resting, "a second sweep under the budget keeps what it decoded");
14566
14567        let read = (0..dictionary.len())
14568            .map(|code| dictionary.try_bytes_at(code).expect("read").expect("a value").to_vec())
14569            .collect::<Vec<_>>();
14570        assert_eq!(swept, read, "a sweep answers what a point read answers");
14571        // A read per value is about what makes the unpacked ends worth building, so whether they
14572        // are built here depends on how many reads the sweep made on the way. They are the one thing
14573        // allowed to grow, by four bytes a value, and nothing of the payload is.
14574        let grown = dictionary.footprint() - after;
14575        assert!(
14576            grown == 0 || grown == dictionary.len() * size_of::<u32>(),
14577            "a point read of a kept block decodes nothing, and {grown} bytes grew"
14578        );
14579        fs::remove_file(path).expect("remove scratch file");
14580    }
14581
14582    #[test]
14583    fn a_narrow_signature_of_an_older_file_answers_by_its_own_width() {
14584        let path = path("narrow-substring-signature");
14585        let blocks = [&b"https://google.com/"[..], b"https://example.org/", b"mail.google.com"];
14586        let mut grams = Vec::new();
14587        for text in blocks {
14588            let mut bits = vec![0_u8; NARROW_GRAM_BYTES];
14589            for gram in text.windows(4) {
14590                for bit in gram_bits(gram, NARROW_GRAM_BYTES) {
14591                    bits[bit / 8] |= 1 << (bit % 8);
14592                }
14593            }
14594            grams.extend(bits);
14595        }
14596        fs::write(&path, &grams).expect("scratch file");
14597        let file = File::open(&path).expect("open scratch file");
14598        let signatures = NativeGrams {
14599            start: 0,
14600            length: grams.len(),
14601            width: NARROW_GRAM_BYTES,
14602            hash: checksum(&grams),
14603            verdicts: Mutex::new(Vec::new()),
14604        };
14605        let verdict = signatures.verdicts(&file, b"google").expect("signatures read");
14606        assert_eq!(&verdict[..], &[true, false, true], "one verdict a block, at the narrow width");
14607        assert!(signatures.footprint() > 0, "a verdict is remembered");
14608        let again = signatures.verdicts(&file, b"google").expect("remembered");
14609        assert!(Arc::ptr_eq(&verdict, &again), "a second question about a literal reads nothing");
14610
14611        let damaged = NativeGrams {
14612            hash: signatures.hash ^ 1,
14613            verdicts: Mutex::new(Vec::new()),
14614            ..signatures
14615        };
14616        let error = damaged.verdicts(&file, b"google").expect_err("a damaged region is refused");
14617        assert!(error.to_string().contains("substring signatures checksum differs"), "{error}");
14618        fs::remove_file(path).expect("remove scratch file");
14619    }
14620
14621    #[test]
14622    fn a_damaged_substring_signature_is_checked_only_when_used() {
14623        let path = path("damaged-substring-signature");
14624        let mut writer =
14625            Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
14626                .expect("new file");
14627        let rows = [Value::Varchar("google".into()), Value::Varchar("example".into())];
14628        writer
14629            .append(
14630                &Chunk::new(vec![
14631                    Vector::from_values(LogicalType::Varchar, &rows).expect("strings"),
14632                ])
14633                .expect("one column"),
14634            )
14635            .expect("stripe written");
14636        writer.finish().expect("commit");
14637
14638        let reader = Reader::open(&path).expect("valid directory");
14639        let page = reader.table.dictionaries[0].expect("string dictionary page");
14640        let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
14641        file.seek(SeekFrom::Start(page.offset + u64::from(page.length) - 1))
14642            .expect("last signature byte");
14643        file.write_all(&[255]).expect("damage signature");
14644        let reader = Reader::open(&path).expect("the directory is still valid");
14645        let dictionary = reader.dictionary(0).expect("index is still valid").expect("dictionary");
14646        let error = dictionary
14647            .text_block_might_contain(0, b"goog")
14648            .expect_err("a used signature checks its own checksum");
14649        assert!(error.message().contains("substring signatures checksum differs"), "{error}");
14650        fs::remove_file(path).expect("remove scratch file");
14651    }
14652
14653    /// A sweep over a block whose second run of offsets is short reads the same values as a point
14654    /// read does.
14655    ///
14656    /// The sweep decodes the offsets of a whole run at a time rather than a value at a time, and a
14657    /// run holds half a block, so the count it asks for is the run length everywhere but at the end
14658    /// of the dictionary. Two thousand five hundred values, which is what the test above writes,
14659    /// never puts a short run second in its block: the last block there begins on a run boundary and
14660    /// holds one run. Two thousand eight hundred does, so the last block is a whole run of five
14661    /// hundred and twelve followed by two hundred and forty, and an off by one in either the count
14662    /// asked for or the slice taken out of the answer shows up as a wrong value or a refusal.
14663    #[test]
14664    fn a_sweep_over_a_block_with_a_short_second_run_reads_what_a_point_read_reads() {
14665        let path = path("dictionary-sweep-short-run");
14666        let spellings = (0..2_800)
14667            .map(|index| Value::Varchar(format!("value {index:08} {}", "x".repeat(index % 40))))
14668            .collect::<Vec<_>>();
14669        let mut writer =
14670            Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
14671                .expect("new file");
14672        for part in spellings.chunks(1_024) {
14673            writer
14674                .append(
14675                    &Chunk::new(vec![
14676                        Vector::from_values(LogicalType::Varchar, part).expect("strings"),
14677                    ])
14678                    .expect("one column"),
14679                )
14680                .expect("stripe written");
14681        }
14682        writer.finish().expect("commit");
14683
14684        let reader = Reader::open(&path).expect("valid directory");
14685        let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
14686        assert_eq!(dictionary.len(), spellings.len(), "every value is distinct");
14687        let last = dictionary.len() % TEXT_PAYLOAD_VALUES;
14688        assert!(last > TEXT_OFFSET_RUN, "the last block has to reach into a second run of offsets");
14689        assert!(last < TEXT_PAYLOAD_VALUES, "and that second run has to be short of a whole one");
14690
14691        let mut swept: Vec<Vec<u8>> = Vec::new();
14692        let mut at = 0;
14693        while at < dictionary.len() {
14694            let stopped = dictionary
14695                .sweep_text(at, dictionary.len(), &mut |index: usize, text: &[u8]| {
14696                    assert_eq!(index, swept.len(), "a sweep hands its values over in order");
14697                    swept.push(text.to_vec());
14698                    Ok(())
14699                })
14700                .expect("a sweep reads");
14701            assert!(stopped > at, "a sweep moves");
14702            at = stopped;
14703        }
14704        let read = (0..dictionary.len())
14705            .map(|code| dictionary.try_bytes_at(code).expect("read").expect("a value").to_vec())
14706            .collect::<Vec<_>>();
14707        assert_eq!(swept, read, "a sweep answers what a point read answers");
14708        fs::remove_file(path).expect("remove scratch file");
14709    }
14710
14711    /// The unpacked ends answer what the packed ends answer, on both sides of the switch.
14712    ///
14713    /// A column asked for one offset at a time reads them out of the packed form until the reads
14714    /// are worth a table and out of the table after that, so every value here is read twice and the
14715    /// two passes are compared against the spellings and against each other. Two thousand eight
14716    /// hundred values is two payload blocks and a bit, which puts the switch in the middle of the
14717    /// first pass and means the pass straddles a block boundary, where the start of a value is zero
14718    /// rather than the end of the value before it.
14719    #[test]
14720    fn the_unpacked_ends_answer_what_the_packed_ends_answer() {
14721        let path = path("dictionary-unpacked-ends");
14722        let spellings = (0..2_800)
14723            .map(|index| Value::Varchar(format!("value {index:08} {}", "x".repeat(index % 40))))
14724            .collect::<Vec<_>>();
14725        let mut writer =
14726            Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
14727                .expect("new file");
14728        for part in spellings.chunks(1_024) {
14729            writer
14730                .append(
14731                    &Chunk::new(vec![
14732                        Vector::from_values(LogicalType::Varchar, part).expect("strings"),
14733                    ])
14734                    .expect("one column"),
14735                )
14736                .expect("stripe written");
14737        }
14738        writer.finish().expect("commit");
14739
14740        let reader = Reader::open(&path).expect("valid directory");
14741        let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
14742        assert_eq!(dictionary.len(), spellings.len(), "every value is distinct");
14743        let wanted = (0..spellings.len())
14744            .map(|index| format!("value {index:08} {}", "x".repeat(index % 40)).into_bytes())
14745            .collect::<Vec<_>>();
14746
14747        let pass = |what: &str| {
14748            for (index, value) in wanted.iter().enumerate() {
14749                let len = dictionary.try_bytes_len_at(index).expect("read").expect("a value");
14750                assert_eq!(len, value.len(), "{what} has the wrong length at {index}");
14751                let bytes = dictionary.try_bytes_at(index).expect("read").expect("a value");
14752                assert_eq!(bytes, value.as_slice(), "{what} has the wrong value at {index}");
14753            }
14754        };
14755        pass("the first pass");
14756        pass("the second pass");
14757
14758        // The whole vector in one call, over the text and through codes into it, which is how a
14759        // scan of a stored column hands it out. The codes run backwards and repeat so that they are
14760        // neither the positions nor in order.
14761        let lens = wanted.iter().map(|value| value.len() as i64).collect::<Vec<_>>();
14762        let mut whole = vec![0i64; wanted.len()];
14763        assert!(dictionary.try_bytes_lens(&mut whole).expect("read"), "the text answers whole");
14764        assert_eq!(whole, lens, "a vector of lengths answers what a length at a time answers");
14765        let codes = (0..4_000_u32).map(|row| (7 * (4_000 - row)) % 2_800).collect::<Vec<_>>();
14766        let coded = Vector::dictionary_over(codes.clone(), dictionary).expect("codes in range");
14767        let mut through = vec![0i64; codes.len()];
14768        assert!(coded.try_bytes_lens(&mut through).expect("read"), "the codes answer whole");
14769        for (row, &code) in codes.iter().enumerate() {
14770            assert_eq!(through[row], lens[code as usize], "row {row} reads code {code}");
14771            let one = coded.try_bytes_len_at(row).expect("read").expect("a value");
14772            assert_eq!(through[row], one as i64, "row {row} a row at a time");
14773        }
14774
14775        // A handful of codes over a column nobody has read yet is short of the table, so the same
14776        // call answers out of the packed ends instead, and has to answer the same.
14777        let fresh = Reader::open(&path).expect("valid directory");
14778        let untouched = fresh.dictionary(0).expect("read").expect("a string column has one");
14779        let few = vec![2_799_u32, 0, 1_024, 1_023, 511, 512];
14780        let coded = Vector::dictionary_over(few.clone(), untouched).expect("in range");
14781        let mut short = vec![0i64; few.len()];
14782        assert!(coded.try_bytes_lens(&mut short).expect("read"), "the codes answer whole");
14783        let expected = few.iter().map(|&code| lens[code as usize]).collect::<Vec<_>>();
14784        assert_eq!(short, expected, "the packed ends answer what the table answers");
14785        fs::remove_file(path).expect("remove scratch file");
14786    }
14787
14788    /// Narrowing a page takes what fits and refuses the page for anything that does not.
14789    ///
14790    /// The edges of the range on both sides and one step past each of them, for every type, because
14791    /// checking a page separately from converting it is only right if the check refuses exactly what
14792    /// `TryFrom` would have refused, and off by one there is a file that reads back a different
14793    /// number than it was given. The check is a bit pattern rather than a comparison, so it is not
14794    /// the shape a reader would guess from the bounds, which is why all six are here. The empty page
14795    /// is here because a check written the obvious way starts with the extremes the wrong way round
14796    /// and refuses it.
14797    #[test]
14798    fn narrowing_a_page_takes_what_fits_and_refuses_what_does_not() {
14799        assert_eq!(fit::<i8>(&[]).expect("an empty page fits anything"), Vec::<i8>::new());
14800        assert_eq!(fit::<i8>(&[-128, 0, 127]).expect("the edges fit"), vec![-128_i8, 0, 127]);
14801        fit::<i8>(&[128]).expect_err("one past the top does not fit");
14802        fit::<i8>(&[-129]).expect_err("one past the bottom does not fit");
14803        assert_eq!(fit::<u8>(&[0, 255]).expect("the edges fit"), vec![0_u8, 255]);
14804        fit::<u8>(&[256]).expect_err("one past the top does not fit");
14805        fit::<u8>(&[-1]).expect_err("a negative does not fit an unsigned page");
14806        assert_eq!(
14807            fit::<i16>(&[-32_768, 0, 32_767]).expect("the edges fit"),
14808            vec![-32_768_i16, 0, 32_767]
14809        );
14810        fit::<i16>(&[32_768]).expect_err("one past the top does not fit");
14811        fit::<i16>(&[-32_769]).expect_err("one past the bottom does not fit");
14812        assert_eq!(fit::<u16>(&[0, 65_535]).expect("the edges fit"), vec![0_u16, 65_535]);
14813        fit::<u16>(&[65_536]).expect_err("one past the top does not fit");
14814        fit::<u16>(&[-1]).expect_err("a negative does not fit an unsigned page");
14815        assert_eq!(
14816            fit::<i32>(&[i64::from(i32::MIN), 0, i64::from(i32::MAX)]).expect("the edges fit"),
14817            vec![i32::MIN, 0, i32::MAX]
14818        );
14819        fit::<i32>(&[i64::from(i32::MAX) + 1]).expect_err("one past the top does not fit");
14820        fit::<i32>(&[i64::from(i32::MIN) - 1]).expect_err("one past the bottom does not fit");
14821        assert_eq!(
14822            fit::<u32>(&[0, 4_294_967_295]).expect("the edges fit"),
14823            vec![0_u32, 4_294_967_295]
14824        );
14825        fit::<u32>(&[4_294_967_296]).expect_err("one past the top does not fit");
14826        fit::<u32>(&[-1]).expect_err("a negative does not fit an unsigned page");
14827
14828        // One value in a page that fits is still a page that does not, which is the thing an or
14829        // into an accumulator could get wrong in a way a page of one value would never show.
14830        fit::<i8>(&[0, 1, 2, 128, 3]).expect_err("one bad value spoils the page");
14831    }
14832
14833    /// The residue says yes to exactly what `TryFrom` says yes to.
14834    ///
14835    /// The edges above are the cases anyone would think to write down. This is the argument that
14836    /// there are no others, made by asking both questions about every value either narrow type could
14837    /// have an opinion about, and then about the values around the wide edges and the ends of an
14838    /// `i64`, which a range that size cannot reach.
14839    #[test]
14840    fn the_residue_agrees_with_a_checked_conversion_everywhere() {
14841        for value in -70_000_i64..70_000 {
14842            assert_eq!(fit::<i8>(&[value]).is_ok(), i8::try_from(value).is_ok(), "{value} as i8");
14843            assert_eq!(fit::<u8>(&[value]).is_ok(), u8::try_from(value).is_ok(), "{value} as u8");
14844            assert_eq!(fit::<i16>(&[value]).is_ok(), i16::try_from(value).is_ok(), "{value} i16");
14845            assert_eq!(fit::<u16>(&[value]).is_ok(), u16::try_from(value).is_ok(), "{value} u16");
14846        }
14847        let wide = [i64::MIN, i64::MIN + 1, i64::from(i32::MIN), 0, i64::from(u32::MAX), i64::MAX];
14848        for edge in wide {
14849            for step in -2_i64..=2 {
14850                let value = edge.saturating_add(step);
14851                assert_eq!(
14852                    fit::<i32>(&[value]).is_ok(),
14853                    i32::try_from(value).is_ok(),
14854                    "{value} as i32"
14855                );
14856                assert_eq!(
14857                    fit::<u32>(&[value]).is_ok(),
14858                    u32::try_from(value).is_ok(),
14859                    "{value} as u32"
14860                );
14861            }
14862        }
14863    }
14864
14865    /// All three block layouts come back as the same values in the same order.
14866    ///
14867    /// Blocks outside the page are what every file this build writes holds. Blocks that say where
14868    /// they are but sit inside the page behind the order are format 26, and blocks behind one
14869    /// another with only their ends recorded are older still. Nothing in the writer produces the
14870    /// last two any more, so the only way to find out whether the reader still understands those
14871    /// files is to write them here. The
14872    /// bytes go straight into a file with no directory around them, because what is under test is
14873    /// [`open_global_dictionary`], which is handed a page and a file and asks the directory for
14874    /// nothing.
14875    ///
14876    /// Three thousand values so that there are three payload blocks and a partial fourth, which is
14877    /// what makes the last block the one place where a length and an end disagree about what they
14878    /// are counting.
14879    #[test]
14880    fn a_dictionary_reads_the_same_whether_its_blocks_say_where_they_are() {
14881        let spellings = (0..3_000)
14882            .map(|index| format!("value {index:08} {}", "y".repeat(index % 40)))
14883            .collect::<Vec<_>>();
14884        let mut read = Vec::new();
14885        for layout in ["outside", "inside", "behind"] {
14886            let mut dictionary = GlobalDictionary::new();
14887            for text in &spellings {
14888                dictionary.code(text).expect("a code for every spelling");
14889            }
14890            dictionary.finish_blocks().expect("the last block encodes");
14891            let order = dictionary.ranked(None).expect("a sorted order");
14892            // Where the blocks go if they start at `from` and follow one another.
14893            let laid = |from: u64| {
14894                let mut at = from;
14895                dictionary
14896                    .blocks
14897                    .iter()
14898                    .map(|block| {
14899                        let place =
14900                            Placed { start: at, length: block.len() as u64, hash: checksum(block) };
14901                        at += block.len() as u64;
14902                        place
14903                    })
14904                    .collect::<Vec<_>>()
14905            };
14906            let payload = dictionary.blocks.concat();
14907            let scattered = layout != "behind";
14908            let (bytes, encoded, offset, length) = if layout == "outside" {
14909                let mut bytes = vec![0; HEADER as usize];
14910                bytes.extend_from_slice(&payload);
14911                let encoded = encode_global_dictionary(&dictionary, &order, &laid(HEADER), true)
14912                    .expect("an encoding");
14913                let offset = bytes.len() as u64;
14914                bytes.extend_from_slice(&encoded.index);
14915                bytes.extend_from_slice(&encoded.ranks);
14916                bytes.extend_from_slice(&encoded.grams);
14917                let length = encoded.index.len() + encoded.ranks.len() + encoded.grams.len();
14918                (bytes, encoded, offset, length)
14919            } else {
14920                // The index is the same length wherever the blocks are, so a first pass says where
14921                // the page ends and the second writes the places that follow it.
14922                let first = encode_global_dictionary(&dictionary, &order, &laid(0), scattered)
14923                    .expect("an encoding");
14924                let body = (first.index.len() + first.ranks.len() + first.grams.len()) as u64;
14925                let encoded = encode_global_dictionary(&dictionary, &order, &laid(body), scattered)
14926                    .expect("an encoding");
14927                let mut bytes = encoded.index.clone();
14928                bytes.extend_from_slice(&encoded.ranks);
14929                bytes.extend_from_slice(&encoded.grams);
14930                bytes.extend_from_slice(&payload);
14931                let length = bytes.len();
14932                (bytes, encoded, 0, length)
14933            };
14934            let path = path(&format!("blocks-{layout}"));
14935            fs::write(&path, &bytes).expect("the dictionary is written on its own");
14936            let file = Arc::new(File::open(&path).expect("it opens again"));
14937            let page = Page {
14938                offset,
14939                length: u32::try_from(length).expect("a test dictionary is small"),
14940                hash: checksum(&encoded.index),
14941            };
14942            let opened =
14943                open_global_dictionary(file, page, &LogicalType::Varchar, TEXT_KEEP_BUDGET)
14944                    .expect("a dictionary laid out either way opens");
14945            let mut swept: Vec<Vec<u8>> = Vec::new();
14946            let mut at = 0;
14947            while at < opened.len() {
14948                at = opened
14949                    .sweep_text(at, opened.len(), &mut |_index: usize, text: &[u8]| {
14950                        swept.push(text.to_vec());
14951                        Ok(())
14952                    })
14953                    .expect("a sweep reads");
14954            }
14955            fs::remove_file(&path).expect("clean up");
14956            read.push(swept);
14957        }
14958        let wanted =
14959            spellings.iter().map(|text| text.as_bytes().to_vec()).collect::<Vec<Vec<u8>>>();
14960        assert_eq!(read[0], wanted, "the blocks outside the page hold the values");
14961        assert_eq!(read[1], read[0], "the blocks inside the page hold the same values");
14962        assert_eq!(read[2], read[0], "the blocks behind one another hold the same values");
14963    }
14964
14965    /// A dictionary at its budget sweeps without keeping, and still answers what it answered.
14966    ///
14967    /// The budget is a quarter of a gigabyte in a running database, which is a fine size for a real
14968    /// column and no size at all for a test, so this opens the same dictionary a second time with a
14969    /// budget of zero. That is the shape of the hundred million row case: `URL` fills the budget
14970    /// somewhere in the middle of itself and everything past that point is read and dropped, which
14971    /// costs the decode again and holds none of it.
14972    #[test]
14973    fn a_dictionary_at_its_budget_sweeps_without_keeping() {
14974        let path = path("dictionary-budget");
14975        let spellings = (0..2_500)
14976            .map(|index| Value::Varchar(format!("value {index:08} {}", "y".repeat(index % 40))))
14977            .collect::<Vec<_>>();
14978        let mut writer =
14979            Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
14980                .expect("new file");
14981        for part in spellings.chunks(1_024) {
14982            writer
14983                .append(
14984                    &Chunk::new(vec![
14985                        Vector::from_values(LogicalType::Varchar, part).expect("strings"),
14986                    ])
14987                    .expect("one column"),
14988                )
14989                .expect("stripe written");
14990        }
14991        writer.finish().expect("commit");
14992
14993        let reader = Reader::open(&path).expect("valid directory");
14994        let page = reader.table.dictionaries[0].expect("a string column has one");
14995        let file = Arc::clone(&reader.file);
14996        let starved = open_global_dictionary(file, page, &LogicalType::Varchar, 0)
14997            .expect("a dictionary opens whatever it may keep");
14998
14999        let resting = starved.footprint();
15000        let mut swept: Vec<Vec<u8>> = Vec::new();
15001        let mut at = 0;
15002        while at < starved.len() {
15003            at = starved
15004                .sweep_text(at, starved.len(), &mut |_index: usize, text: &[u8]| {
15005                    swept.push(text.to_vec());
15006                    Ok(())
15007                })
15008                .expect("a sweep reads");
15009        }
15010        assert_eq!(swept.len(), spellings.len(), "a starved sweep still reads every value");
15011        assert_eq!(starved.footprint(), resting, "and keeps no block it decoded");
15012
15013        let generous = reader.dictionary(0).expect("read").expect("a string column has one");
15014        let read = (0..generous.len())
15015            .map(|code| generous.try_bytes_at(code).expect("read").expect("a value").to_vec())
15016            .collect::<Vec<_>>();
15017        assert_eq!(swept, read, "a starved sweep answers what a point read answers");
15018        fs::remove_file(path).expect("remove scratch file");
15019    }
15020
15021    #[test]
15022    fn damaged_membership_cannot_skip_a_string_page() {
15023        let path = path("damaged-membership");
15024        let mut writer = Writer::create(
15025            &path,
15026            "items",
15027            vec![
15028                Field::required("id", LogicalType::Integer),
15029                Field::new("text", LogicalType::Varchar),
15030            ],
15031        )
15032        .expect("new file");
15033        writer.append(&sample()).expect("stripe written");
15034        writer.finish().expect("commit");
15035
15036        let reader = Reader::open(&path).expect("valid directory");
15037        let membership = reader.table.stripes[0].memberships.get(1).expect("string membership");
15038        let mut file = OpenOptions::new().write(true).open(&path).expect("open membership page");
15039        file.seek(SeekFrom::Start(membership.offset)).expect("membership start");
15040        file.write_all(&[255]).expect("damage membership");
15041        let error = reader.skips_codes(0, 1, &[3]).expect_err("corruption must not skip rows");
15042        assert!(error.message().contains("membership page checksum differs"), "{error}");
15043        fs::remove_file(path).expect("remove scratch file");
15044    }
15045
15046    #[test]
15047    fn membership_delta_stream_is_sorted_exact_and_bounded() {
15048        let unique = unique_codes(&[900, 4, 4, 72, 9, u32::MAX]);
15049        assert_eq!(unique, [4, 9, 72, 900, u32::MAX]);
15050        let encoded = encode_membership(&unique);
15051        assert_eq!(
15052            decode_membership(&encoded).expect("valid membership"),
15053            [4, 9, 72, 900, u32::MAX]
15054        );
15055        // A stripe's index is the union of its parts', so a code in two of them is in it once and
15056        // the result is still one ascending run of deltas.
15057        let merged = merged_codes(vec![vec![4, 900], vec![9, 900, u32::MAX], vec![72]]);
15058        assert_eq!(merged, [4, 9, 72, 900, u32::MAX]);
15059        assert_eq!(
15060            decode_membership(&encode_membership(&merged)).expect("valid membership"),
15061            unique
15062        );
15063        assert!(decode_membership(&[1, 0x80]).is_err(), "a truncated varint is invalid");
15064        assert!(
15065            decode_membership(&[1, 0xff, 0xff, 0xff, 0xff, 0x10]).is_err(),
15066            "a value past u32 is invalid"
15067        );
15068    }
15069
15070    #[test]
15071    fn a_global_dictionary_may_be_larger_than_one_column_page() {
15072        let dictionary = Page {
15073            offset: HEADER,
15074            length: u32::try_from(MAX_PAGE + 1).expect("the page bound fits on disk"),
15075            hash: 0,
15076        };
15077        let table = Table {
15078            name: "items".to_owned(),
15079            fields: vec![Field::new("text", LogicalType::Varchar)],
15080            stripes: Vec::new(),
15081            rows: 0,
15082            dictionaries: vec![Some(dictionary)],
15083            dictionary_payloads: Vec::new(),
15084            distincts: vec![None],
15085            frequencies: vec![None],
15086            pair_frequencies: Vec::new(),
15087            frequency_texts: Vec::new(),
15088            host_groups: None,
15089            clustering: None,
15090            generation: 1,
15091            sections: Vec::new(),
15092        };
15093        let directory = encode_directory(&table).expect("directory");
15094        let file_size = dictionary.offset + u64::from(dictionary.length) + 1;
15095
15096        let decoded = decode_directory(&directory, file_size).expect("large lazy dictionary");
15097        assert_eq!(decoded.dictionaries[0].expect("dictionary").length, dictionary.length);
15098    }
15099
15100    #[test]
15101    fn a_column_with_one_value_everywhere_costs_almost_nothing_a_row() {
15102        let path = path("constant-codes");
15103        let mut writer =
15104            Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
15105                .expect("new file");
15106        let empty = vec![Value::Varchar(String::new()); 1024];
15107        for _ in 0..4 {
15108            let column = Vector::from_values(LogicalType::Varchar, &empty).expect("strings");
15109            writer.append(&Chunk::new(vec![column]).expect("one column")).expect("a part");
15110        }
15111        writer.finish().expect("commit");
15112
15113        let reader = Reader::open(&path).expect("valid directory");
15114        let pages = reader.layout().columns.first().expect("one column").pages;
15115        // This column used to cost four bytes a row, 16,384 of them, the same as a column of four
15116        // thousand distinct URLs would. The cascade calls each part a constant, so what is left is
15117        // a tag, a count and the value, and the row count stops being what drives the number.
15118        assert!(pages < 256, "{pages} bytes of pages for 4,096 rows of one value");
15119        let read = reader.read(3, &[0]).expect("the last part back");
15120        assert_eq!(read.value_at(0, 0), Value::Varchar(String::new()));
15121        assert_eq!(read.value_at(1023, 0), Value::Varchar(String::new()));
15122        fs::remove_file(path).expect("remove scratch file");
15123    }
15124
15125    #[test]
15126    fn a_cascade_value_too_wide_for_its_column_is_refused_rather_than_cut() {
15127        // What a damaged page looks like from here: the cascade decoded, so the bytes are not
15128        // truncated, but the values do not belong to the column the directory says they do.
15129        let over = vec![i64::from(i32::MAX) + 1];
15130        let error = narrowed(&LogicalType::Integer, over).expect_err("a page that disagrees");
15131        assert!(format!("{error}").contains("not of its type"), "{error}");
15132        assert!(narrowed(&LogicalType::BigInt, vec![i64::MIN]).is_ok(), "bigint holds all of i64");
15133        assert!(narrowed(&LogicalType::Varchar, vec![0]).is_err(), "strings are not integers");
15134    }
15135
15136    #[test]
15137    fn a_code_stream_the_cascade_cannot_shrink_is_left_alone() {
15138        // A shift register rather than a run, because an arithmetic run is the one wide shape the
15139        // cascade does shrink. This is what a column with tens of millions of distinct values hands
15140        // over: full width codes with no order to them.
15141        let mut state: u32 = 0x9e37_79b9;
15142        let spread: Vec<u32> = (0..1024)
15143            .map(|_| {
15144                state ^= state << 13;
15145                state ^= state >> 17;
15146                state ^= state << 5;
15147                state
15148            })
15149            .collect();
15150        assert_eq!(encoded_codes(&spread).expect("no failure"), None);
15151        let near: Vec<u32> = (0..1024).collect();
15152        let coded = encoded_codes(&near).expect("no failure").expect("counting up is packable");
15153        assert!(coded.len() < near.len() * 4, "{} bytes for a run of 1,024", coded.len());
15154    }
15155
15156    /// The columns of a stripe are encoded on whichever thread got to them, so the one thing that
15157    /// must not depend on which thread that was is the file. Two writes of the same rows are
15158    /// compared byte for byte rather than value for value, because a dictionary that two columns
15159    /// somehow shared would still read back correctly and would hand out its codes in the order the
15160    /// threads happened to run in, which is exactly what this is here to catch.
15161    #[test]
15162    fn two_writes_of_the_same_rows_give_the_same_bytes() {
15163        fn written(path: &PathBuf) {
15164            let fields = (0..40)
15165                .map(|column| {
15166                    let ty =
15167                        if column % 4 == 0 { LogicalType::Varchar } else { LogicalType::BigInt };
15168                    Field::new(format!("c{column}"), ty)
15169                })
15170                .collect::<Vec<_>>();
15171            let mut writer = Writer::create(path, "wide", fields).expect("new file");
15172            for part in 0..70_u64 {
15173                let columns = (0..40)
15174                    .map(|column| {
15175                        let values = (0..64_u64)
15176                            .map(|row| {
15177                                let seed = part.wrapping_mul(31).wrapping_add(row);
15178                                if column % 4 == 0 {
15179                                    Value::Varchar(format!("v{}", seed % 17))
15180                                } else {
15181                                    Value::BigInt(i64::try_from(seed % 97).expect("small"))
15182                                }
15183                            })
15184                            .collect::<Vec<_>>();
15185                        let ty = if column % 4 == 0 {
15186                            LogicalType::Varchar
15187                        } else {
15188                            LogicalType::BigInt
15189                        };
15190                        Vector::from_values(ty, &values).expect("a column")
15191                    })
15192                    .collect::<Vec<_>>();
15193                writer.append(&Chunk::new(columns).expect("forty columns")).expect("a part");
15194            }
15195            writer.finish().expect("commit");
15196        }
15197
15198        let first = path("repeatable-one");
15199        let second = path("repeatable-two");
15200        written(&first);
15201        written(&second);
15202        let left = fs::read(&first).expect("the first file");
15203        let right = fs::read(&second).expect("the second file");
15204        assert_eq!(left.len(), right.len(), "two writes of the same rows differ in length");
15205        assert!(left == right, "two writes of the same rows differ in their bytes");
15206
15207        // And the rows are still there, since a pair of identically wrong files would pass the
15208        // comparison above on its own.
15209        let reader = Reader::open(&first).expect("valid directory");
15210        assert_eq!(reader.table().rows(), 70 * 64);
15211        let read = reader.read(0, &[0, 1]).expect("the first part back");
15212        assert_eq!(read.value_at(0, 0), Value::Varchar("v0".to_owned()));
15213        assert_eq!(read.value_at(0, 1), Value::BigInt(0));
15214        fs::remove_file(first).expect("remove scratch file");
15215        fs::remove_file(second).expect("remove scratch file");
15216    }
15217
15218    /// Three tables of different shapes in one file, read back by name.
15219    fn three_tables(path: &PathBuf) {
15220        let writer = Writer::create(
15221            path,
15222            "region",
15223            vec![
15224                Field::new("r_key", LogicalType::Integer),
15225                Field::new("r_name", LogicalType::Varchar),
15226            ],
15227        )
15228        .expect("new file");
15229        let mut writer = writer;
15230        writer
15231            .append(
15232                &Chunk::new(vec![
15233                    Vector::from_values(
15234                        LogicalType::Integer,
15235                        &[Value::Integer(0), Value::Integer(1)],
15236                    )
15237                    .expect("keys"),
15238                    Vector::from_values(
15239                        LogicalType::Varchar,
15240                        &[Value::Varchar("AFRICA".to_owned()), Value::Varchar("ASIA".to_owned())],
15241                    )
15242                    .expect("names"),
15243                ])
15244                .expect("two columns"),
15245            )
15246            .expect("a part");
15247        let mut writer = writer
15248            .next("empty", vec![Field::new("nothing", LogicalType::BigInt)])
15249            .expect("a second table");
15250        writer
15251            .append(
15252                &Chunk::new(vec![
15253                    Vector::from_values(LogicalType::BigInt, &[Value::BigInt(7)]).expect("a row"),
15254                ])
15255                .expect("one column"),
15256            )
15257            .expect("a part");
15258        let mut writer =
15259            writer.next("wide", vec![Field::new("n", LogicalType::BigInt)]).expect("a third table");
15260        for part in 0..70_i64 {
15261            let values = (0..64).map(|row| Value::BigInt(part * 64 + row)).collect::<Vec<_>>();
15262            writer
15263                .append(
15264                    &Chunk::new(vec![
15265                        Vector::from_values(LogicalType::BigInt, &values).expect("a column"),
15266                    ])
15267                    .expect("one column"),
15268                )
15269                .expect("a part");
15270        }
15271        writer.finish().expect("commit");
15272    }
15273
15274    #[test]
15275    fn three_tables_in_one_file_read_back_by_name() {
15276        let file = path("three-tables");
15277        three_tables(&file);
15278        let catalog = Catalog::open(&file).expect("a committed catalog");
15279        assert_eq!(catalog.names().collect::<Vec<_>>(), ["region", "empty", "wide"]);
15280
15281        let region = catalog.table("region").expect("the first table");
15282        assert_eq!(region.table().rows(), 2);
15283        assert_eq!(
15284            region.read(0, &[1]).expect("names").value_at(1, 0),
15285            Value::Varchar("ASIA".to_owned())
15286        );
15287
15288        let wide = catalog.table("wide").expect("the third table");
15289        assert_eq!(wide.table().rows(), 70 * 64);
15290        assert_eq!(wide.read(0, &[0]).expect("the first part").value_at(0, 0), Value::BigInt(0));
15291
15292        // The middle table is reached without the one after it having been touched, which is what
15293        // a directory per table buys over one directory of everything.
15294        let empty = catalog.table("empty").expect("the second table");
15295        assert_eq!(empty.table().rows(), 1);
15296        assert_eq!(empty.read(0, &[0]).expect("the row").value_at(0, 0), Value::BigInt(7));
15297
15298        fs::remove_file(file).expect("remove scratch file");
15299    }
15300
15301    #[test]
15302    fn a_name_the_file_does_not_hold_is_an_error_rather_than_the_first_table() {
15303        let file = path("three-tables-missing");
15304        three_tables(&file);
15305        let catalog = Catalog::open(&file).expect("a committed catalog");
15306        let error = catalog.table("nation").expect_err("no such table");
15307        assert!(error.message().contains("nation"), "{}", error.message());
15308        fs::remove_file(file).expect("remove scratch file");
15309    }
15310
15311    #[test]
15312    fn a_file_of_three_tables_will_not_open_as_one() {
15313        let file = path("three-tables-unnamed");
15314        three_tables(&file);
15315        let error = Reader::open(&file).expect_err("more than one table");
15316        assert!(error.message().contains("more than one table"), "{}", error.message());
15317        fs::remove_file(file).expect("remove scratch file");
15318    }
15319
15320    /// One column per storage width, because the width is what decides how many bytes a row costs.
15321    #[test]
15322    fn decimals_of_every_storage_width_round_trip() {
15323        let file = path("decimals");
15324        let widths = [(4_u8, 2_u8), (9, 2), (18, 4), (38, 6)];
15325        let fields = widths
15326            .iter()
15327            .enumerate()
15328            .map(|(index, (width, scale))| {
15329                Field::new(
15330                    format!("d{index}"),
15331                    LogicalType::decimal(*width, *scale).expect("a decimal type"),
15332                )
15333            })
15334            .collect::<Vec<_>>();
15335        let mut writer = Writer::create(&file, "money", fields).expect("new file");
15336        let rows: [i128; 3] = [-1234, 0, 999];
15337        let columns = widths
15338            .iter()
15339            .map(|(width, scale)| {
15340                let values = rows
15341                    .iter()
15342                    .map(|unscaled| Value::Decimal {
15343                        unscaled: *unscaled,
15344                        width: *width,
15345                        scale: *scale,
15346                    })
15347                    .collect::<Vec<_>>();
15348                Vector::from_values(
15349                    LogicalType::decimal(*width, *scale).expect("a decimal type"),
15350                    &values,
15351                )
15352                .expect("a decimal column")
15353            })
15354            .collect::<Vec<_>>();
15355        writer.append(&Chunk::new(columns).expect("four columns")).expect("a part");
15356        writer.finish().expect("commit");
15357
15358        let reader = Reader::open(&file).expect("a committed file");
15359        for (index, (width, scale)) in widths.iter().enumerate() {
15360            assert_eq!(
15361                reader.table().fields()[index].ty,
15362                LogicalType::decimal(*width, *scale).expect("a decimal type"),
15363                "column {index} came back as another type"
15364            );
15365            let column = reader.read(0, &[index]).expect("the column");
15366            for (row, unscaled) in rows.iter().enumerate() {
15367                assert_eq!(
15368                    column.value_at(row, 0),
15369                    Value::Decimal { unscaled: *unscaled, width: *width, scale: *scale },
15370                    "column {index} row {row}"
15371                );
15372            }
15373        }
15374        fs::remove_file(file).expect("remove scratch file");
15375    }
15376
15377    #[test]
15378    fn two_tables_of_one_name_are_refused_before_anything_is_committed() {
15379        let file = path("two-of-a-name");
15380        let writer = Writer::create(&file, "t", vec![Field::new("a", LogicalType::BigInt)])
15381            .expect("new file");
15382        let error = writer
15383            .next("t", vec![Field::new("a", LogicalType::BigInt)])
15384            .expect_err("the same name twice");
15385        assert!(error.message().contains("same name"), "{}", error.message());
15386        fs::remove_file(file).expect("remove scratch file");
15387    }
15388
15389    #[test]
15390    fn opening_the_catalog_reads_no_table_directory() {
15391        let file = path("catalog-only");
15392        three_tables(&file);
15393        let catalog = Catalog::open(&file).expect("a committed catalog");
15394        // The header and one slot, and nothing under it. The third table's directory covers seventy
15395        // stripes and reading it here would be the whole point of the two levels thrown away.
15396        assert_eq!(catalog.opening.reads, 2, "opening the catalog read more than the slot");
15397        assert_eq!(catalog.names().len(), 3);
15398        fs::remove_file(file).expect("remove scratch file");
15399    }
15400
15401    /// The checksum answers what it has always answered, at every length its branches split on.
15402    ///
15403    /// This is a compatibility test rather than a correctness one. Nothing about the hash has to be
15404    /// any particular function, but a file already on disk carries the answers the version that
15405    /// wrote it gave, so a change here is a change that makes every stored file fail to verify. The
15406    /// lengths are the ones the code makes decisions about: nothing, under a block, a block exactly,
15407    /// a block and a word, a word and a half word, and a half word and a byte.
15408    ///
15409    /// The empty answer is the published xxHash64 vector for an empty input at seed zero, which is
15410    /// also a check that this is the function it says it is.
15411    #[test]
15412    fn the_checksum_answers_what_it_has_always_answered() {
15413        let bytes: Vec<u8> =
15414            (0..1000_u32).map(|at| (at.wrapping_mul(31).wrapping_add(7) % 251) as u8).collect();
15415        for (length, expected) in [
15416            (0, 0xef46_db37_51d8_e999),
15417            (1, 0xa96c_7f0c_e858_bbb7),
15418            (3, 0x56e6_9576_32a4_87f9),
15419            (4, 0xc60d_15b1_e3ff_8f04),
15420            (5, 0x8088_1585_8624_dd4e),
15421            (7, 0xafbe_fc3d_6c6f_9a8e),
15422            (8, 0x3da5_c7aa_2696_83e0),
15423            (9, 0x465e_c429_b13c_3892),
15424            (15, 0xdee8_9d8a_065a_6233),
15425            (16, 0x1330_489a_7767_9c80),
15426            (31, 0x3391_303d_485e_846e),
15427            (32, 0x40b7_aff7_5d45_bbc8),
15428            (33, 0x4997_cae4_951c_17a5),
15429            (39, 0x5807_28fd_5c14_5739),
15430            (40, 0xf95c_f6f5_c08a_3d3b),
15431            (63, 0x2944_b4da_fc69_b206),
15432            (64, 0xbb76_f6ef_19bd_5a1b),
15433            (65, 0x814e_0c65_4a9f_d640),
15434            (127, 0x00de_aab1_31cf_f89b),
15435            (1000, 0x9e33_00c1_cde3_c58d),
15436        ] {
15437            assert_eq!(checksum(&bytes[..length]), expected, "the checksum of {length} bytes");
15438        }
15439        assert_eq!(checksum(b"the quick brown fox jumps over the lazy dog"), 0xed71_4233_c5a9_a792);
15440    }
15441    /// A declared order survives the file, and a table that declared none stays as it was.
15442    ///
15443    /// The second half is the one worth a test. The clustering section is written only when there
15444    /// is a declaration, so a file of two tables where one is clustered exercises both the present
15445    /// and the absent branch of the decoder in one directory, which is where a length bug would
15446    /// show up as one table reading the other's bytes.
15447    #[test]
15448    fn a_declared_order_comes_back_out_of_the_file() {
15449        let path = path("clustered");
15450        let shipped = vec![
15451            Field::new("key", LogicalType::BigInt),
15452            Field::new("line", LogicalType::Integer),
15453            Field::new("shipdate", LogicalType::Date),
15454        ];
15455        let plain = vec![Field::new("a", LogicalType::Integer)];
15456        let stage_zero = Clustering::new(vec![2, 0, 1], Width::Month, &shipped).expect("valid");
15457
15458        let mut writer = Writer::create(&path, "lineitem", shipped)
15459            .expect("new file")
15460            .declare(stage_zero.clone())
15461            .expect("the columns are the table's");
15462        let column = |ty: LogicalType, values: &[Value]| {
15463            Vector::from_values(ty, values).expect("the values match the type")
15464        };
15465        writer
15466            .append(
15467                &Chunk::new(vec![
15468                    column(
15469                        LogicalType::BigInt,
15470                        &[Value::BigInt(0), Value::BigInt(1), Value::BigInt(2), Value::BigInt(3)],
15471                    ),
15472                    column(
15473                        LogicalType::Integer,
15474                        &[
15475                            Value::Integer(1),
15476                            Value::Integer(1),
15477                            Value::Integer(1),
15478                            Value::Integer(1),
15479                        ],
15480                    ),
15481                    column(
15482                        LogicalType::Date,
15483                        &[Value::Date(0), Value::Date(1), Value::Date(2), Value::Date(3)],
15484                    ),
15485                ])
15486                .expect("three columns"),
15487            )
15488            .expect("four rows");
15489        let mut writer = writer.next("nation", plain).expect("a second table");
15490        writer
15491            .append(
15492                &Chunk::new(vec![column(LogicalType::Integer, &[Value::Integer(7)])])
15493                    .expect("one column"),
15494            )
15495            .expect("one row");
15496        writer.finish().expect("commit");
15497
15498        let catalog = Catalog::open(&path).expect("reopen");
15499        let lineitem = catalog.table("lineitem").expect("the clustered table");
15500        assert_eq!(lineitem.table().clustering(), Some(&stage_zero));
15501        let nation = catalog.table("nation").expect("the plain table");
15502        assert_eq!(nation.table().clustering(), None, "nobody declared one here");
15503
15504        // And the rows are still the rows, because the section goes on the end of the directory
15505        // and the easy way to break that is to leave the cursor somewhere the next read trusts.
15506        assert_eq!(lineitem.table().rows(), 4);
15507        assert_eq!(nation.table().rows(), 1);
15508        fs::remove_file(&path).ok();
15509    }
15510
15511    /// A declaration naming a column the table does not have is refused where it is made.
15512    #[test]
15513    fn a_declaration_off_the_end_of_the_table_never_reaches_the_file() {
15514        let path = path("clustered-bad");
15515        let writer = Writer::create(&path, "items", vec![Field::new("a", LogicalType::Integer)])
15516            .expect("new file");
15517        let four =
15518            (0..4).map(|at| Field::new(format!("c{at}"), LogicalType::Integer)).collect::<Vec<_>>();
15519        let wrong = Clustering::new(vec![3], Width::Exact, &four).expect("valid against four");
15520        assert!(writer.declare(wrong).is_err(), "the table has one column, not four");
15521        fs::remove_file(&path).ok();
15522    }
15523
15524    /// The sorted order is the byte order, whatever the values do before they differ.
15525    ///
15526    /// A block handed out of the writer's lock to be encoded, and given back in whatever order the
15527    /// stripes happen to finish in, is the same block with the same signature as one encoded in
15528    /// place, and lands in the same position.
15529    #[test]
15530    fn blocks_handed_out_and_given_back_out_of_order_are_the_blocks_encoded_in_place() {
15531        let values = (0..PAYLOAD_SAMPLE_BLOCKS * TEXT_PAYLOAD_VALUES * 2 + 100)
15532            .map(|at| format!("http://example{}.test/page/{at:06}", at % 7))
15533            .collect::<Vec<_>>();
15534        let filled = || {
15535            let mut dictionary = GlobalDictionary::new();
15536            for value in &values {
15537                dictionary.code(value).expect("a code for every value");
15538            }
15539            dictionary.settle().expect("a shape");
15540            dictionary
15541        };
15542        let mut in_place = filled();
15543        in_place.finish_blocks().expect("every block encodes");
15544
15545        let mut handed = filled();
15546        let out = handed.hand_out(3);
15547        assert_eq!(out.len(), PAYLOAD_SAMPLE_BLOCKS * 2, "every sealed block goes out");
15548        assert!(handed.waiting.is_empty(), "and none is left to be encoded under the lock");
15549        for job in out.iter().rev() {
15550            assert_eq!(job.place().0, 3, "a block goes back to the column it came from");
15551            handed.take_back(job.place().1, job.encode().expect("encodes")).expect("taken back");
15552        }
15553        assert!(handed.early.is_empty(), "nothing is waiting on a gap");
15554        handed.finish_blocks().expect("the last block encodes");
15555
15556        assert_eq!(handed.blocks, in_place.blocks, "the same blocks in the same order");
15557        assert_eq!(handed.grams, in_place.grams, "with the same signatures");
15558    }
15559
15560    /// A block given back twice is a bug in whoever gave it, and is said rather than written twice.
15561    #[test]
15562    fn a_block_given_back_twice_is_refused() {
15563        let mut dictionary = GlobalDictionary::new();
15564        for at in 0..PAYLOAD_SAMPLE_BLOCKS * TEXT_PAYLOAD_VALUES {
15565            dictionary.code(&format!("value {at}")).expect("a code");
15566        }
15567        dictionary.settle().expect("a shape");
15568        let out = dictionary.hand_out(0);
15569        let last = out.last().expect("blocks went out");
15570        let at = last.place().1;
15571        dictionary.take_back(at, last.encode().expect("encodes")).expect("taken back once");
15572        assert!(dictionary.take_back(at, last.encode().expect("encodes")).is_err());
15573    }
15574
15575    /// The values here are the shape the sort is built for and the shape a comparison sort is worst
15576    /// at: a common scheme, a handful of hosts, and a path that only decides the pair thirty bytes
15577    /// in. They also cover what the bucketing has to get right at the edges, which is a value that
15578    /// has run out where another carries on, the empty value, and enough entries to take the range
15579    /// down through several passes and out the bottom into the comparison that finishes it.
15580    #[test]
15581    fn the_dictionary_order_is_the_byte_order_however_deep_the_values_agree() {
15582        let mut values = vec![String::new(), "http://".to_owned()];
15583        for host in 0..7 {
15584            for path in 0..30 {
15585                values.push(format!("http://example{host}.test/page/{path:04}/index.html"));
15586                values.push(format!("http://example{host}.test/page/{path:04}"));
15587            }
15588        }
15589        values.push("http://example0.test/page/0000/index.htmlx".to_owned());
15590
15591        let mut dictionary = GlobalDictionary::new();
15592        for value in &values {
15593            dictionary.code(value).expect("a code for every value");
15594        }
15595        dictionary.finish_blocks().expect("the last block encodes");
15596        let ranked = dictionary.ranked(None).expect("a sorted order");
15597        assert_eq!(ranked.len(), values.len(), "one entry a distinct value");
15598
15599        let spellings = dictionary_values(&dictionary);
15600        let seen = ranked
15601            .iter()
15602            .map(|&(_, code)| {
15603                String::from_utf8(spellings[code as usize].clone()).expect("text in, text out")
15604            })
15605            .collect::<Vec<_>>();
15606        let mut wanted = values.clone();
15607        wanted.sort_unstable();
15608        assert_eq!(seen, wanted, "the order is the order the bytes give");
15609
15610        for &(carried, code) in &ranked {
15611            let value = &spellings[code as usize];
15612            assert_eq!(carried, head(value), "the head belongs to the value it is filed with");
15613        }
15614    }
15615
15616    /// Picking the commonest entries leaves exactly what sorting all of them and cutting left.
15617    ///
15618    /// The counts here are deliberately full of ties, including a tie that straddles the cut, which
15619    /// is where a partition and a sort can disagree if the comparison they are given is not total.
15620    #[test]
15621    fn the_commonest_entries_are_the_ones_a_full_sort_would_have_kept() {
15622        let entry =
15623            |value: u32, count: u64| FrequencyEntry { value: FrequencyValue::Code(value), count };
15624        let mut all = (0..FREQUENCY_ENTRIES as u32 * 3)
15625            .map(|code| entry(code, u64::from(code % 7) + 1))
15626            .collect::<Vec<_>>();
15627        all.push(FrequencyEntry { value: FrequencyValue::Null, count: 4 });
15628
15629        let mut sorted = all.clone();
15630        sorted.sort_unstable_by(|left, right| {
15631            right.count.cmp(&left.count).then_with(|| frequency_order(left.value, right.value))
15632        });
15633        let wanted_omitted = sorted[FREQUENCY_ENTRIES].count;
15634        sorted.truncate(FREQUENCY_ENTRIES);
15635
15636        let mut picked = all.clone();
15637        let omitted = keep_most_frequent(&mut picked);
15638        assert_eq!(omitted, wanted_omitted, "the largest count that did not make the cut");
15639        assert_eq!(picked.len(), FREQUENCY_ENTRIES, "the cut is where it says it is");
15640        assert!(
15641            picked
15642                .iter()
15643                .zip(&sorted)
15644                .all(|(one, two)| one.value == two.value && one.count == two.count),
15645            "the same entries in the same order"
15646        );
15647
15648        let mut short = all[..FREQUENCY_ENTRIES - 1].to_vec();
15649        let omitted = keep_most_frequent(&mut short);
15650        assert_eq!(omitted, 0, "nothing is omitted when everything fits");
15651        assert!(short.windows(2).all(|pair| pair[0].count >= pair[1].count), "still in order");
15652    }
15653
15654    /// A dictionary too small to bucket, and one with nothing in it, come back in order too.
15655    #[test]
15656    fn a_short_dictionary_sorts_without_a_bucketing_pass() {
15657        let empty = GlobalDictionary::new();
15658        assert!(empty.ranked(None).expect("an empty order").is_empty(), "nothing in, nothing out");
15659
15660        let mut dictionary = GlobalDictionary::new();
15661        for value in ["pear", "apple", "", "apples", "app"] {
15662            dictionary.code(value).expect("a code for every value");
15663        }
15664        dictionary.finish_blocks().expect("the one block encodes");
15665        let spellings = dictionary_values(&dictionary);
15666        let seen = dictionary
15667            .ranked(None)
15668            .expect("a sorted order")
15669            .iter()
15670            .map(|&(_, code)| spellings[code as usize].clone())
15671            .collect::<Vec<_>>();
15672        let wanted: Vec<Vec<u8>> =
15673            [&b""[..], b"app", b"apple", b"apples", b"pear"].iter().map(|v| v.to_vec()).collect();
15674        assert_eq!(seen, wanted, "shorter first where one runs out inside another");
15675    }
15676}