Skip to main content

rudb_native/
lib.rs

1//! Rudb's single-file columnar snapshot format.
2//!
3//! A committed directory names independently readable column pages. It has two levels: a catalog
4//! directory naming every table in the file, which is what a footer slot points at and what opening
5//! a database reads, and one directory per table under it holding that table's stripes, pages and
6//! statistics. One slot write publishes all of them, so a commit is atomic across tables.
7//!
8//! This version handles scalar columns; the file header has two generation slots so an unfinished
9//! replacement directory cannot hide the last complete one. See
10//! `spec/storage-v3/12-many-tables-in-one-file.md`.
11//!
12//! # Parts and stripes
13//!
14//! A part is one appended chunk, which is a thousand rows, and it is the unit a scan decodes and
15//! hands to the pipeline. A stripe is sixty four parts, and it is the unit the directory describes
16//! and the unit the file is laid out in: one page per column per stripe, holding that column's
17//! sixty four part payloads end to end.
18//!
19//! The two are separate because they are sized by different pressures. A part wants to be small
20//! because it is a vector and vectors live in cache. A stripe wants to be large because everything
21//! the directory holds is per stripe and the directory is one buffer that has to be read and
22//! decoded before a single row can be answered. A hundred million rows of the hundred and five
23//! column ClickBench table is ninety seven thousand parts, and a directory with a page entry and a
24//! pair of bounds per part per column is several hundred megabytes, which is what made that load
25//! fail before this split existed. Sixty four parts to a stripe divides that by sixty four.
26//!
27//! Where the parts of a page start is not in the directory either, for the same reason. Each
28//! stripe writes one index page holding a length and a checksum per part per column, and a reader
29//! preads the sixty four entries belonging to the column it wants. A scan reads the whole column
30//! page once and slices it; a sparse row fetch reads the index entries and then only the part it
31//! needs.
32
33#![forbid(unsafe_code)]
34
35use std::borrow::Cow;
36use std::cmp::{Ordering, Reverse};
37use std::collections::{BTreeMap, HashMap, VecDeque};
38use std::fs::File;
39use std::mem::{size_of, size_of_val};
40use std::path::Path;
41use std::slice;
42use std::sync::atomic::{AtomicBool, AtomicUsize, Ordering as Atomic};
43use std::sync::{Arc, Condvar, Mutex, OnceLock, PoisonError, Weak};
44
45use rudb_common::bounds::{self, Bound, Op, scaled_as};
46use rudb_common::{Clustering, Error, Field, LogicalType, PhysicalType, Result, Value, Width};
47use rudb_encoding::{bitpack, chooser, integer, string};
48use rudb_io::{Filesystem, OpenMode, RealFilesystem};
49use rudb_metrics::{LoadProfile, Stage};
50use rudb_storage::sieve::Sieve;
51use rudb_storage::{Probe, Range, Zone};
52use rudb_vector::string::StringColumn;
53use rudb_vector::validity::Validity;
54use rudb_vector::{Buffer, Chunk, Data, Packed, TextSource, Vector, search_below};
55
56mod distinct;
57pub mod graph;
58pub mod host;
59mod prepare;
60mod projection;
61mod run_projection;
62use prepare::Lent;
63pub mod section;
64pub mod stats;
65mod zones;
66
67pub use prepare::{Building, DICTIONARY_CAP_BYTES, Merged, Merger, Paged, Prepared, Preparer};
68pub use projection::build_sorted_projection;
69pub use run_projection::build_run_projection;
70pub use section::Section;
71pub use zones::{Common, Stripes, ascending, distincts};
72
73const MAGIC: &[u8; 8] = b"RUDBNV10";
74const DIRECTORY: &[u8; 8] = b"RUDBDI10";
75const CATALOG: &[u8; 8] = b"RUDBCA10";
76const NONZERO_COUNTS: &[u8; 8] = b"RUDBNZ10";
77const AGGREGATE_SUMS: &[u8; 8] = b"RUDBAG10";
78const DISTINCT_COUNTS: &[u8; 8] = b"RUDBDC10";
79const INTEGER_EXTREMES: &[u8; 8] = b"RUDBEX10";
80const COMPLETE_FREQUENCIES: &[u8; 8] = b"RUDBFQ10";
81const MAX_CATALOG_FREQUENCIES: usize = 64;
82const FORMAT: u32 = 29;
83
84/// Formats this build can open.
85///
86/// More than one, for the first time, and the reason is spec/graph/10-milestones.md's G1 exit
87/// criterion: a build with the section table in it has to open a file written before the section
88/// table existed, unchanged and without a rewrite. Formats 22 and 23 are those files, and both read
89/// as a table with an empty section table, which is exactly what section 3.1 says a table with no
90/// graph sections is.
91///
92/// All three of the older ones are readable for the same reason. What took the format from 22 to 23
93/// was tags for fourteen more column types, and a file written before that has none of them in it,
94/// so nothing in an older file is a tag this build cannot read. What took it from 23 to 24 is the
95/// section table, which a file written before it simply does not have. What takes it from 24 to 25
96/// is the view section on the end of the catalog, which an older file does not have either, and a
97/// catalog that ends where the tables end reads as a catalog with no views in it. What takes it
98/// from 25 to 26 is that a global dictionary's payload blocks now say where they are, and a file
99/// written before that has them behind one another, which [`open_global_dictionary`] reads by
100/// turning the ends it finds into the same places the newer files name outright. What takes it
101/// from 26 to 27 is that those blocks are written into the file as the load goes, between the
102/// stripes, rather than behind the dictionary's index at the end, so the dictionary's page is the
103/// index and the sorted order and nothing else. A format 26 file has its blocks inside the page,
104/// and the reader tells the two apart by whether the page has room left over for them.
105///
106/// Format 28 adds per-payload-block substring signatures to global string dictionaries. Older
107/// files have no signatures and use the ordinary exact string filter. Format 29 makes each
108/// signature four times as wide, which a dictionary says with [`DICTIONARY_WIDE_GRAMS`], and a
109/// format 28 file is read with the narrow ones it has.
110///
111/// This is not a general compatibility promise. Seven formats are readable because there was a
112/// specific reason for each, and the list shrinks again the moment the older ones stop being worth
113/// carrying.
114const READABLE: &[u32] = &[22, 23, 24, 25, 26, 27, 28, FORMAT];
115
116const HEADER: u64 = 80;
117const SLOT_BYTES: usize = 28;
118const MAX_PAGE: usize = 256 * 1024 * 1024;
119const MAX_DIRECTORY: usize = 128 * 1024 * 1024;
120const FREQUENCIES_V2: &[u8; 8] = b"RUDBFQ2\0";
121const FREQUENCIES: &[u8; 8] = b"RUDBFQ3\0";
122/// Inline spellings for string entries in the bounded frequency synopsis.
123///
124/// A planner usually asks about one literal such as the empty string. Without this block it opens
125/// a multi-million-value global dictionary and visits the payload blocks of every retained entry
126/// merely to compare that literal with at most 512 heavy hitters. The spellings are already in
127/// memory while the writer sorts the dictionary, so storing this bounded copy makes planning a
128/// directory read and leaves the dictionary unopened.
129const FREQUENCY_TEXTS: &[u8; 8] = b"RUDBFT1\0";
130/// Certified host aggregate state for the version-one anchored replacement expression.
131const HOST_GROUPS: &[u8; 8] = b"RUDBHG1\0";
132/// Exact leading counts for a bounded pair of dictionary-backed grouping keys.
133///
134/// This is a separate optional directory block rather than another frequency format. Readers that
135/// predate it still understand every earlier directory, and a table without a pair worth keeping
136/// writes no block at all.
137const PAIR_FREQUENCIES: &[u8; 8] = b"RUDBPF1\0";
138/// The clustering declaration, written after the frequencies and only when there is one.
139///
140/// No format bump for this, which is the convention the frequency section set in #728: a new
141/// optional trailing section with its own magic leaves every file that does not use it byte for
142/// byte what it was, and the version is bumped for a change to a layout that already exists, as
143/// #1029 did. A file with no declaration is the same bytes this build wrote yesterday.
144///
145/// The width byte in this block gained a fifth value for #1285, for a declaration that leaves the
146/// bucket to the row count, and that did not bump the format either. It is the one case where the
147/// reasoning needs saying out loud, because it is a new value in a layout that already exists
148/// rather than a new section. A build without it reading one of these says `clustering width
149/// tag differs` and refuses the table, which is what that message was written for. Bumping the
150/// format instead would have made every file this build writes unreadable to an older one, whether
151/// it has a declaration in it or not, to warn about a case that only arises when it does.
152const CLUSTERING: &[u8; 8] = b"RUDBCL1\0";
153/// The string columns whose global dictionary stopped taking values partway through the load.
154///
155/// Section 5.5 of the encoding spec: a column whose stripes are nearly all new values, or the
156/// fastest growing one once the dictionaries together pass their cap, stops adding to its
157/// dictionary, and every stripe after that is written plainly. The stripes before keep their codes,
158/// so the dictionary is still written and still decodes them, but it no longer holds every value of
159/// the column, and nothing that reads it as if it did can be trusted: not the distinct count, not
160/// the frequencies, not the sorted order's first and last value, and not the codes as a group key
161/// or a membership index. A reader that finds a column named here decodes its coded pages to plain
162/// strings and answers everything else the way it answers a column with no dictionary.
163///
164/// Same convention as [`CLUSTERING`], written only when a column was demoted, so a file with none
165/// is the bytes it always was. A build that predates it refuses a file that has one with
166/// `directory extension magic differs`, which is the right answer, because that build would trust
167/// the dictionary.
168///
169/// A stripe written after the demotion has no membership index for the column. Its slot in the
170/// stripe is written as a page of no bytes, which no real membership index is, since the smallest
171/// one holds its code count.
172const DEMOTED: &[u8; 8] = b"RUDBDM1\0";
173/// The graph section table, written after the clustering declaration and written even when empty.
174///
175/// Same convention and the same reason as the block above it, with one difference: this one is
176/// always there, so a file written by this build says which sections it has rather than leaving a
177/// reader to infer it from where the bytes ran out. Section 3.1 of the graph spec is what makes
178/// that safe to add without a format bump, because a table with no sections answers every query
179/// the way it did before, only without the graph path.
180const SECTIONS: &[u8; 8] = b"RUDBSE1\0";
181/// How many bytes of each column's global dictionary live outside its page, written only when any do.
182///
183/// From format 27 a dictionary's payload blocks are written into the file while the load runs, so
184/// they sit between the stripes and the dictionary's page covers only its index and sorted order.
185/// Nothing needs the total to read the file, because the index names every block. It is here for
186/// what a file costs a column, which [`Reader::layout`] and the statistics budget both report, and
187/// which would otherwise lose most of the bytes of every large string column.
188const DICTIONARY_PAYLOADS: &[u8; 8] = b"RUDBDP1\0";
189
190/// The most sections one table's directory may name.
191///
192/// A relationship contributes at most three sections, so this bounds a table at a few thousand
193/// relationships, which is far past anything a schema has. The bound is here so that a torn
194/// directory naming four billion of them is refused at decode rather than turned into an
195/// allocation, the same reason the extent count has one.
196const MAX_SECTIONS: usize = 4096;
197const FREQUENCY_CANDIDATES: usize = 32_768;
198const FREQUENCY_ENTRIES: usize = 512;
199const FREQUENCY_BUILD_RANK: usize = 10;
200const FREQUENCY_ORDINALS: usize = 131_072;
201const MAX_PAIR_FREQUENCIES: usize = 1024;
202/// The most exact heavy-hitter text one column may copy into the directory.
203///
204/// A column with unusually large leading values keeps the old code-only synopsis instead. The
205/// optimization must never turn a valid load into a directory-size failure.
206const FREQUENCY_TEXT_BUDGET: usize = 1024 * 1024;
207/// The most threads the two per column passes at the end of a commit are spread over.
208///
209/// A table like `hits` has ninety numeric columns, so on a machine with more cores than this the
210/// cap is what decides how long the frequencies take rather than the columns are. It is here at all
211/// because each worker holds a candidate table and a decoded part, and a hundred of those at once
212/// on a narrow machine would be worse than waiting.
213const MAX_FREQUENCY_WORKERS: usize = 32;
214
215/// How many threads the passes at the end of a commit are spread over on this machine.
216fn close_workers() -> usize {
217    std::thread::available_parallelism().map_or(1, usize::from).min(MAX_FREQUENCY_WORKERS)
218}
219
220/// How many bytes the columns closing at the same time may hold between them.
221///
222/// Closing a global dictionary decodes every value it holds, sorts them and drops them, and #1356
223/// took the columns one at a time so that five of them decoded at once were not the peak of a load.
224/// A numeric column's frequencies hold a candidate table and, past it, an exact set of its distinct
225/// values that reaches 512 MiB. The two used to run side by side with only the dictionaries under a
226/// bound, and on the ClickBench `hits` 10M load the close took a load that had held 3.1 GB to 4.8
227/// GB. A column is taken while the ones already closing leave room for it under this, and always
228/// when nothing else is closing, so every dictionary of `hits` at 10M rows closes at once and `URL`
229/// at 100M, which is past this alone, still closes on its own.
230const CLOSE_BYTES: usize = 1 << 30;
231
232/// What a numeric column's frequencies hold before its exact distinct set, which is the candidate
233/// table, its recount and the page being read, with room to spare.
234const NUMERIC_CLOSE_BYTES: usize = 4 << 20;
235
236/// The most threads one stripe's encode is spread over.
237///
238/// Higher than the frequency cap because this is the load itself rather than a pass at the end of
239/// it, and the work is one column of sixty four parts, which is large enough that a thread that
240/// takes one is not a thread that was started for nothing. A machine with more cores than this has
241/// the rest of them on the Parquet read, which is still one thread and is the other half of #808.
242const MAX_ENCODE_WORKERS: usize = 32;
243
244/// How much a writer appends before it asks the kernel to start writing it to the device.
245///
246/// Without it every byte of a load waits in the page cache for the sync at the commit, and that
247/// sync was 1.3 to 1.7 s of a ClickBench `hits` 10M load of 8 to 9 s on the 32 core box. With it
248/// the device writes while the load is still encoding. Thirty two megabytes is a few stripes of
249/// `hits`, big enough that the call costs nothing next to the write, and small enough that what
250/// is left for the commit is one stretch.
251const WRITEBACK_STRETCH: u64 = 32 << 20;
252
253/// The most bytes one column of one part may spend on a membership sieve.
254///
255/// A part is a thousand rows, so a filter sized for every one of them being distinct is about
256/// thirteen hundred bytes and this never binds in practice. It is here so that a part that somehow
257/// arrives much wider than a vector cannot put an unbounded index in the file. What does bind is the
258/// rule in `Writer::encode_pages` that a sieve may not be as large as the part it indexes, which is a cap
259/// per column rather than one number for the whole file.
260const SIEVE_BUDGET: usize = 8 * 1024;
261
262/// The most bytes one end of a per part range may spend on a string.
263///
264/// A bound is allowed to be wider than the truth and never narrower, so a long string is cut down to
265/// this many bytes for the low end and cut down and then stepped up for the high end. The reason for
266/// a cap at all is that there are nine hundred and seventy four parts of a hundred and five columns
267/// in a million rows of ClickBench and `URL` runs to hundreds of bytes, so keeping every end whole
268/// would put more in the directory than the skipping is worth. Twenty four bytes is past the point
269/// where two URLs of the same site still look alike.
270const PART_BOUND_BYTES: usize = 24;
271
272fn io(error: std::io::Error) -> Error {
273    Error::io(error.to_string())
274}
275
276fn invalid(message: &str) -> Error {
277    Error::invalid_input(format!("invalid rudb native file: {message}"))
278}
279
280/// Adds a sequence of byte counts without an overflow the caller has to think about.
281fn sum(counts: impl Iterator<Item = u64>) -> u64 {
282    counts.fold(0, u64::saturating_add)
283}
284
285/// One column's span out of a per column list, or zero when the list is shorter than the column.
286fn span_bytes(spans: &[Span], at: usize) -> u64 {
287    spans.get(at).map_or(0, |span| u64::from(span.length))
288}
289
290/// One column's page out of a per column list, or zero when that column has no page at all.
291fn page_bytes(pages: &[Option<Page>], at: usize) -> u64 {
292    pages.get(at).and_then(Option::as_ref).map_or(0, Page::bytes)
293}
294
295/// Everything one column's global dictionary costs the file, its page and the blocks outside it.
296fn dictionary_bytes(table: &Table, at: usize) -> u64 {
297    page_bytes(&table.dictionaries, at)
298        .saturating_add(table.dictionary_payloads.get(at).copied().unwrap_or(0))
299}
300
301/// The xxHash64 of `bytes`, which is what every span this format stores is checked against.
302///
303/// It walks the input as chunks rather than as offsets into it, and that is the only thing about it
304/// worth a comment. The offset form reads `bytes[at..at + 8]`, and neither the slicing nor the
305/// `try_into` behind it can be proved in range by a compiler that does not know where `at` stopped,
306/// so each of the four lanes paid for a bounds check and a length check on every thirty two bytes.
307/// A chunk carries its own length, so both fold away and the loop is the multiplies and rotates it
308/// was meant to be. That loop runs over every byte of every span a query reads, which on ClickBench
309/// 8 is about five percent of the query.
310fn checksum(bytes: &[u8]) -> u64 {
311    seeded_checksum(bytes, 0)
312}
313
314/// A hundred and twenty eight bit name for `bytes`, as two xxHash64 walks under different seeds,
315/// with the format this build writes folded in so that a name made by one format is never taken
316/// for the name of a file in another.
317///
318/// For a caller outside this crate that has to name a file by what went into it, which is what a
319/// Parquet mirror's key is. See the global dictionary's use of the same pair for the arithmetic.
320#[must_use]
321pub fn content_name(bytes: &[u8]) -> u128 {
322    let seed = u64::from(FORMAT);
323    u128::from(seeded_checksum(bytes, seed)) << 64 | u128::from(seeded_checksum(bytes, !seed))
324}
325
326/// [`content_name`] of bytes that arrive in pieces, which gives the same name as the pieces joined.
327///
328/// A Parquet mirror is named for the file's footer, and that is 930 KB on the ten million row
329/// ClickBench file. Read whole to be hashed it is a freed megabyte in every process that opens the
330/// mirror, which the allocator keeps. Read a window at a time it is a window.
331#[derive(Debug, Clone)]
332pub struct ContentNamer {
333    seeds: [u64; 2],
334    lanes: [[u64; 4]; 2],
335    held: [u8; 32],
336    filled: usize,
337    length: u64,
338}
339
340impl Default for ContentNamer {
341    fn default() -> Self {
342        let seed = u64::from(FORMAT);
343        let seeds = [seed, !seed];
344        let lanes = seeds.map(|seed| {
345            [
346                seed.wrapping_add(XXH_P1).wrapping_add(XXH_P2),
347                seed.wrapping_add(XXH_P2),
348                seed,
349                seed.wrapping_sub(XXH_P1),
350            ]
351        });
352        Self { seeds, lanes, held: [0; 32], filled: 0, length: 0 }
353    }
354}
355
356impl ContentNamer {
357    /// Takes the next piece.
358    pub fn update(&mut self, mut bytes: &[u8]) {
359        self.length += bytes.len() as u64;
360        if self.filled > 0 {
361            let take = (32 - self.filled).min(bytes.len());
362            self.held[self.filled..self.filled + take].copy_from_slice(&bytes[..take]);
363            self.filled += take;
364            bytes = &bytes[take..];
365            if self.filled < 32 {
366                return;
367            }
368            let block = self.held;
369            self.lanes.iter_mut().for_each(|lanes| checksum_block(lanes, &block));
370            self.filled = 0;
371        }
372        let mut blocks = bytes.chunks_exact(32);
373        for block in blocks.by_ref() {
374            self.lanes.iter_mut().for_each(|lanes| checksum_block(lanes, block));
375        }
376        let rest = blocks.remainder();
377        self.held[..rest.len()].copy_from_slice(rest);
378        self.filled = rest.len();
379    }
380
381    /// The name of everything taken so far.
382    #[must_use]
383    pub fn finish(&self) -> u128 {
384        let rest = &self.held[..self.filled];
385        let [first, second] = [0, 1].map(|at| {
386            if self.length < 32 {
387                checksum_tail(self.seeds[at].wrapping_add(XXH_P5).wrapping_add(self.length), rest)
388            } else {
389                finish_checksum(self.lanes[at], rest, self.length)
390            }
391        });
392        u128::from(first) << 64 | u128::from(second)
393    }
394}
395
396/// The xxHash64 of `bytes` started from `seed`, which is the same walk with a different beginning.
397///
398/// A seed is here for one caller: a global dictionary decides whether two values are the same by
399/// their hashes rather than by their bytes, and one sixty four bit hash is not enough to do that
400/// with. Twenty million distinct values collide on sixty four bits about once in a hundred thousand
401/// loads, which for a wrong answer is far too often. Two hashes of the same value under different
402/// seeds are independent, so the pair is a hundred and twenty eight bits and the same arithmetic
403/// puts that at around one in 1e24.
404fn seeded_checksum(bytes: &[u8], seed: u64) -> u64 {
405    // Asked for before the loop rather than after it, because a `ChunksExact` settles what it
406    // cannot divide when it is built and hands back the same tail whether it has been walked or not.
407    let mut blocks = bytes.chunks_exact(32);
408    let rest = blocks.remainder();
409    if bytes.len() < 32 {
410        return checksum_tail(seed.wrapping_add(XXH_P5).wrapping_add(bytes.len() as u64), rest);
411    }
412    let mut lanes = [
413        seed.wrapping_add(XXH_P1).wrapping_add(XXH_P2),
414        seed.wrapping_add(XXH_P2),
415        seed,
416        seed.wrapping_sub(XXH_P1),
417    ];
418    for block in blocks.by_ref() {
419        checksum_block(&mut lanes, block);
420    }
421    finish_checksum(lanes, rest, bytes.len() as u64)
422}
423
424const XXH_P1: u64 = 11_400_714_785_074_694_791;
425const XXH_P2: u64 = 14_029_467_366_897_019_727;
426const XXH_P3: u64 = 1_609_587_929_392_839_161;
427const XXH_P4: u64 = 9_650_029_242_287_828_579;
428const XXH_P5: u64 = 2_870_177_450_012_600_261;
429
430fn checksum_round(state: u64, word: u64) -> u64 {
431    state.wrapping_add(word.wrapping_mul(XXH_P2)).rotate_left(31).wrapping_mul(XXH_P1)
432}
433
434fn checksum_word(chunk: &[u8]) -> u64 {
435    u64::from_le_bytes(chunk.try_into().expect("eight checksum bytes"))
436}
437
438/// One thirty two byte block into the four lanes.
439fn checksum_block(lanes: &mut [u64; 4], block: &[u8]) {
440    for (lane, chunk) in lanes.iter_mut().zip(block.chunks_exact(8)) {
441        *lane = checksum_round(*lane, checksum_word(chunk));
442    }
443}
444
445/// The lanes after every whole block, folded together with what was left over and the length.
446fn finish_checksum(lanes: [u64; 4], rest: &[u8], length: u64) -> u64 {
447    let merge = |state: u64, lane: u64| {
448        (state ^ checksum_round(0, lane)).wrapping_mul(XXH_P1).wrapping_add(XXH_P4)
449    };
450    let [one, two, three, four] = lanes;
451    let combined = one
452        .rotate_left(1)
453        .wrapping_add(two.rotate_left(7))
454        .wrapping_add(three.rotate_left(12))
455        .wrapping_add(four.rotate_left(18));
456    let hash = merge(merge(merge(merge(combined, one), two), three), four);
457    checksum_tail(hash.wrapping_add(length), rest)
458}
459
460/// The fewer than thirty two bytes after the last whole block, and the final mix.
461fn checksum_tail(mut hash: u64, mut rest: &[u8]) -> u64 {
462    let mut words = rest.chunks_exact(8);
463    for chunk in words.by_ref() {
464        hash ^= checksum_round(0, checksum_word(chunk));
465        hash = hash.rotate_left(27).wrapping_mul(XXH_P1).wrapping_add(XXH_P4);
466    }
467    rest = words.remainder();
468    if rest.len() >= 4 {
469        let (head, tail) = rest.split_at(4);
470        let quarter = u32::from_le_bytes(head.try_into().expect("four checksum bytes"));
471        hash ^= u64::from(quarter).wrapping_mul(XXH_P1);
472        hash = hash.rotate_left(23).wrapping_mul(XXH_P2).wrapping_add(XXH_P3);
473        rest = tail;
474    }
475    for &byte in rest {
476        hash ^= u64::from(byte).wrapping_mul(XXH_P5);
477        hash = hash.rotate_left(11).wrapping_mul(XXH_P1);
478    }
479    hash ^= hash >> 33;
480    hash = hash.wrapping_mul(XXH_P2);
481    hash ^= hash >> 29;
482    hash = hash.wrapping_mul(XXH_P3);
483    hash ^ (hash >> 32)
484}
485
486/// The checksum of `length` bytes of `file` from `offset`, read [`DIRECTORY_WINDOW`] at a time.
487///
488/// The same xxHash64 as [`checksum`], carried across reads rather than over one buffer, so that a
489/// directory can be checked without all of it being in memory at once. The four lanes take whole
490/// thirty two byte blocks, and a read that ends partway through one keeps the tail for the next.
491fn file_checksum(file: &File, offset: u64, length: usize) -> Result<u64> {
492    walk_checksummed(file, offset, length, DIRECTORY_WINDOW, |_| Ok(()))
493}
494
495/// Reads `length` bytes at `offset` a window at a time, hands each window to `each`, and answers
496/// the checksum of all of them.
497///
498/// `window` is a multiple of thirty two, so every window but the last is whole blocks of the hash
499/// and nothing has to be carried from one read to the next.
500fn walk_checksummed(
501    file: &File,
502    offset: u64,
503    length: usize,
504    window: usize,
505    mut each: impl FnMut(&[u8]) -> Result<()>,
506) -> Result<u64> {
507    debug_assert!(window % 32 == 0 && window > 0, "a window is whole blocks of the hash");
508    if length < 32 {
509        let mut bytes = vec![0; length];
510        read_at(file, offset, &mut bytes)?;
511        each(&bytes)?;
512        return Ok(checksum(&bytes));
513    }
514    let mut lanes = [XXH_P1.wrapping_add(XXH_P2), XXH_P2, 0, 0_u64.wrapping_sub(XXH_P1)];
515    let mut buffer = vec![0; window.min(length)];
516    let mut read = 0;
517    let (mut whole, mut filled) = (0, 0);
518    while read < length {
519        filled = buffer.len().min(length - read);
520        read_at(file, offset + read as u64, &mut buffer[..filled])?;
521        read += filled;
522        each(&buffer[..filled])?;
523        whole = filled / 32 * 32;
524        for block in buffer[..whole].chunks_exact(32) {
525            checksum_block(&mut lanes, block);
526        }
527    }
528    Ok(finish_checksum(lanes, &buffer[whole..filled], length as u64))
529}
530
531#[derive(Debug, Clone, Copy)]
532struct Slot {
533    offset: u64,
534    length: u32,
535    generation: u64,
536    hash: u64,
537}
538
539impl Slot {
540    fn bytes(self) -> [u8; SLOT_BYTES] {
541        let mut result = [0; SLOT_BYTES];
542        result[..8].copy_from_slice(&self.offset.to_le_bytes());
543        result[8..12].copy_from_slice(&self.length.to_le_bytes());
544        result[12..20].copy_from_slice(&self.generation.to_le_bytes());
545        result[20..28].copy_from_slice(&self.hash.to_le_bytes());
546        result
547    }
548
549    fn read(bytes: &[u8]) -> Self {
550        Self {
551            offset: u64::from_le_bytes(bytes[..8].try_into().expect("eight bytes")),
552            length: u32::from_le_bytes(bytes[8..12].try_into().expect("four bytes")),
553            generation: u64::from_le_bytes(bytes[12..20].try_into().expect("eight bytes")),
554            hash: u64::from_le_bytes(bytes[20..28].try_into().expect("eight bytes")),
555        }
556    }
557}
558
559#[derive(Debug, Clone, Copy)]
560struct Page {
561    offset: u64,
562    length: u32,
563    hash: u64,
564}
565
566impl Page {
567    /// How much of the file this page takes, for [`Reader::layout`].
568    fn bytes(&self) -> u64 {
569        u64::from(self.length)
570    }
571}
572
573#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
574enum FrequencyValue {
575    Null,
576    Integer(i128),
577    Code(u32),
578}
579
580/// A table keyed by the sixty four bits of the values the numeric frequency pass counts.
581///
582/// Every integer of every numeric column goes through one of these at least once when a table
583/// closes, and with the standard hasher that was a fifth of the close on its own, all of it SipHash
584/// guarding against an attacker who would have to choose the rows of the file being written.
585type FrequencyMap<V> = HashMap<u64, V, Spread>;
586
587/// The first pass of [`Writer::numeric_frequency`]: a Misra-Gries candidate table keyed by a value's
588/// sixty four bits, with the null counted beside it.
589///
590/// The table is an open addressed one of its own rather than a `HashMap`. On a column that is near
591/// unique, which `hits` has a dozen of, nearly every row is a value the table has not seen, and a
592/// `HashMap` spent a lookup and then a second hash and probe to insert it, and a `retain` over every
593/// bucket each time the table filled. Those were 6 percent of the CPU of loading the 10m ClickBench
594/// file, and the slowest of those columns decided how long the whole frequency step took. Here a
595/// value is found or given the empty slot it stopped at in one probe, and a decrement rebuilds the
596/// table from the few candidates that outlive it.
597///
598/// What the table holds after a stream of rows is the same set of counts either way, since that is
599/// fixed by the algorithm and not by where the counts live.
600#[derive(Debug)]
601struct Candidates {
602    /// A power of two number of slots, at most half of them in use. A count of zero is an empty
603    /// slot, which no candidate ever is, because one whose count reaches zero is dropped.
604    slots: Vec<Candidate>,
605    held: usize,
606    nulls: u32,
607    decrements: u64,
608    /// The candidates that outlive a decrement, kept so that each decrement is not an allocation.
609    survivors: Vec<Candidate>,
610}
611
612/// One slot of [`Candidates`], the value's bits beside its count so a probe reads one line.
613#[derive(Debug, Default, Clone, Copy)]
614struct Candidate {
615    bits: u64,
616    count: u32,
617}
618
619/// The slots a candidate table starts with, grown by doubling as it fills.
620const FIRST_CANDIDATE_SLOTS: usize = 64;
621
622impl Default for Candidates {
623    fn default() -> Self {
624        Self {
625            slots: vec![Candidate::default(); FIRST_CANDIDATE_SLOTS],
626            held: 0,
627            nulls: 0,
628            decrements: 0,
629            survivors: Vec::new(),
630        }
631    }
632}
633
634impl Candidates {
635    /// Counts `times` rows of `bits` and ends in the state `times` rows counted one at a time would.
636    ///
637    /// A value already held, or one there is room to hold, takes the whole run at once, because
638    /// every row after the first would find it held. A value the full table turns away goes a row
639    /// at a time, because each of its rows decrements every candidate and one of those decrements
640    /// can free the place the next row takes.
641    fn add(&mut self, bits: Option<u64>, mut times: u32) {
642        while times > 0 {
643            let room = self.held + usize::from(self.nulls != 0) < FREQUENCY_CANDIDATES;
644            match bits {
645                Some(bits) => {
646                    let (at, found) = self.find(bits);
647                    if found {
648                        self.slots[at].count = self.slots[at].count.saturating_add(times);
649                        return;
650                    }
651                    if room {
652                        self.place(at, bits, times);
653                        return;
654                    }
655                }
656                None if self.nulls != 0 => {
657                    self.nulls = self.nulls.saturating_add(times);
658                    return;
659                }
660                None if room => {
661                    self.nulls = times;
662                    return;
663                }
664                None => {}
665            }
666            self.decrement();
667            times -= 1;
668        }
669    }
670
671    /// The slot holding `bits` and `true`, or the empty slot a search for it stopped at and `false`.
672    fn find(&self, bits: u64) -> (usize, bool) {
673        let mask = self.slots.len() - 1;
674        let mut at = home(bits, self.slots.len());
675        loop {
676            let slot = self.slots[at];
677            if slot.count == 0 {
678                return (at, false);
679            }
680            if slot.bits == bits {
681                return (at, true);
682            }
683            at = (at + 1) & mask;
684        }
685    }
686
687    /// Where `bits` is held, for the recount, which reads the table without changing it.
688    fn position(&self, bits: u64) -> Option<usize> {
689        match self.find(bits) {
690            (at, true) => Some(at),
691            (_, false) => None,
692        }
693    }
694
695    /// Puts a new candidate in the empty slot `at`, which a search for it just stopped at, doubling
696    /// the table first when that would fill more than half of it.
697    fn place(&mut self, at: usize, bits: u64, count: u32) {
698        let at = if (self.held + 1) * 2 > self.slots.len() {
699            let wider = self.slots.len() * 2;
700            let old = std::mem::replace(&mut self.slots, vec![Candidate::default(); wider]);
701            for slot in old.into_iter().filter(|slot| slot.count != 0) {
702                let (to, _) = self.find(slot.bits);
703                self.slots[to] = slot;
704            }
705            self.find(bits).0
706        } else {
707            at
708        };
709        self.slots[at] = Candidate { bits, count };
710        self.held += 1;
711    }
712
713    /// Takes one from every candidate and the null, dropping the ones that reach zero.
714    fn decrement(&mut self) {
715        let mut survivors = std::mem::take(&mut self.survivors);
716        survivors.clear();
717        survivors.extend(
718            self.slots
719                .iter()
720                .filter(|slot| slot.count > 1)
721                .map(|slot| Candidate { bits: slot.bits, count: slot.count - 1 }),
722        );
723        self.slots.fill(Candidate::default());
724        self.held = survivors.len();
725        for &slot in &survivors {
726            let (at, _) = self.find(slot.bits);
727            self.slots[at] = slot;
728        }
729        self.survivors = survivors;
730        self.nulls = self.nulls.saturating_sub(1);
731        self.decrements = self.decrements.saturating_add(1);
732    }
733
734    /// Every candidate's bits and count, in no particular order.
735    fn pairs(&self) -> impl Iterator<Item = (u64, u32)> + '_ {
736        self.slots.iter().filter(|slot| slot.count != 0).map(|slot| (slot.bits, slot.count))
737    }
738}
739
740/// The slot a search for `bits` starts at in a table of `slots`, a power of two.
741///
742/// The top bits of a multiply by the golden ratio, which every bit of the value reaches, so a
743/// timestamp column whose values are all multiples of a million still spreads over the table.
744fn home(bits: u64, slots: usize) -> usize {
745    (bits.wrapping_mul(0x9E37_79B9_7F4A_7C15) >> (64 - slots.trailing_zeros())) as usize
746}
747
748/// Equal rows in a row, gathered so they are counted once.
749#[derive(Debug, Default)]
750struct Run {
751    bits: Option<u64>,
752    times: u32,
753}
754
755impl Run {
756    /// Adds one row, and hands back the run it ended if it was not the same value.
757    fn push(&mut self, bits: Option<u64>) -> Option<(Option<u64>, u32)> {
758        if self.times != 0 && self.bits == bits && self.times < u32::MAX {
759            self.times += 1;
760            return None;
761        }
762        let ended = self.take();
763        self.bits = bits;
764        self.times = 1;
765        ended
766    }
767
768    /// The run being gathered, if there is one, leaving none.
769    fn take(&mut self) -> Option<(Option<u64>, u32)> {
770        let times = std::mem::take(&mut self.times);
771        (times != 0).then_some((self.bits, times))
772    }
773}
774
775/// Builds the hasher for [`FrequencyMap`].
776#[derive(Debug, Default, Clone, Copy)]
777struct Spread;
778
779impl std::hash::BuildHasher for Spread {
780    type Hasher = SpreadHasher;
781
782    fn build_hasher(&self) -> SpreadHasher {
783        SpreadHasher(0)
784    }
785}
786
787/// Folds each word in with a full width multiply whose two halves are xored together.
788///
789/// A plain multiply leaves the low bits of the hash as poor as the low bits of the key, and the
790/// table picks its bucket from the low bits, so a timestamp column, whose values are all multiples
791/// of a million microseconds, would pile into a sixty fourth of the buckets. Folding the high half
792/// of the product back in is what gives the low bits the whole word.
793#[derive(Debug)]
794struct SpreadHasher(u64);
795
796impl SpreadHasher {
797    fn mix(&mut self, word: u64) {
798        let product = u128::from(self.0 ^ word) * 0x9E37_79B9_7F4A_7C15_u128;
799        self.0 = (product as u64) ^ ((product >> 64) as u64);
800    }
801}
802
803impl std::hash::Hasher for SpreadHasher {
804    fn write(&mut self, bytes: &[u8]) {
805        for part in bytes.chunks(8) {
806            let mut word = [0; 8];
807            word[..part.len()].copy_from_slice(part);
808            self.mix(u64::from_le_bytes(word));
809        }
810    }
811
812    fn write_u32(&mut self, value: u32) {
813        self.mix(u64::from(value));
814    }
815
816    fn write_u64(&mut self, value: u64) {
817        self.mix(value);
818    }
819
820    fn write_i128(&mut self, value: i128) {
821        self.mix(value as u64);
822        self.mix((value >> 64) as u64);
823    }
824
825    fn write_isize(&mut self, value: isize) {
826        self.mix(value as u64);
827    }
828
829    fn finish(&self) -> u64 {
830        self.0
831    }
832}
833
834#[derive(Debug, Clone)]
835struct FrequencyEntry {
836    value: FrequencyValue,
837    count: u64,
838}
839
840/// Exact leading frequencies for one column.
841///
842/// Values outside `entries` occur at most `omitted_max` times. This lets a count-descending TopN
843/// use the synopsis only when its last winner is strictly above every omitted value.
844#[derive(Debug, Clone)]
845struct FrequencySummary {
846    entries: Vec<FrequencyEntry>,
847    omitted_max: u64,
848    ordinals: Vec<u64>,
849    ordinal_entries: Vec<u16>,
850}
851
852#[derive(Debug, Clone)]
853struct PairFrequencyEntry {
854    first_entry: u16,
855    second: Option<u32>,
856    count: u64,
857}
858
859/// Exact leading counts for one numeric frequency anchor and one stable string code space.
860///
861/// `omitted_max` covers both first-key values outside the numeric synopsis and pairs below the
862/// retained prefix. A TopN may therefore use the entries only when its boundary strictly exceeds
863/// this number.
864#[derive(Debug, Clone)]
865struct PairFrequencySummary {
866    first: u16,
867    second: u16,
868    entries: Vec<PairFrequencyEntry>,
869    omitted_max: u64,
870}
871
872/// One column's frequency synopsis, in memory or left where it is in the file.
873///
874/// A writer holds what it counted. A reader leaves every synopsis in the file and reads one back
875/// when a query asks about its column, because they are the largest thing in a directory once they
876/// are decoded, forty eight bytes an entry and nearly twenty thousand entries over `hits`, and
877/// most queries ask about none of them. Where one sits is found at open, by reading it through and
878/// checking it, so a torn synopsis is still refused when the table is opened.
879#[derive(Debug, Clone)]
880enum Frequencies {
881    Held(FrequencySummary),
882    /// Where the synopsis sits, and whether it was written with the value of each ordinal, which
883    /// is what the directory's frequency magic says and the synopsis itself does not.
884    Stored {
885        span: Span,
886        values: bool,
887    },
888}
889
890/// The values one column's frequency synopsis lists, with a bound on everything it left out.
891///
892/// What [`Reader::frequency_prefix`] answers. The counts are exact, and `omitted_max` is how many
893/// rows any value not in the list can hold, which is zero when nothing was left out at all.
894#[derive(Debug, Clone)]
895pub struct FrequencyPrefix {
896    /// Every value the synopsis lists, with the number of rows holding it, count descending.
897    pub entries: Vec<(Value, u64)>,
898    /// How many rows the most common value outside the list holds, and zero for a complete list.
899    pub omitted_max: u64,
900}
901
902/// Sparse row ordinals covered by a numeric frequency candidate set.
903#[derive(Debug, Clone, PartialEq)]
904pub struct FrequencyOccurrences {
905    /// Upper bound for the frequency of every value absent from the fetched rows.
906    pub omitted_max: u64,
907    /// Table-wide row ordinals in ascending order.
908    pub ordinals: Vec<u64>,
909    /// The retained heavy-hitter values named by `anchor_indices`.
910    pub anchors: Vec<Value>,
911    /// The index in `anchors` at each ordinal, or empty for a legacy FQ2 directory.
912    pub anchor_indices: Vec<u16>,
913}
914
915/// Exact grouped counts for a pair of values, in descending count order.
916pub type PairFrequencyCounts = Vec<(Vec<Value>, u64)>;
917
918/// Where one column's page for one stripe sits in the file.
919///
920/// A column page has no checksum of its own because every part inside it carries one, and the
921/// stripe's index page holds those. Checking a part on the way out of the page covers exactly the
922/// bytes a reader is about to decode, and covers them once whether the reader took the whole page
923/// or pulled one part out of the middle of it.
924#[derive(Debug, Clone, Copy, Default)]
925struct Span {
926    offset: u64,
927    length: u32,
928}
929
930/// One optional page for each column of a stripe, holding only the pages that are there.
931///
932/// A stripe has three of these, the membership, sieve and part range pages. As a
933/// `Vec<Option<Page>>` each was thirty two bytes a column whether the page was there or not, and
934/// over the ten million rows of `hits` that is half a megabyte at open for 7171 pages out of 16380
935/// slots. Kept sparse and packed, a page that is there is twenty four bytes and one that is not is
936/// nothing.
937#[derive(Debug, Clone, Default)]
938struct Pages {
939    columns: usize,
940    held: Box<[StripePage]>,
941}
942
943/// A page and the column it is for, packed so that the column sits where the padding was.
944#[derive(Debug, Clone, Copy)]
945struct StripePage {
946    offset: u64,
947    hash: u64,
948    length: u32,
949    column: u32,
950}
951
952impl Pages {
953    /// The pages of `columns` columns, one slot each in column order.
954    fn from_slots(slots: Vec<Option<Page>>) -> Result<Self> {
955        let mut held = Vec::with_capacity(slots.iter().flatten().count());
956        for (column, page) in slots.iter().enumerate() {
957            if let Some(page) = page {
958                let column =
959                    u32::try_from(column).map_err(|_| invalid("too many columns for a page"))?;
960                held.push(StripePage {
961                    offset: page.offset,
962                    hash: page.hash,
963                    length: page.length,
964                    column,
965                });
966            }
967        }
968        Ok(Self { columns: slots.len(), held: held.into_boxed_slice() })
969    }
970
971    /// The page of one column, if it has one.
972    fn get(&self, column: usize) -> Option<Page> {
973        let at = self.held.binary_search_by_key(&column, |placed| placed.column as usize).ok()?;
974        let placed = self.held[at];
975        Some(Page { offset: placed.offset, length: placed.length, hash: placed.hash })
976    }
977
978    /// One slot per column, in column order, the way the directory writes them.
979    fn slots(&self) -> impl Iterator<Item = Option<Page>> + '_ {
980        (0..self.columns).map(|column| self.get(column))
981    }
982
983    /// How much of the file one column's page takes, or zero when it has none.
984    fn bytes(&self, column: usize) -> u64 {
985        self.get(column).map_or(0, |page| page.bytes())
986    }
987}
988
989/// One independently readable stripe of a table.
990#[derive(Debug, Clone)]
991pub struct Stripe {
992    rows: usize,
993    /// Rows in each part, in source order. Kept in the directory so that mapping a row ordinal to a
994    /// part, which every sparse fetch does, never reads the file.
995    parts: Vec<u32>,
996    /// The index page: one section per column, holding a length and a checksum for every part and
997    /// then a checksum of the section itself, so that a reader can pread one column's section and
998    /// still know it is intact.
999    index: Span,
1000    pages: Vec<Span>,
1001    memberships: Pages,
1002    /// One page per column holding the membership sieve of every part of the stripe, for the
1003    /// columns that have one. A column whose parts all declined a sieve has no page at all.
1004    sieves: Pages,
1005    /// One page per column holding the two ends and the null count of every part of the stripe.
1006    ///
1007    /// The stripe's own `zone` below covers sixty four times as many rows, and on a column that is
1008    /// not the one the rows are ordered by that is the difference between skipping half the file and
1009    /// skipping all but three percent of it. On ClickBench 24 the cutoff the answer settles at
1010    /// leaves eight stripes of sixteen alive and thirty parts of nine hundred and seventy four.
1011    ///
1012    /// A page per column rather than one page for the stripe, so that a query that compares one
1013    /// column reads the ends of that column and not of the hundred and four beside it. Read lazily
1014    /// for the same reason, like the sieves.
1015    part_ranges: Pages,
1016    zone: Zone,
1017}
1018
1019impl Stripe {
1020    /// Number of rows in this stripe.
1021    #[must_use]
1022    pub fn rows(&self) -> usize {
1023        self.rows
1024    }
1025
1026    /// Number of parts in this stripe.
1027    #[must_use]
1028    pub fn parts(&self) -> usize {
1029        self.parts.len()
1030    }
1031
1032    /// The two ends and the null count of every column over the whole stripe.
1033    ///
1034    /// In the directory and so in memory, which is what makes it the one a planner can ask. The
1035    /// finer ones are a page per column per stripe in the file, read by [`Reader::skips`] when a
1036    /// scan wants to know which parts to open.
1037    #[must_use]
1038    pub fn zone(&self) -> &Zone {
1039        &self.zone
1040    }
1041}
1042
1043/// The committed table directory.
1044#[derive(Debug, Clone)]
1045pub struct Table {
1046    name: String,
1047    fields: Vec<Field>,
1048    stripes: Vec<Stripe>,
1049    rows: usize,
1050    dictionaries: Vec<Option<Page>>,
1051    /// Bytes of each column's dictionary payload that are outside its page, which is all of them
1052    /// from format 27 and none of them before. See [`DICTIONARY_PAYLOADS`].
1053    ///
1054    /// Empty rather than a row of zeros on a table that has none, and read with `get` for that
1055    /// reason, so that a table built by hand in a test does not have to know about it.
1056    dictionary_payloads: Vec<u64>,
1057    /// The columns whose dictionary stopped taking values partway through the load, see
1058    /// [`DEMOTED`].
1059    ///
1060    /// Empty rather than a row of `false` on a table that has none, and read with `get`, for the
1061    /// same reason `dictionary_payloads` is.
1062    demoted: Vec<bool>,
1063    frequencies: Vec<Option<Frequencies>>,
1064    pair_frequencies: Vec<PairFrequencySummary>,
1065    /// String spellings aligned with each column's frequency entries.
1066    ///
1067    /// Empty for files written before `RUDBFT1`. A `None` entry is the null frequency entry; every
1068    /// code entry in a column named by the block has its exact bytes here.
1069    frequency_texts: Vec<Vec<Option<Vec<u8>>>>,
1070    /// Exact candidate host aggregates and an upper bound for every omitted host.
1071    host_groups: Option<host::HostSummary>,
1072    /// How many distinct values each column holds, for the columns that know.
1073    ///
1074    /// A dictionary entry is made the first time a value is seen and nothing ever removes one, so
1075    /// the size of the dictionary is the number of distinct values in the column. That is the whole
1076    /// story for a column with no null in it, and the wrong number by one for a column with a null
1077    /// in it, because a null row is written as the code for the empty string and makes an entry the
1078    /// dictionary would not otherwise have. The writer knows which case it is, since it counts the
1079    /// non-null rows that use each code while it builds the frequency summary, and the reader cannot
1080    /// work it out from the dictionary alone. So the writer settles it here.
1081    distincts: Vec<Option<u64>>,
1082    /// The order the rows of this table are meant to be stored in, if anybody declared one.
1083    ///
1084    /// A declaration and not a measurement. Nothing here checks that the stripes actually arrived
1085    /// in this order, and the reason it is worth storing anyway is that the order is the only thing
1086    /// about a table that a rewrite destroys without anybody noticing. The fragment ranges prune on
1087    /// whatever order the rows came in, so a table loaded sorted prunes and the same table after a
1088    /// checkpoint that did not know to keep the order quietly stops pruning and nothing says why.
1089    clustering: Option<Clustering>,
1090    /// The file generation of the commit that last wrote this table's column pages.
1091    ///
1092    /// This is what spec/graph/03-the-file-format.md section 3.2 calls the table generation, and
1093    /// the definition is deliberately about the pages rather than about the directory. A graph
1094    /// section is a restatement of a column in terms of row ids, so what invalidates one is the
1095    /// rows being renumbered, and nothing else. Adding a second table to the file, or attaching a
1096    /// section to this one, commits a new file generation without touching a single row of this
1097    /// table, and a definition that moved with those would declare every section in the file stale
1098    /// for no reason.
1099    ///
1100    /// Zero on a table written before format 23, where nothing recorded it. Real generations start
1101    /// at one, so zero can never match a section's stamp, and a table from format 22 has no
1102    /// sections for it to match anyway.
1103    generation: u64,
1104    /// The graph sections this table carries, per spec/graph/03-the-file-format.md section 3.2.
1105    ///
1106    /// Empty for every table written before the section table existed, and empty is not a
1107    /// degraded state: section 3.1 says deleting every graph section from a file changes no answer,
1108    /// only the time, so a table with none here answers every query the same way and slower. That
1109    /// is what lets this field arrive without a migration.
1110    sections: Vec<Section>,
1111}
1112
1113impl Table {
1114    /// The SQL table name held by this snapshot.
1115    #[must_use]
1116    pub fn name(&self) -> &str {
1117        &self.name
1118    }
1119
1120    /// Columns in their SQL order.
1121    #[must_use]
1122    pub fn fields(&self) -> &[Field] {
1123        &self.fields
1124    }
1125
1126    /// Committed row count.
1127    #[must_use]
1128    pub fn rows(&self) -> usize {
1129        self.rows
1130    }
1131
1132    /// Independently readable stripes.
1133    #[must_use]
1134    pub fn stripes(&self) -> &[Stripe] {
1135        &self.stripes
1136    }
1137
1138    /// The order the rows are meant to be stored in, if this table was declared with one.
1139    #[must_use]
1140    pub fn clustering(&self) -> Option<&Clustering> {
1141        self.clustering.as_ref()
1142    }
1143
1144    /// The generation every section of this table is judged against.
1145    ///
1146    /// See the field. A caller deciding whether to read a section asks [`Section::usable`] with
1147    /// this.
1148    #[must_use]
1149    pub fn generation(&self) -> u64 {
1150        self.generation
1151    }
1152
1153    /// Every graph section this table names, including the kinds this build does not know.
1154    ///
1155    /// Including them is the point. A caller that wants only the ones it can use asks
1156    /// [`Section::usable`], and a caller rewriting the directory carries the rest through, so a
1157    /// file opened by an older build and written again does not silently lose a section that build
1158    /// had no name for.
1159    #[must_use]
1160    pub fn sections(&self) -> &[Section] {
1161        &self.sections
1162    }
1163}
1164
1165/// One table's line in the catalog directory.
1166///
1167/// The small level of the two. It holds what opening a database needs and nothing else: the name to
1168/// bind, the shape to plan against, the row count, and where the table's own directory sits. A file
1169/// of eight tables is eight of these, and reading them costs the same whether the tables hold a
1170/// thousand rows or a billion.
1171///
1172/// The name, the fields and the row count are repeated here rather than pointed at inside the table
1173/// directory, which is the entire point of having two levels. A catalog that pointed at them would
1174/// have to read every table directory at open to answer what tables there are, which is the cost
1175/// this level exists to avoid.
1176#[derive(Debug, Clone)]
1177struct Entry {
1178    name: String,
1179    fields: Vec<Field>,
1180    rows: usize,
1181    /// Where this table's own directory sits, with the checksum it was committed under.
1182    directory: Page,
1183    /// Legacy nonzero counts. New files leave these empty and derive filtered counts from
1184    /// reusable column frequencies when a query needs them.
1185    nonzero: Vec<Option<u64>>,
1186    /// Exact sum and non-null count for signed integer columns.
1187    aggregates: Vec<Option<(i128, u64)>>,
1188    /// Exact non-null distinct values when the writer finished counting the column.
1189    distincts: Vec<Option<u64>>,
1190    /// Exact integer or date bounds; the inner `None` means every row is null.
1191    extremes: Vec<StoredIntegerExtremes>,
1192    /// Complete bounded numeric frequencies, including NULL when present.
1193    frequencies: Vec<StoredNumericFrequencies>,
1194}
1195
1196type StoredIntegerExtremes = Option<Option<(i128, i128)>>;
1197type StoredNumericFrequencies = Option<NumericFrequencies>;
1198
1199/// One view's line in the catalog directory.
1200///
1201/// A view has no pages, so unlike a table it is entirely here and there is no second level under it.
1202/// What it is made of is text: the body the binder binds again at every reference, and the whole
1203/// statement written back out, which is what `duckdb_views()` reports and nothing else reads.
1204///
1205/// The columns are a cache and they are written down anyway, which is worth saying out loud because
1206/// a cache in a file looks like a mistake. It is what the pin does. Create a view on a file, open
1207/// the file again in another process, and `duckdb_views()` answers `column_count` and `is_bound`
1208/// true without anything having bound the body, so the list survived the write. Not writing it
1209/// would answer null and false there, and the only way back would be to bind every view at open,
1210/// which is the thing the cache exists to avoid.
1211#[derive(Debug, Clone, PartialEq, Eq)]
1212pub struct ViewEntry {
1213    /// The view's own name, without the schema, the way a table entry holds its name.
1214    pub name: String,
1215    /// The query the view stands for, as the text that was written.
1216    pub sql: String,
1217    /// The whole `CREATE VIEW` written back out.
1218    pub statement: String,
1219    /// The column names the statement gave, which rename a prefix of what the body produces.
1220    pub aliases: Vec<String>,
1221    /// The columns the last bind of the body produced.
1222    pub columns: Vec<Field>,
1223}
1224
1225/// Where one column's bytes went, taken from the directory rather than by reading pages.
1226#[derive(Debug, Clone)]
1227pub struct ColumnLayout {
1228    /// The column's name, so a report does not have to carry the field list beside this.
1229    pub name: String,
1230    /// The type, spelled the way the catalog spells it.
1231    pub kind: String,
1232    /// Every stripe's page of this column added up, which is the encoded data itself.
1233    pub pages: u64,
1234    /// Every stripe's exact code membership page for this column.
1235    pub memberships: u64,
1236    /// Every stripe's membership sieve page for this column.
1237    pub sieves: u64,
1238    /// Every stripe's per part range page for this column.
1239    pub part_ranges: u64,
1240    /// The table wide dictionary of this column, if it has one.
1241    pub dictionary: u64,
1242}
1243
1244impl ColumnLayout {
1245    /// Everything this column costs, which is what the file would lose if the column went.
1246    #[must_use]
1247    pub fn total(&self) -> u64 {
1248        self.pages
1249            .saturating_add(self.memberships)
1250            .saturating_add(self.sieves)
1251            .saturating_add(self.part_ranges)
1252            .saturating_add(self.dictionary)
1253    }
1254}
1255
1256/// Where a whole file's bytes went.
1257///
1258/// Every number here comes out of the committed directory, so taking it costs one directory read
1259/// however large the file is. That is the point: a 45 GB table has to be able to say where it went
1260/// without being read, or nobody will ask.
1261///
1262/// The parts that are not a column are kept apart rather than shared out over the columns. The
1263/// stripe index page holds a section per column and could be split, and the directory and the
1264/// header cannot be, so splitting one of the three and not the others would read as if the columns
1265/// accounted for everything. They do not, and the gap is the thing worth looking at.
1266#[derive(Debug, Clone)]
1267pub struct Layout {
1268    /// The size of the file on disk.
1269    pub file: u64,
1270    /// Committed rows.
1271    pub rows: usize,
1272    /// Committed stripes.
1273    pub stripes: usize,
1274    /// Committed parts, which is how many chunks a scan reads.
1275    pub parts: usize,
1276    /// One entry per column, in the table's column order.
1277    pub columns: Vec<ColumnLayout>,
1278    /// Every stripe's index page, which carries a length and a checksum for every part of every
1279    /// column and is charged per stripe rather than per column.
1280    pub indexes: u64,
1281    /// The committed directory itself, the one that was read to build this.
1282    pub directory: u64,
1283    /// The fixed header, which holds the magic, the format and the two directory slots.
1284    pub header: u64,
1285}
1286
1287impl Layout {
1288    /// Everything the columns cost together.
1289    #[must_use]
1290    pub fn columns_total(&self) -> u64 {
1291        self.columns.iter().map(ColumnLayout::total).fold(0, u64::saturating_add)
1292    }
1293
1294    /// What the file holds that this does not account for.
1295    ///
1296    /// A committed file is written once and never rewritten in place, so an earlier directory and
1297    /// the pages of an earlier snapshot are still in it. That is the honest place for them: they
1298    /// are bytes on disk that no column owns.
1299    #[must_use]
1300    pub fn unaccounted(&self) -> u64 {
1301        self.file
1302            .saturating_sub(self.columns_total())
1303            .saturating_sub(self.indexes)
1304            .saturating_sub(self.directory)
1305            .saturating_sub(self.header)
1306    }
1307}
1308
1309/// How one part of one column is stored, which is one row of `pragma_storage_info`.
1310///
1311/// Everything here is read off the file rather than worked out from the schema, because the whole
1312/// question this answers is what the encoder chose, and the encoder chooses per part. Two files
1313/// holding the same rows in a different order give different answers and that difference is the
1314/// reason to ask.
1315///
1316/// The encoding costs a read of the column's page, so this is not free the way [`Layout`] is. It is
1317/// one read per column per stripe rather than one per part, because a part is a few kilobytes out
1318/// of a page that is a quarter of a megabyte.
1319#[derive(Debug, Clone)]
1320pub struct StoredPart {
1321    /// Which stripe the part belongs to.
1322    pub stripe: usize,
1323    /// Which part of that stripe it is, counting from zero inside the stripe.
1324    pub part: usize,
1325    /// The table wide row number the part starts at.
1326    pub row: usize,
1327    /// How many rows it holds.
1328    pub rows: usize,
1329    /// What the encoder made of it, as a line of text like `DICT(PACKED, PACKED)`.
1330    pub encoding: String,
1331    /// The stored bytes of the part, which is what it costs in the file.
1332    pub bytes: u64,
1333    /// Where in the file the column page holding this part starts.
1334    pub page: u64,
1335    /// Where in that page the part starts.
1336    pub offset: u64,
1337    /// The smallest value the part holds, when the stored ranges say.
1338    pub low: Option<Value>,
1339    /// The largest, same.
1340    pub high: Option<Value>,
1341    /// How many of its rows are null, when the stored ranges say.
1342    pub nulls: Option<usize>,
1343}
1344
1345/// Seeds the second hash a global dictionary tells its values apart by.
1346///
1347/// Any value that is not zero does, since zero is the seed [`checksum`] already uses and the point
1348/// is only that the two hashes of one value are not the same number. This one is the fractional part
1349/// of the golden ratio in sixty four bits, which is the constant everything else here is built out
1350/// of and is as good a nothing-up-my-sleeve number as any.
1351const DICTIONARY_CHECK_SEED: u64 = 11_400_714_819_323_198_485;
1352
1353/// One column's table wide dictionary while the load is running.
1354///
1355/// The thing to understand about this is what it does not hold. A dictionary of `URL` at a hundred
1356/// million ClickBench rows has about eighteen million distinct values and 1.3 GB of bytes in them,
1357/// and five columns like it are twelve of the seventeen gigabytes a load of `hits` peaks at. So the
1358/// bytes are not kept. A value's bytes go into [`GlobalDictionary::filling`], and when that reaches
1359/// [`TEXT_PAYLOAD_VALUES`] values the block is sealed, handed out at the end of the merge that
1360/// sealed it to be encoded with the stripe's pages, and never seen in that form again. What is left
1361/// is the encoded block, which is two to three times smaller, and that is the same bytes the file
1362/// is going to hold anyway.
1363///
1364/// Two things needed the raw bytes and neither needs them now. Deciding whether a value has been
1365/// seen before was a hash lookup and then a comparison of the bytes, and is now a hash lookup and a
1366/// comparison of a second hash under a different seed, which is [`DICTIONARY_CHECK_SEED`] and the
1367/// argument for why that is sound. Sorting the values at the end needed all of them at once, and
1368/// now reads the blocks back through [`GlobalDictionary::decoded`] one column at a time, which is
1369/// one column's bytes rather than every column's.
1370///
1371/// The offsets going block relative comes free with it, and takes the four gigabyte wall with it.
1372/// They were `u32` into a per column payload, so a column could not hold more than four gigabytes of
1373/// values however much memory the machine had, and `URL` and `Referer` are within a small factor of
1374/// that at a hundred million rows. A `u32` into a block of 1,024 values is not a bound anything real
1375/// reaches. The stored form is unchanged, because [`encode_offsets`] was already subtracting a per
1376/// block base before writing.
1377#[derive(Debug)]
1378struct GlobalDictionary {
1379    /// Keyed by the value's hash, which is already well spread, so the maps hash it once more
1380    /// with a multiply rather than with SipHash. SipHash here was one percent of a ClickBench load,
1381    /// and every stripe's merge of a column waits on the one before it.
1382    primary: HashMap<u64, u32, Spread>,
1383    collisions: HashMap<u64, Vec<u32>, Spread>,
1384    /// Every value's hash under [`DICTIONARY_CHECK_SEED`], in code order.
1385    checks: Vec<u64>,
1386    /// Where every value ends inside the payload block it is in, in code order.
1387    ends: Vec<u32>,
1388    counts: Vec<u64>,
1389    nulls: u64,
1390    /// The values of the block being filled, back to back.
1391    filling: Vec<u8>,
1392    /// One conservative four-byte substring signature per encoded payload block, in block order.
1393    ///
1394    /// Made where the block is encoded rather than where it is sealed, because sealing is under the
1395    /// writer's lock and every byte of every value going through [`gram_bits`] was 2.9 of the 14
1396    /// seconds the 10m ClickBench load spent on the 32 core box.
1397    grams: Vec<[u8; TEXT_GRAM_BYTES]>,
1398    /// Blocks that have filled and not been handed out to be encoded yet, each with its block number.
1399    ///
1400    /// Empty except inside the merge that filled them, and while the column is still too small to
1401    /// settle a shape on.
1402    waiting: Vec<(usize, Vec<u8>)>,
1403    /// Blocks kept raw to settle a shape on, spread across the column, each with its number.
1404    ///
1405    /// At most [`PAYLOAD_SAMPLE_BLOCKS`] of them and so at most a few megabytes. Spread rather than
1406    /// taken off the front for the reason [`settle_shape`] gives, and kept rather than read back
1407    /// because reading back is a decode and this is a sample of a column that is still growing.
1408    sample: Vec<(usize, Vec<u8>)>,
1409    /// How far apart the blocks in `sample` are, which doubles every time there are too many.
1410    stride: usize,
1411    /// What the blocks encoded so far were encoded with, once the column is big enough to settle it.
1412    shape: Option<chooser::Settled>,
1413    /// How many blocks had filled when that shape was settled.
1414    settled: usize,
1415    /// The blocks that are encoded and not yet in the file, in block order, following `placed`.
1416    ///
1417    /// Empty between stripes, because [`Writer::place_blocks`] writes them the moment they come
1418    /// back. Only a dictionary that never meets a writer, which is a test's, keeps them here.
1419    blocks: Vec<Vec<u8>>,
1420    /// Blocks that came back encoded ahead of a block before them, by block number.
1421    ///
1422    /// Two stripes merged one after the other can have their pages built in the other order, and a
1423    /// block cannot go into `blocks` until every block before it is there. They wait here until the
1424    /// gap closes, which is at most until the stripe merged just before this one is written.
1425    early: BTreeMap<usize, EncodedBlock>,
1426    /// Where every block already written to the file is, in block order.
1427    placed: Vec<Placed>,
1428    /// What the dictionary held the last time it was asked, see [`Self::recharge`], which is also
1429    /// what the load profile was told when there is one.
1430    charged: u64,
1431    /// Whether the dictionary stopped taking values, see [`Self::demote`].
1432    demoted: bool,
1433}
1434
1435/// Where one payload block of a global dictionary is in the file, and its checksum.
1436#[derive(Debug, Clone, Copy)]
1437struct Placed {
1438    start: u64,
1439    length: u64,
1440    hash: u64,
1441}
1442
1443/// Sorted `(head, code)` entries and the decoded bytes and block bases they were sorted over.
1444type RankedDictionary = (Vec<(u64, u32)>, Vec<u8>, Vec<u64>);
1445
1446impl GlobalDictionary {
1447    fn new() -> Self {
1448        Self {
1449            primary: HashMap::default(),
1450            collisions: HashMap::default(),
1451            checks: Vec::new(),
1452            ends: Vec::new(),
1453            counts: Vec::new(),
1454            nulls: 0,
1455            filling: Vec::new(),
1456            grams: Vec::new(),
1457            waiting: Vec::new(),
1458            sample: Vec::new(),
1459            stride: 1,
1460            shape: None,
1461            settled: 0,
1462            blocks: Vec::new(),
1463            early: BTreeMap::new(),
1464            placed: Vec::new(),
1465            charged: 0,
1466            demoted: false,
1467        }
1468    }
1469
1470    /// How many distinct values this dictionary holds, which is one past its largest code.
1471    fn values(&self) -> usize {
1472        self.ends.len()
1473    }
1474
1475    /// About how many bytes closing this dictionary holds at once: every value decoded, and a
1476    /// sort entry and a code for each.
1477    fn closing_bytes(&self) -> usize {
1478        let values = self.values();
1479        let decoded = (0..values.div_ceil(TEXT_PAYLOAD_VALUES))
1480            .map(|block| self.ends[((block + 1) * TEXT_PAYLOAD_VALUES).min(values) - 1] as usize)
1481            .sum::<usize>();
1482        decoded.saturating_add(values.saturating_mul(size_of::<(u64, u32)>() + size_of::<u32>()))
1483    }
1484
1485    /// About what the dictionary holds in memory, by capacity rather than by length.
1486    ///
1487    /// A hash table is charged its buckets, which is a power of two over eight sevenths of what it
1488    /// says it can hold, and a byte of control per bucket. The blocks waiting to be encoded and the
1489    /// ones kept to settle a shape on are counted one by one, and there are only ever a few.
1490    fn held_bytes(&self) -> u64 {
1491        fn table<K, V, S>(map: &HashMap<K, V, S>) -> usize {
1492            (map.capacity() * 8 / 7).next_power_of_two() * (size_of::<(K, V)>() + 1)
1493        }
1494        fn spilled<T>(values: &Vec<T>) -> usize {
1495            values.capacity() * size_of::<T>()
1496        }
1497        let raw = |blocks: &Vec<(usize, Vec<u8>)>| {
1498            spilled(blocks) + blocks.iter().map(|(_, block)| block.capacity()).sum::<usize>()
1499        };
1500        let bytes = table(&self.primary)
1501            + table(&self.collisions)
1502            + self.collisions.values().map(spilled).sum::<usize>()
1503            + spilled(&self.checks)
1504            + spilled(&self.ends)
1505            + spilled(&self.counts)
1506            + self.filling.capacity()
1507            + spilled(&self.grams)
1508            + raw(&self.waiting)
1509            + raw(&self.sample)
1510            + self.blocks.iter().map(Vec::capacity).sum::<usize>()
1511            + spilled(&self.placed);
1512        bytes as u64
1513    }
1514
1515    /// Tells `profile` what the dictionary has grown or shrunk by since the last time, and hands
1516    /// back what it held then and what it holds now.
1517    fn recharge(&mut self, profile: Option<&LoadProfile>) -> (u64, u64) {
1518        let before = self.charged;
1519        let now = self.held_bytes();
1520        if let Some(profile) = profile {
1521            if now >= before {
1522                profile.hold(now - before);
1523            } else {
1524                profile.release(before - now);
1525            }
1526        }
1527        self.charged = now;
1528        (before, now)
1529    }
1530
1531    /// Stops the dictionary taking values, for good.
1532    ///
1533    /// The block being filled is sealed so that it goes out with the others, and what the
1534    /// dictionary keeps for looking values up is let go of, which on a column of mostly new values
1535    /// is most of what it holds. What stays is what the close needs to write the dictionary's page:
1536    /// where every value ends, how often each was seen and where its blocks went. The stripes that
1537    /// were coded against it still need that page to be read. See [`DEMOTED`].
1538    fn demote(&mut self) {
1539        if self.demoted {
1540            return;
1541        }
1542        self.seal_rest();
1543        self.release_lookup();
1544        self.demoted = true;
1545    }
1546
1547    /// Frees what the dictionary keeps for coding new values, once none are coming.
1548    ///
1549    /// The hash tables, the check hash of every value and the blocks kept to settle a shape on are
1550    /// what a merge looks values up in. The close reads the counts, the ends and the written blocks
1551    /// and none of these, which are most of what the dictionary holds per value, so they go before
1552    /// the close takes memory of its own rather than after.
1553    fn release_lookup(&mut self) {
1554        self.primary = HashMap::default();
1555        self.collisions = HashMap::default();
1556        self.checks = Vec::new();
1557        self.sample = Vec::new();
1558        self.filling = Vec::new();
1559    }
1560
1561    /// How many blocks are encoded, written or not, which is the number the next one has to have.
1562    fn encoded(&self) -> usize {
1563        self.placed.len() + self.blocks.len()
1564    }
1565
1566    #[cfg(test)]
1567    fn code(&mut self, text: &str) -> Result<u32> {
1568        let bytes = text.as_bytes();
1569        self.code_hashed(bytes, checksum(bytes), seeded_checksum(bytes, DICTIONARY_CHECK_SEED))
1570    }
1571
1572    /// The code for a value whose two hashes the caller already has.
1573    ///
1574    /// A stripe prepared outside the writer's lock hashed every value it holds while it was coding
1575    /// them, and merging it into this dictionary is one of these a distinct value rather than two
1576    /// hashes of every row. See [`prepare`].
1577    fn code_hashed(&mut self, text: &[u8], hash: u64, check: u64) -> Result<u32> {
1578        if let Some(&code) = self.primary.get(&hash) {
1579            if self.checks.get(code as usize) == Some(&check) {
1580                return Ok(code);
1581            }
1582            if let Some(codes) = self.collisions.get(&hash) {
1583                if let Some(code) =
1584                    codes.iter().copied().find(|&code| self.checks[code as usize] == check)
1585                {
1586                    return Ok(code);
1587                }
1588            }
1589            let code = self.insert(text, check)?;
1590            self.collisions.entry(hash).or_default().push(code);
1591            return Ok(code);
1592        }
1593        let code = self.insert(text, check)?;
1594        self.primary.insert(hash, code);
1595        Ok(code)
1596    }
1597
1598    fn insert(&mut self, text: &[u8], check: u64) -> Result<u32> {
1599        if self.demoted {
1600            return Err(Error::internal("a value was coded against a demoted dictionary"));
1601        }
1602        let code = u32::try_from(self.ends.len())
1603            .map_err(|_| invalid("global dictionary has too many values"))?;
1604        self.filling.extend_from_slice(text);
1605        self.ends.push(
1606            u32::try_from(self.filling.len())
1607                .map_err(|_| invalid("a global dictionary value exceeds 4 GiB"))?,
1608        );
1609        self.checks.push(check);
1610        self.counts.push(0);
1611        if self.ends.len() % TEXT_PAYLOAD_VALUES == 0 {
1612            self.seal();
1613        }
1614        Ok(code)
1615    }
1616
1617    /// Closes the block being filled and puts it in the queue to be encoded.
1618    ///
1619    /// Also keeps a copy of it if it lands on the sample's stride, and halves the sample when that
1620    /// has left too many, which is what keeps the kept blocks spread evenly over however much of the
1621    /// column exists rather than bunched at whichever end was cheap to remember.
1622    fn seal(&mut self) {
1623        let at = self.ends.len().div_ceil(TEXT_PAYLOAD_VALUES) - 1;
1624        let bytes = std::mem::take(&mut self.filling);
1625        if at % self.stride == 0 {
1626            self.sample.push((at, bytes.clone()));
1627            if self.sample.len() > PAYLOAD_SAMPLE_BLOCKS {
1628                self.stride *= 2;
1629                let stride = self.stride;
1630                self.sample.retain(|(at, _)| at % stride == 0);
1631            }
1632        }
1633        self.waiting.push((at, bytes));
1634    }
1635
1636    /// The values of one block, as slices into the bytes the block was filled with.
1637    fn slices<'a>(&self, at: usize, bytes: &'a [u8]) -> Vec<&'a [u8]> {
1638        block_values(self.block_ends(at), bytes)
1639    }
1640
1641    /// Where every value of one block ends, relative to the block.
1642    fn block_ends(&self, at: usize) -> &[u32] {
1643        let first = (at * TEXT_PAYLOAD_VALUES).min(self.ends.len());
1644        let last = (first + TEXT_PAYLOAD_VALUES).min(self.ends.len());
1645        &self.ends[first..last]
1646    }
1647
1648    /// Takes every waiting block out to be encoded somewhere else, if the column has a shape to
1649    /// encode them with.
1650    ///
1651    /// This is what keeps the encoding out of the writer's lock. A block needs its bytes, where its
1652    /// values end and the shape, and nothing else of the dictionary, so it goes out with a copy of
1653    /// the four kilobytes of ends it has and comes back through [`GlobalDictionary::take_back`].
1654    fn hand_out(&mut self, column: usize) -> Vec<Unencoded> {
1655        let Some(shape) = &self.shape else { return Vec::new() };
1656        let waiting = std::mem::take(&mut self.waiting);
1657        waiting
1658            .into_iter()
1659            .map(|(at, bytes)| Unencoded {
1660                column,
1661                at,
1662                ends: self.block_ends(at).to_vec(),
1663                bytes,
1664                shape: shape.clone(),
1665            })
1666            .collect()
1667    }
1668
1669    /// Takes back one block that was handed out, and moves every block that is now next in line
1670    /// into `blocks`.
1671    fn take_back(&mut self, at: usize, block: EncodedBlock) -> Result<()> {
1672        if at < self.encoded() || self.early.insert(at, block).is_some() {
1673            return Err(Error::internal("a dictionary block came back twice"));
1674        }
1675        while let Some(block) = self.early.remove(&self.encoded()) {
1676            self.push_block(block);
1677        }
1678        Ok(())
1679    }
1680
1681    /// Appends the next encoded block and its signature.
1682    fn push_block(&mut self, (bytes, grams): EncodedBlock) {
1683        self.blocks.push(bytes);
1684        self.grams.push(*grams);
1685    }
1686
1687    /// Settles the shape the waiting blocks are about to be encoded with, if there is enough column
1688    /// to settle one on.
1689    ///
1690    /// Settled again once the column has grown fourfold, because the sample it was settled on then
1691    /// covered a quarter of what exists now and a dictionary in first seen order does not look the
1692    /// same at both ends. Blocks already encoded keep the shape they were encoded with. They can,
1693    /// because a block says what it is: nothing reading one asks the column what shape to expect.
1694    fn settle(&mut self) -> Result<()> {
1695        if self.sample.len() < PAYLOAD_SAMPLE_BLOCKS {
1696            return Ok(());
1697        }
1698        self.settle_on_sample()
1699    }
1700
1701    /// Settles a shape on whatever sample there is, for a column the load ended before it had
1702    /// enough of to settle one the usual way.
1703    ///
1704    /// Such a column has fewer than [`PAYLOAD_SAMPLE_BLOCKS`] blocks, so the sample is every block
1705    /// it has. Trying every candidate on each of them instead runs at two to six megabytes a second,
1706    /// and once `hits` stored its string columns with a dictionary, the forty or so small ones were
1707    /// more than half the CPU of a million row load, all of it in the close.
1708    fn settle_rest(&mut self) -> Result<()> {
1709        if self.shape.is_some() || self.sample.is_empty() {
1710            return Ok(());
1711        }
1712        self.settle_on_sample()
1713    }
1714
1715    fn settle_on_sample(&mut self) -> Result<()> {
1716        let complete = self.ends.len() / TEXT_PAYLOAD_VALUES;
1717        if self.shape.is_some() && complete < self.settled.saturating_mul(4) {
1718            return Ok(());
1719        }
1720        let sample =
1721            self.sample.iter().map(|(at, bytes)| self.slices(*at, bytes)).collect::<Vec<_>>();
1722        self.shape = Some(string::with_symbols(settle_shape(&sample)?, &sample));
1723        self.settled = complete;
1724        Ok(())
1725    }
1726
1727    /// Seals the part block at the end of the load, if there is one.
1728    fn seal_rest(&mut self) {
1729        // Asked of the values rather than of the bytes, because a block of empty strings has values
1730        // in it and no bytes, and a column of nulls is exactly that. A demoted dictionary sealed its
1731        // part block when it was demoted and has taken nothing since.
1732        if !self.demoted && self.ends.len() % TEXT_PAYLOAD_VALUES != 0 {
1733            self.seal();
1734        }
1735    }
1736
1737    /// Encodes the waiting block at `at`, with the settled shape when there is one and by trying
1738    /// everything when the column was too small to settle one.
1739    fn encode_waiting(&self, at: usize) -> Result<EncodedBlock> {
1740        let (block, bytes) = &self.waiting[at];
1741        let values = self.slices(*block, bytes);
1742        let encoded = match &self.shape {
1743            Some(shape) => string::encode_with(&values, shape)?,
1744            None => string::encode(&values)?,
1745        };
1746        Ok((encoded, block_grams(&values)))
1747    }
1748
1749    /// [`finish_dictionaries`] for one dictionary on this thread, for the tests that hold one.
1750    #[cfg(test)]
1751    fn finish_blocks(&mut self) -> Result<()> {
1752        self.seal_rest();
1753        let made = (0..self.waiting.len())
1754            .map(|at| self.encode_waiting(at))
1755            .collect::<Result<Vec<_>>>()?;
1756        for ((at, _), block) in std::mem::take(&mut self.waiting).into_iter().zip(made) {
1757            if self.encoded() != at {
1758                return Err(Error::internal("a dictionary block was encoded out of order"));
1759            }
1760            self.push_block(block);
1761        }
1762        Ok(())
1763    }
1764
1765    /// Every value of this dictionary read back out of its encoded blocks, as the bytes back to back
1766    /// and where each block starts in them.
1767    ///
1768    /// This is the one place the whole column is in memory at once and the reason [`Writer::close`]
1769    /// takes the columns one at a time rather than across threads. One column's values is 1.3 GB on
1770    /// the worst ClickBench column, and five columns of that at once is the peak this was all meant
1771    /// to remove.
1772    ///
1773    /// The blocks are spread over threads instead. Each block's decoded length is already known from
1774    /// the ends of its values, so the answer is laid out before anything is decoded and every thread
1775    /// decodes its own run of blocks straight into its own part of it. On the 10m ClickBench sample
1776    /// this was a second of the close for `URL` alone, on one core of thirty two, and the close is
1777    /// what a load waits on once its stripes are written.
1778    ///
1779    /// The blocks already written are read back out of `file`, so what a thread holds beyond the
1780    /// answer is one encoded block. They were written moments or minutes ago and are almost always
1781    /// still in the page cache, so this is a copy rather than a read of the disk.
1782    fn decoded(&self, file: Option<&dyn rudb_io::File>) -> Result<(Vec<u8>, Vec<u64>)> {
1783        let count = self.placed.len() + self.blocks.len();
1784        if count != self.values().div_ceil(TEXT_PAYLOAD_VALUES) {
1785            return Err(invalid("global dictionary blocks do not cover its values"));
1786        }
1787        let mut bases = Vec::with_capacity(count);
1788        let mut total = 0_usize;
1789        for block in 0..count {
1790            bases.push(total as u64);
1791            let last = ((block + 1) * TEXT_PAYLOAD_VALUES).min(self.values()) - 1;
1792            total = total
1793                .checked_add(self.ends[last] as usize)
1794                .ok_or_else(|| invalid("global dictionary does not fit in memory"))?;
1795        }
1796        let mut flat = vec![0_u8; total];
1797        let mut outs = Vec::with_capacity(count);
1798        let mut rest = flat.as_mut_slice();
1799        for block in 0..count {
1800            let end = bases.get(block + 1).map_or(total, |&base| base as usize);
1801            let (out, after) = rest.split_at_mut(end - bases[block] as usize);
1802            outs.push((block, out));
1803            rest = after;
1804        }
1805        let one = |run: &mut [(usize, &mut [u8])]| -> Result<()> {
1806            let mut stored = Vec::new();
1807            for (block, out) in run {
1808                let encoded = match self.placed.get(*block) {
1809                    Some(place) => {
1810                        let file = file.ok_or_else(|| {
1811                            Error::internal("a written dictionary block has no file")
1812                        })?;
1813                        let length = usize::try_from(place.length).map_err(|_| {
1814                            invalid("global dictionary block does not fit in memory")
1815                        })?;
1816                        stored.resize(length, 0);
1817                        read_at(file, place.start, &mut stored)?;
1818                        if checksum(&stored) != place.hash {
1819                            return Err(invalid(
1820                                "a global dictionary block did not read back as written",
1821                            ));
1822                        }
1823                        stored.as_slice()
1824                    }
1825                    None => &self.blocks[*block - self.placed.len()],
1826                };
1827                let decoded = string::decode_flat(encoded)?;
1828                if decoded.bytes().len() != out.len() {
1829                    return Err(invalid(
1830                        "a global dictionary block is not the length its ends say",
1831                    ));
1832                }
1833                out.copy_from_slice(decoded.bytes());
1834            }
1835            Ok(())
1836        };
1837        // Sixteen blocks a thread at the least, because a thread costs about what decoding a few
1838        // blocks does and most columns have one or two.
1839        let workers = close_workers().min(count / 16).max(1);
1840        if workers <= 1 {
1841            one(&mut outs)?;
1842        } else {
1843            let per = count.div_ceil(workers);
1844            std::thread::scope(|scope| {
1845                outs.chunks_mut(per)
1846                    .map(|run| scope.spawn(|| one(run)))
1847                    .collect::<Vec<_>>()
1848                    .into_iter()
1849                    .try_for_each(|handle| {
1850                        handle.join().map_err(|_| {
1851                            Error::internal("a global dictionary decode worker panicked")
1852                        })?
1853                    })
1854            })?;
1855        }
1856        drop(outs);
1857        Ok((flat, bases))
1858    }
1859
1860    /// Where the value at `code` sits in the bytes [`GlobalDictionary::decoded`] handed back.
1861    ///
1862    /// A block's first value starts at the block, and every other value starts where the one before
1863    /// it ended, which is what makes 1,024 values 1,024 numbers rather than 1,025.
1864    fn value_span(ends: &[u32], bases: &[u64], code: usize) -> (usize, usize) {
1865        let Some(&base) = bases.get(code / TEXT_PAYLOAD_VALUES) else { return (0, 0) };
1866        let Some(&end) = ends.get(code) else { return (0, 0) };
1867        let base = base as usize;
1868        let from = if code % TEXT_PAYLOAD_VALUES == 0 { 0 } else { ends[code - 1] as usize };
1869        (base + from, base + end as usize)
1870    }
1871
1872    /// This dictionary's values in sorted order, each as the first eight bytes of the value and the
1873    /// code that holds it, so entry `rank` describes the value that sits at `rank` when the values
1874    /// are sorted by their bytes.
1875    ///
1876    /// Codes themselves stay in first appearance order, which is what lets the writer hand one out
1877    /// the moment it sees a value rather than waiting for the last stripe, and which also keeps a
1878    /// stripe's codes close together because the data is clustered. This is what puts the values
1879    /// back in order for anything that needs it, and it is separate from the codes so that getting
1880    /// it costs a sort of the distinct values at the end rather than a rewrite of every code page.
1881    ///
1882    /// The order is the byte order of the values and nothing else. The heads are attached after the
1883    /// sort rather than sorted on, because padding with zero on the right is order preserving for
1884    /// byte strings and so sorting by head and then by bytes lands in the same place as sorting by
1885    /// bytes: a shorter value differs from a longer one that starts the same way at a position
1886    /// where the shorter one has run out, and zero is below every byte that could be there.
1887    ///
1888    /// The heads are kept because a reader searching this order wants a comparison it can make out
1889    /// of the index alone. What they buy there depends entirely on the column and is much less than
1890    /// it looks on the columns that cost the most, which [`sort_by_value`] measures.
1891    fn ranked_with_values(&self, file: Option<&dyn rudb_io::File>) -> Result<RankedDictionary> {
1892        let (flat, bases) = self.decoded(file)?;
1893        let value = |code: u32| {
1894            let (from, to) = Self::value_span(&self.ends, &bases, code as usize);
1895            flat.get(from..to).unwrap_or_default()
1896        };
1897        let mut codes = (0..self.values() as u32).collect::<Vec<_>>();
1898        sort_by_value_across(&mut codes, value, close_workers());
1899        let order = codes.into_iter().map(|code| (head(value(code)), code)).collect();
1900        Ok((order, flat, bases))
1901    }
1902
1903    #[cfg(test)]
1904    fn ranked(&self, file: Option<&dyn rudb_io::File>) -> Result<Vec<(u64, u32)>> {
1905        self.ranked_with_values(file).map(|(order, _, _)| order)
1906    }
1907}
1908
1909/// Appends pages and commits a new directory.
1910///
1911/// One writer covers a whole file rather than one table. [`Writer::next`] closes the table it is on
1912/// and opens another over the same file, and [`Writer::finish`] commits every table it has closed in
1913/// one generation. That is what makes a checkpoint atomic across tables: there is one slot write at
1914/// the end of it and a reader sees every table at the generation before it or every table at the
1915/// generation after it.
1916#[derive(Debug)]
1917pub struct Writer {
1918    /// The file, through `rudb-io` rather than `std::fs`, so that a test can hand the writer a
1919    /// simulated filesystem and crash a load at every call it makes.
1920    file: Box<dyn rudb_io::File>,
1921    /// Where the next write goes, counted here rather than asked of the file.
1922    ///
1923    /// The file's own cursor is not ours. Building the numeric frequencies reads pages back through
1924    /// [`read_at`], and a positional read is only positional about where it reads from: `pread`
1925    /// leaves the cursor alone, and the call Windows has for it moves the cursor to the end of what
1926    /// it read. A writer that asked the file where it was would then write the directory over a
1927    /// page it had already written, which is what it did.
1928    at: u64,
1929    /// How far into the file the kernel has been asked to start writing, see [`WRITEBACK_STRETCH`].
1930    written_back: u64,
1931    table: Table,
1932    generation: u64,
1933    /// The first and the last source position in every stripe, in the order the stripes were
1934    /// written.
1935    order: Vec<((u64, u64), (u64, u64))>,
1936    next_order: u64,
1937    dictionaries: Vec<Option<GlobalDictionary>>,
1938    /// Which columns still have a global dictionary, shared with every [`Preparer`] this writer
1939    /// hands out so that a stripe prepared after a column lost its dictionary is not coded for it.
1940    coded: Arc<prepare::Coding>,
1941    /// One per column, folding the rows into a summary and a sketch as they go past.
1942    ///
1943    /// `None` for a column with no hash rule, which is the interval and the nested types. See
1944    /// [`stats::Gather`] for why the statistics are built here rather than by reading the file back
1945    /// once it is committed.
1946    gathers: Vec<Option<stats::Gather>>,
1947    /// The dictionaries and the statistics while a [`Merger`] has them, which is from
1948    /// [`Writer::merger`] until the table is closed. `dictionaries` and `gathers` are empty then.
1949    lent: Option<Arc<Lent>>,
1950    pending: Vec<PendingChunk>,
1951    /// The tables already closed in this generation, in the order they were written.
1952    closed: Vec<Entry>,
1953    /// The views the next commit writes down, which [`Writer::with_views`] sets.
1954    ///
1955    /// Carried forward from the committed generation by [`Writer::open`], so a writer that was only
1956    /// opened to append a table does not have to know about views to avoid dropping them.
1957    views: Vec<ViewEntry>,
1958    /// Where the stages this writer runs are charged, which [`Writer::with_profile`] sets.
1959    ///
1960    /// The writer runs the page builder, the dictionary blocks, the writes and the publish, and it
1961    /// charges them once per stripe and once per worker, never per chunk. See
1962    /// `rudb_metrics::LoadProfile` for why that is the grain.
1963    profile: Option<Arc<LoadProfile>>,
1964}
1965
1966/// A chunk that has arrived and is waiting for the rest of its stripe.
1967///
1968/// The rows are kept rather than the pages they encode to, which is the whole of #808's first half.
1969/// Encoding on arrival put every column of every part on the thread that called `append_at`, and
1970/// that thread is the only one the load has. Encoding at the flush instead means a stripe's worth
1971/// of work is on the table at once, and a stripe splits by column into a hundred and five pieces
1972/// that share nothing.
1973#[derive(Debug)]
1974struct PendingChunk {
1975    order: (u64, u64),
1976    chunk: Chunk,
1977}
1978
1979/// What the writer still needs of a part once its columns are encoded: where in the source it came
1980/// from, how many rows it has and how large those rows were.
1981///
1982/// A stripe waiting for the writer's lock carries these rather than its chunks, so its rows are
1983/// freed as soon as they are encoded and not after the stripe is written. See [`prepare`].
1984#[derive(Debug, Clone, Copy)]
1985struct Part {
1986    order: (u64, u64),
1987    rows: usize,
1988    footprint: usize,
1989}
1990
1991impl Part {
1992    fn of(pending: &PendingChunk) -> Self {
1993        Self {
1994            order: pending.order,
1995            rows: pending.chunk.len(),
1996            footprint: pending.chunk.footprint(),
1997        }
1998    }
1999}
2000
2001/// One column's share of a stripe, which is what one encode worker produces.
2002///
2003/// Indexed by part, so a stripe is a column of these and the write loop reads down one of them.
2004/// That is also the order the loop wanted: `flush_pending` walks a column at a time and lays its
2005/// parts next to each other, and it used to reach across a row of parts to do it.
2006#[derive(Debug, Default)]
2007struct ColumnStripe {
2008    pages: Vec<Vec<u8>>,
2009    codes: Vec<Option<Vec<u32>>>,
2010    sieves: Vec<Option<Sieve>>,
2011    ranges: Vec<Range>,
2012}
2013
2014/// Whether a column of this type is coded against a global dictionary.
2015///
2016/// A dictionary, its codes and the membership index beside them are about bytes and not about
2017/// text, so a blob gets one the same as a varchar does. ClickBench's `hits.parquet` stores every
2018/// string column as a plain byte array, which reads back as a blob, and those columns were being
2019/// written as a length and the bytes for every row: 533 MB for the first million rows where DuckDB
2020/// writes 142.
2021fn coded_type(ty: &LogicalType) -> bool {
2022    matches!(ty, LogicalType::Varchar | LogicalType::Blob)
2023}
2024
2025/// The tag a directory gives a column's global dictionary.
2026///
2027/// A varchar's is 1, as it always was. A blob's is 2, so that a reader from before blobs had
2028/// dictionaries meets a tag it does not know and refuses the file, rather than laying the rest of
2029/// the directory out as if the blob columns had no dictionary and reading everything after the
2030/// first one from the wrong place.
2031fn dictionary_tag(ty: &LogicalType) -> u8 {
2032    if ty == &LogicalType::Blob { 2 } else { 1 }
2033}
2034
2035/// Roughly what encoding a column of this type costs, for ordering the encode queue.
2036///
2037/// Only the order matters and only roughly. A string column hashes and copies every value into a
2038/// dictionary and is in a different class from everything else, and among the fixed widths the wide
2039/// ones carry more bytes through the cascade than the narrow ones. Anything finer than that would
2040/// be a cost model, and the queue already absorbs a wrong guess: it only has to avoid finishing on
2041/// a column nobody else can help with.
2042fn weight(ty: &LogicalType) -> usize {
2043    match ty {
2044        LogicalType::Varchar | LogicalType::Blob | LogicalType::Bit => 64,
2045        LogicalType::HugeInt
2046        | LogicalType::UHugeInt
2047        | LogicalType::Uuid
2048        | LogicalType::Interval => 16,
2049        LogicalType::BigInt
2050        | LogicalType::UBigInt
2051        | LogicalType::Timestamp
2052        | LogicalType::Time
2053        | LogicalType::TimeTz
2054        | LogicalType::TimestampTz
2055        | LogicalType::TimestampS
2056        | LogicalType::TimestampMs
2057        | LogicalType::TimestampNs
2058        | LogicalType::Double
2059        | LogicalType::Decimal { .. } => 8,
2060        LogicalType::Integer | LogicalType::UInteger | LogicalType::Date | LogicalType::Float => 4,
2061        LogicalType::SmallInt | LogicalType::USmallInt => 2,
2062        _ => 1,
2063    }
2064}
2065
2066/// Parts in one stripe.
2067///
2068/// Sixty four thousand rows is the smallest stripe that keeps the ClickBench directory in single
2069/// digit megabytes at a hundred million rows, and it puts a four byte column's page at a quarter of
2070/// a megabyte, which is the size a sequential read wants. Larger stripes buy a smaller directory
2071/// and cost a sparse fetch, which has to read a page index before it can reach one part.
2072pub const STRIPE_PARTS: usize = 64;
2073
2074/// How many rows the writer wants to see before it decides whether a varchar column gets to keep
2075/// its global dictionary.
2076///
2077/// See [`prepare::drops_dictionary`]. A stripe is up to [`STRIPE_PARTS`] parts, so most tables give it
2078/// far more than this and it binds only on a table that is smaller than one stripe. A handful of
2079/// rows says nothing about whether a column repeats itself, and the answer that costs nothing when
2080/// the sample is that small is the one the writer has always given, which is to keep the dictionary.
2081const DICTIONARY_DECIDE_ROWS: usize = 4_096;
2082
2083/// Out of ten. A varchar column loses its dictionary when more than this many rows in ten of the
2084/// first stripe held a value that stripe had not seen before.
2085///
2086/// See [`prepare::drops_dictionary`]. Nine and not five, because the properties a dictionary buys are
2087/// worth keeping everywhere they are real. On ClickBench the widest string column is `Referer` at
2088/// 0.131 of its first stripe and every other one is below that, so nothing there is near this and
2089/// every one of them keeps its dictionary, which is what a group by on codes wants. On TPC-H
2090/// `o_comment` and `c_comment` are at 0.97 and are what this catches.
2091///
2092/// `l_comment` sits at 0.883 and so keeps its dictionary. Eight was built and measured rather than
2093/// argued about, and it is not a clear win: it takes `select l_comment from lineitem` from 4.335 G
2094/// instructions to 3.473 G and the file from 280.2 MB to 260.4 MB, and it takes a `like` over the
2095/// same column from 3.29 G to 4.27 G, because a dictionary runs the predicate once a distinct value
2096/// and there are 3.6 M of those to 6.0 M rows. The 22 query suite came out 6.91 s against 7.07 s in
2097/// favour of nine. So nine stays until there is a reason to prefer one of those shapes. See #1137.
2098const DICTIONARY_DISTINCT_IN_TEN: usize = 9;
2099
2100/// Bytes one part takes in a stripe's index page: four for the length, eight for the checksum.
2101const INDEX_ENTRY: usize = size_of::<u32>() + size_of::<u64>();
2102
2103/// Bytes one column's section of a stripe's index page takes, including its own trailing checksum.
2104fn index_section(parts: usize) -> Result<usize> {
2105    parts
2106        .checked_mul(INDEX_ENTRY)
2107        .and_then(|bytes| bytes.checked_add(size_of::<u64>()))
2108        .ok_or_else(|| invalid("index page length overflow"))
2109}
2110
2111impl Writer {
2112    /// Opens a committed file and starts a table in the generation after the one it holds.
2113    ///
2114    /// The tables already in the file are carried forward by name and by directory pointer, and
2115    /// their pages are not read. Nothing in the file is overwritten: the new table's pages and the
2116    /// new catalog go on the end, past the catalog the committed generation points at, and the one
2117    /// write that is not an append is the slot in the header that [`Writer::finish`] does last.
2118    ///
2119    /// That slot is the other one. A file committed at generation 1 is named by the slot at 16 and
2120    /// generation 2 writes the one at 44, so until the last four bytes of the commit land the file
2121    /// still reads as the generation before it, and a slot torn across a write fails its checksum
2122    /// and the reader falls back to the one beside it. This is what the second slot has always been
2123    /// for.
2124    ///
2125    /// # Errors
2126    ///
2127    /// If the file has no valid committed directory, is not this build's format, repeats the name
2128    /// of a table already in it that holds rows, has a field with no scalar encoding, or cannot be
2129    /// written.
2130    pub fn open(
2131        path: impl AsRef<Path>,
2132        name: impl Into<String>,
2133        fields: Vec<Field>,
2134    ) -> Result<Self> {
2135        Self::open_in(&RealFilesystem::new(), path, name, fields)
2136    }
2137
2138    /// [`Writer::open`] on a file in `fs`, which is how a crash test runs an append against the
2139    /// simulated filesystem.
2140    ///
2141    /// # Errors
2142    ///
2143    /// The same as [`Writer::open`].
2144    pub fn open_in(
2145        fs: &dyn Filesystem,
2146        path: impl AsRef<Path>,
2147        name: impl Into<String>,
2148        fields: Vec<Field>,
2149    ) -> Result<Self> {
2150        for field in &fields {
2151            type_tag(&field.ty)?;
2152        }
2153        let name = name.into();
2154        let file = fs.open(path.as_ref(), OpenMode::ReadWrite)?;
2155        let size = file.len()?;
2156        let (slot, bytes, _) = committed_slot(&*file, size)?;
2157        let (mut closed, views) = decode_catalog(&bytes, size)?;
2158        // A table already in the file under this name is only in the way if it holds rows. One that
2159        // holds none has no pages for this generation to carry and no reader that could lose
2160        // anything, so the table being started here takes its place in the catalog rather than
2161        // colliding with it, and `finish` writes the new entry where the old one was.
2162        //
2163        // That is not a corner. It is the shape every loading script writes: the schema goes in one
2164        // statement and the rows go in the next, and a checkpoint between them commits the empty
2165        // table. Before this, the second statement had to build the whole table in memory because
2166        // the first had already put the name in the file, which is how a load of a table larger
2167        // than memory became a load that needed memory the size of the table.
2168        if let Some(at) = closed.iter().position(|held| held.name == name) {
2169            if closed[at].rows > 0 {
2170                return Err(invalid("two tables in one native file have the same name"));
2171            }
2172            closed.remove(at);
2173        }
2174        // The generation of the slot whose bytes checksummed, and not the highest number in the
2175        // header. A slot torn across a write can hold any number at all, and taking that one would
2176        // be choosing which slot to overwrite from a value nothing has vouched for, which is how a
2177        // half written commit gets to destroy the one good copy beside it.
2178        let generation = slot
2179            .generation
2180            .checked_add(1)
2181            .ok_or_else(|| invalid("native file generation overflow"))?;
2182        Ok(Self {
2183            file,
2184            // The end of the file, so that the committed generation's catalog stays where its slot
2185            // says it is and keeps naming a file a reader can still open.
2186            at: size,
2187            written_back: size,
2188            dictionaries: fields
2189                .iter()
2190                .map(|field| coded_type(&field.ty).then(GlobalDictionary::new))
2191                .collect(),
2192            coded: Arc::new(prepare::Coding::new(fields.iter().map(|field| coded_type(&field.ty)))),
2193            gathers: fields.iter().map(|field| stats::Gather::new(&field.ty, generation)).collect(),
2194            lent: None,
2195            table: Table {
2196                name,
2197                dictionaries: vec![None; fields.len()],
2198                dictionary_payloads: Vec::new(),
2199                demoted: Vec::new(),
2200                distincts: vec![None; fields.len()],
2201                fields,
2202                stripes: Vec::new(),
2203                rows: 0,
2204                frequencies: Vec::new(),
2205                pair_frequencies: Vec::new(),
2206                frequency_texts: Vec::new(),
2207                host_groups: None,
2208                clustering: None,
2209                generation,
2210                sections: Vec::new(),
2211            },
2212            generation,
2213            order: Vec::new(),
2214            next_order: 0,
2215            pending: Vec::with_capacity(STRIPE_PARTS),
2216            closed,
2217            views,
2218            profile: None,
2219        })
2220    }
2221
2222    /// Creates a new v10 file and its first table.
2223    ///
2224    /// # Errors
2225    ///
2226    /// If the file exists, a field has no scalar encoding, or the path cannot be written.
2227    pub fn create(
2228        path: impl AsRef<Path>,
2229        name: impl Into<String>,
2230        fields: Vec<Field>,
2231    ) -> Result<Self> {
2232        Self::create_in(&RealFilesystem::new(), path, name, fields)
2233    }
2234
2235    /// [`Writer::create`] with the file made in `fs` rather than on the real filesystem.
2236    ///
2237    /// Every call the writer makes on the file from here to [`Writer::finish`] goes to that
2238    /// filesystem, which is what lets a test built on `rudb_io::SimFilesystem` stop a load at any
2239    /// one of them and look at what a crash there would leave on the disk.
2240    ///
2241    /// # Errors
2242    ///
2243    /// The same as [`Writer::create`].
2244    pub fn create_in(
2245        fs: &dyn Filesystem,
2246        path: impl AsRef<Path>,
2247        name: impl Into<String>,
2248        fields: Vec<Field>,
2249    ) -> Result<Self> {
2250        for field in &fields {
2251            type_tag(&field.ty)?;
2252        }
2253        let file = fs.open(path.as_ref(), OpenMode::CreateNew)?;
2254        let mut header = [0; HEADER as usize];
2255        header[..8].copy_from_slice(MAGIC);
2256        header[8..12].copy_from_slice(&FORMAT.to_le_bytes());
2257        file.write_at(0, &header)?;
2258        Ok(Self {
2259            file,
2260            at: HEADER,
2261            written_back: HEADER,
2262            dictionaries: fields
2263                .iter()
2264                .map(|field| coded_type(&field.ty).then(GlobalDictionary::new))
2265                .collect(),
2266            coded: Arc::new(prepare::Coding::new(fields.iter().map(|field| coded_type(&field.ty)))),
2267            gathers: fields.iter().map(|field| stats::Gather::new(&field.ty, 1)).collect(),
2268            lent: None,
2269            table: Table {
2270                name: name.into(),
2271                dictionaries: vec![None; fields.len()],
2272                dictionary_payloads: Vec::new(),
2273                demoted: Vec::new(),
2274                distincts: vec![None; fields.len()],
2275                fields,
2276                stripes: Vec::new(),
2277                rows: 0,
2278                frequencies: Vec::new(),
2279                pair_frequencies: Vec::new(),
2280                frequency_texts: Vec::new(),
2281                host_groups: None,
2282                clustering: None,
2283                generation: 1,
2284                sections: Vec::new(),
2285            },
2286            generation: 1,
2287            order: Vec::new(),
2288            next_order: 0,
2289            pending: Vec::with_capacity(STRIPE_PARTS),
2290            closed: Vec::new(),
2291            views: Vec::new(),
2292            profile: None,
2293        })
2294    }
2295
2296    /// Creates a new file that holds no table at all, committed and ready to open.
2297    ///
2298    /// A database somebody dropped the last table out of is still a database, and until this there
2299    /// was no way to write one down. Every other way into this file goes through a table, because
2300    /// [`Writer::create`] takes the first one and [`Writer::finish`] commits the one it is on, so a
2301    /// catalog with nothing in it could be read and not written. The format already allowed it: the
2302    /// catalog is a count and that many entries, and a count of nought encodes and decodes the same
2303    /// way every other count does, which is why nothing here is a version change.
2304    ///
2305    /// It hands back nothing rather than a writer, because a writer with no table is a writer with
2306    /// nothing to append to. A file that is going to hold a table is [`Writer::create`], and one
2307    /// that is going to have a table added to it later is [`Writer::open`], which reads what this
2308    /// wrote the same way it reads any other generation.
2309    ///
2310    /// It takes the views anyway, because a database with no table can still have views in it. A
2311    /// view over `range` or over another view names no table, so dropping the last table out of a
2312    /// database does not have to leave the catalog with nothing worth writing down.
2313    ///
2314    /// # Errors
2315    ///
2316    /// If the file exists or the path cannot be written.
2317    pub fn empty(path: impl AsRef<Path>, views: &[ViewEntry]) -> Result<()> {
2318        let file = RealFilesystem::new().open(path.as_ref(), OpenMode::CreateNew)?;
2319        let mut header = [0; HEADER as usize];
2320        header[..8].copy_from_slice(MAGIC);
2321        header[8..12].copy_from_slice(&FORMAT.to_le_bytes());
2322        file.write_at(0, &header)?;
2323        let catalog = encode_catalog(&[], views)?;
2324        file.write_at(HEADER, &catalog)?;
2325        // The same two syncs in the same order as [`Writer::finish`], and for the same reason. The
2326        // catalog is on the disk before the slot names it, so a file this is interrupted in the
2327        // middle of is a header with no valid slot rather than a slot pointing at nothing.
2328        file.sync()?;
2329        let slot = Slot {
2330            offset: HEADER,
2331            length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
2332            generation: 1,
2333            hash: checksum(&catalog),
2334        };
2335        file.write_at(slot_offset(1), &slot.bytes())?;
2336        file.sync()?;
2337        Ok(())
2338    }
2339
2340    /// Closes the table this writer is on and starts another one in the same file.
2341    ///
2342    /// Nothing is published here. The closed table's directory is written so that the bytes are on
2343    /// disk and its span is known, and the catalog that names it is only written by
2344    /// [`Writer::finish`], so a crash between two tables leaves the previous generation intact.
2345    ///
2346    /// # Errors
2347    ///
2348    /// If the name repeats a table already closed, a field has no scalar encoding, or the table
2349    /// being closed cannot be written.
2350    pub fn next(mut self, name: impl Into<String>, fields: Vec<Field>) -> Result<Self> {
2351        for field in &fields {
2352            type_tag(&field.ty)?;
2353        }
2354        let name = name.into();
2355        let entry = self.close()?;
2356        if self.closed.iter().chain(std::iter::once(&entry)).any(|held| held.name == name) {
2357            return Err(invalid("two tables in one native file have the same name"));
2358        }
2359        let Self { file, at, generation, mut closed, views, .. } = self;
2360        closed.push(entry);
2361        Ok(Self {
2362            file,
2363            written_back: at,
2364            at,
2365            generation,
2366            closed,
2367            views,
2368            profile: None,
2369            dictionaries: fields
2370                .iter()
2371                .map(|field| coded_type(&field.ty).then(GlobalDictionary::new))
2372                .collect(),
2373            coded: Arc::new(prepare::Coding::new(fields.iter().map(|field| coded_type(&field.ty)))),
2374            gathers: fields.iter().map(|field| stats::Gather::new(&field.ty, generation)).collect(),
2375            lent: None,
2376            table: Table {
2377                name,
2378                dictionaries: vec![None; fields.len()],
2379                dictionary_payloads: Vec::new(),
2380                demoted: Vec::new(),
2381                distincts: vec![None; fields.len()],
2382                fields,
2383                stripes: Vec::new(),
2384                rows: 0,
2385                frequencies: Vec::new(),
2386                pair_frequencies: Vec::new(),
2387                frequency_texts: Vec::new(),
2388                host_groups: None,
2389                clustering: None,
2390                generation,
2391                sections: Vec::new(),
2392            },
2393            order: Vec::new(),
2394            next_order: 0,
2395            pending: Vec::with_capacity(STRIPE_PARTS),
2396        })
2397    }
2398
2399    /// Sets the views the next commit writes down, replacing whatever was carried forward.
2400    ///
2401    /// It replaces rather than adds because the caller has the whole catalog in front of it and the
2402    /// writer does not. A view that was dropped is a view that is not in the list any more, and
2403    /// there is no other way for the writer to hear about that, since nothing else it is told about
2404    /// mentions views at all.
2405    ///
2406    /// A writer that is never told anything writes back the views it read at [`Writer::open`], so a
2407    /// checkpoint that only had a table to append does not quietly drop them.
2408    #[must_use]
2409    pub fn with_views(mut self, views: Vec<ViewEntry>) -> Self {
2410        self.views = views;
2411        self
2412    }
2413
2414    /// Charges the stages this writer runs to `profile`.
2415    ///
2416    /// For the table being written now. [`Writer::next`] starts the next table without one,
2417    /// because a second table's stripes charged to the first table's load would be a profile of
2418    /// neither.
2419    #[must_use]
2420    pub fn with_profile(mut self, profile: Arc<LoadProfile>) -> Self {
2421        self.profile = Some(profile);
2422        self
2423    }
2424
2425    /// Sets what the table's global dictionaries may hold between them before the one growing
2426    /// fastest stops taking values, which is [`DICTIONARY_CAP_BYTES`] unless this says
2427    /// otherwise. It applies to every [`Preparer`] and [`Merger`] this writer has handed out too.
2428    #[must_use]
2429    pub fn with_dictionary_cap(self, bytes: u64) -> Self {
2430        self.coded.cap(bytes);
2431        self
2432    }
2433
2434    /// Records the order this table's rows are meant to be stored in.
2435    ///
2436    /// The declaration goes in the table directory and comes back out of
2437    /// [`Table::clustering`]. Nothing here sorts anything, and nothing here checks that the rows
2438    /// handed to [`Writer::append`] arrive in the order this claims. That is deliberate for now:
2439    /// the thing that was missing was a place to write the order down, and a loader that honours
2440    /// the declaration is the next piece rather than this one.
2441    ///
2442    /// The declaration applies to the table the writer is currently on, so it is set after
2443    /// [`Writer::next`] rather than once for the file.
2444    ///
2445    /// # Errors
2446    ///
2447    /// If the declaration names a column this table does not have.
2448    pub fn declare(mut self, clustering: Clustering) -> Result<Self> {
2449        // Rebuilt against this table's own column count rather than trusted, because the caller
2450        // built it against a catalog entry and the two could have drifted.
2451        self.table.clustering = Some(Clustering::new(
2452            clustering.columns().to_vec(),
2453            clustering.width(),
2454            &self.table.fields,
2455        )?);
2456        Ok(self)
2457    }
2458
2459    /// Appends bytes at the end of the file and moves the writer's own offset past them.
2460    ///
2461    /// Every write in here goes through this, so that [`Writer::at`] is the only answer to where
2462    /// anything is and the file's cursor is never consulted for it.
2463    fn put(&mut self, bytes: &[u8]) -> Result<()> {
2464        self.file.write_at(self.at, bytes)?;
2465        self.at = self
2466            .at
2467            .checked_add(bytes.len() as u64)
2468            .ok_or_else(|| invalid("native file length overflow"))?;
2469        if self.at - self.written_back >= WRITEBACK_STRETCH {
2470            self.file.start_writeback(self.written_back, self.at - self.written_back);
2471            self.written_back = self.at;
2472        }
2473        Ok(())
2474    }
2475
2476    /// Writes one chunk as independently readable column pages.
2477    ///
2478    /// # Errors
2479    ///
2480    /// If its width or types differ from the declared table, or a page exceeds its bound.
2481    pub fn append(&mut self, chunk: &Chunk) -> Result<()> {
2482        let order = (self.next_order, 0);
2483        self.next_order = self.next_order.saturating_add(1);
2484        self.append_at(order, chunk)
2485    }
2486
2487    /// Writes one chunk and records its source position for directory ordering.
2488    ///
2489    /// Pages may be encoded by parallel pipeline instances and reach the file in completion order.
2490    /// The stripe they land in is sorted by this key at commit, and [`Self::finish`] rejects a
2491    /// sequence whose parts do not come out in source order once the stripes are sorted, because a
2492    /// stripe groups whatever arrived together and cannot put a late part back where it belongs.
2493    ///
2494    /// # Errors
2495    ///
2496    /// The same as [`Self::append`].
2497    pub fn append_at(&mut self, order: (u64, u64), chunk: &Chunk) -> Result<()> {
2498        if chunk.is_empty() {
2499            return Ok(());
2500        }
2501        self.admit(chunk)?;
2502        if self.pending.last().is_some_and(|last| last.order > order) {
2503            self.flush_pending()?;
2504        }
2505        // Cloned rather than encoded, and a clone of a chunk that owns its buffers is a copy of
2506        // them. Sixty four parts of a hundred and five columns is tens of megabytes held for the
2507        // length of a stripe and a few seconds of memory traffic over a whole ClickBench load,
2508        // against the hundreds of seconds of encode this is what lets off one thread.
2509        self.pending.push(PendingChunk { order, chunk: chunk.clone() });
2510        if self.pending.len() == STRIPE_PARTS {
2511            self.flush_pending()?;
2512        }
2513        Ok(())
2514    }
2515
2516    /// Writes a run of chunks as one stripe of its own.
2517    ///
2518    /// [`Self::append_at`] decides where a stripe ends by watching the orders go past, which works
2519    /// when one caller hands over every chunk in source order and does not when several do. A
2520    /// writer being fed by more than one pipeline instance sees the orders interleave, and a stripe
2521    /// that ends every time two of them cross is a stripe of one or two parts.
2522    ///
2523    /// So the grouping moves to the caller. Whoever is buffering hands over a run it already knows
2524    /// is contiguous and in order, and gets a stripe holding exactly that run. The orders still
2525    /// have to come out in source order once the stripes are sorted, which [`Self::finish`] checks,
2526    /// so the runs from different callers may interleave with each other but may not overlap.
2527    ///
2528    /// # Errors
2529    ///
2530    /// The same as [`Self::append`], and if the run is longer than [`STRIPE_PARTS`].
2531    pub fn append_stripe(&mut self, parts: Vec<((u64, u64), Chunk)>) -> Result<()> {
2532        if parts.len() > STRIPE_PARTS {
2533            return Err(invalid("a stripe was handed more parts than it holds"));
2534        }
2535        // Whatever an earlier caller left behind is its own stripe rather than the front of this
2536        // one, because the two runs are from different places in the source and a stripe is a run.
2537        self.flush_pending()?;
2538        for (order, chunk) in parts {
2539            if chunk.is_empty() {
2540                continue;
2541            }
2542            self.admit(&chunk)?;
2543            self.pending.push(PendingChunk { order, chunk });
2544        }
2545        self.flush_pending()
2546    }
2547
2548    /// Checks a chunk against the declared table and counts its rows in.
2549    fn admit(&mut self, chunk: &Chunk) -> Result<()> {
2550        if chunk.width() != self.table.fields.len() {
2551            return Err(invalid("chunk width differs from table schema"));
2552        }
2553        for (index, field) in self.table.fields.iter().enumerate() {
2554            if chunk.column(index)?.logical_type() != &field.ty {
2555                return Err(invalid("chunk type differs from table schema"));
2556            }
2557        }
2558        self.table.rows = self
2559            .table
2560            .rows
2561            .checked_add(chunk.len())
2562            .ok_or_else(|| invalid("row count overflow"))?;
2563        Ok(())
2564    }
2565
2566    /// One column's parts of a stripe as pages, for a column with no global dictionary.
2567    fn encode_pages(columns: &[&Vector]) -> Result<ColumnStripe> {
2568        let mut stripe = ColumnStripe {
2569            pages: Vec::with_capacity(columns.len()),
2570            codes: Vec::with_capacity(columns.len()),
2571            sieves: Vec::with_capacity(columns.len()),
2572            ranges: Vec::with_capacity(columns.len()),
2573        };
2574        let mut settling = Settling::default();
2575        for &column in columns {
2576            Self::encode_page(&mut stripe, &mut settling, column)?;
2577        }
2578        Ok(stripe)
2579    }
2580
2581    /// One more part of a column with no global dictionary as a page, after the ones already in
2582    /// `stripe`. The parts have to come in order, since `settling` carries from one to the next.
2583    fn encode_page(
2584        stripe: &mut ColumnStripe,
2585        settling: &mut Settling,
2586        column: &Vector,
2587    ) -> Result<()> {
2588        let bytes = encode(column, settling)?;
2589        if bytes.len() > MAX_PAGE {
2590            return Err(invalid("column page exceeds the configured bound"));
2591        }
2592        // The range is built first because the sieve reads it rather than walking the column a
2593        // second time to find out how wide it is.
2594        let range = Range::of(column);
2595        // A sieve at least as large as the part it indexes is not written. A reader reads the
2596        // sieve to decide whether to read the part, so when the sieve is the larger of the two
2597        // it has already spent more than the read it is trying to avoid, and that holds even if
2598        // it rejects every time. It is a necessary condition rather than the whole rule, which
2599        // is that a sieve pays when its bytes are under the rejection rate times the part's,
2600        // but the rejection rate depends on what a query probes for and the writer does not
2601        // know that. The necessary half needs two numbers that are both in hand here.
2602        //
2603        // A column with a global dictionary gets none, because it already has an exact
2604        // membership index per stripe. Those do not come through here. See [`prepare`].
2605        let sieve =
2606            Sieve::of(column, &range, SIEVE_BUDGET).filter(|sieve| sieve.len() < bytes.len());
2607        stripe.pages.push(bytes);
2608        stripe.codes.push(None);
2609        stripe.sieves.push(sieve);
2610        stripe.ranges.push(range);
2611        Ok(())
2612    }
2613
2614    /// Writes every encoded dictionary block that is not in the file yet and forgets its bytes.
2615    ///
2616    /// This is what keeps a load from holding its dictionaries' payload. The blocks land between
2617    /// stripes wherever the writer is, which is fine because the index says where each one is.
2618    fn place_blocks(&mut self) -> Result<()> {
2619        if let Some(lent) = self.lent.clone() {
2620            return self.place_lent_blocks(&lent);
2621        }
2622        let mut dictionaries = std::mem::take(&mut self.dictionaries);
2623        let placed = dictionaries.iter_mut().flatten().try_for_each(|dictionary| {
2624            for block in std::mem::take(&mut dictionary.blocks) {
2625                let start = self.at;
2626                self.put(&block)?;
2627                dictionary.placed.push(Placed {
2628                    start,
2629                    length: block.len() as u64,
2630                    hash: checksum(&block),
2631                });
2632            }
2633            Ok(())
2634        });
2635        self.dictionaries = dictionaries;
2636        placed
2637    }
2638
2639    /// [`Writer::place_blocks`] while a [`Merger`] has the dictionaries.
2640    ///
2641    /// A column whose merge is running is passed over rather than waited for, because the writer's
2642    /// lock is held here and a merge of `URL` can take tens of milliseconds. Its blocks go out with
2643    /// a later stripe, or at the close.
2644    fn place_lent_blocks(&mut self, lent: &Lent) -> Result<()> {
2645        for column in lent.columns() {
2646            let Ok(mut held) = column.try_lock() else { continue };
2647            let Some(dictionary) = held.dictionary.as_mut() else { continue };
2648            for block in std::mem::take(&mut dictionary.blocks) {
2649                let start = self.at;
2650                self.put(&block)?;
2651                dictionary.placed.push(Placed {
2652                    start,
2653                    length: block.len() as u64,
2654                    hash: checksum(&block),
2655                });
2656            }
2657        }
2658        Ok(())
2659    }
2660
2661    /// Takes the dictionaries and the statistics back from the [`Merger`] that has them.
2662    ///
2663    /// A merge that starts after this is refused, since whatever it merged would be lost.
2664    fn reclaim(&mut self) -> Result<()> {
2665        let Some(lent) = self.lent.take() else { return Ok(()) };
2666        let (dictionaries, gathers) = lent.reclaim()?;
2667        self.dictionaries = dictionaries;
2668        self.gathers = gathers;
2669        Ok(())
2670    }
2671
2672    /// Writes the buffered parts as one stripe, each column's parts contiguous on disk.
2673    ///
2674    /// The same four steps a caller holding this writer behind a lock takes, with nobody else
2675    /// waiting between them. See [`prepare`].
2676    fn flush_pending(&mut self) -> Result<()> {
2677        if self.pending.is_empty() {
2678            return Ok(());
2679        }
2680        let held = std::mem::take(&mut self.pending);
2681        let prepared = self.preparer().prepare_held(held)?;
2682        let merged = self.merge_held(prepared)?;
2683        let paged = merged.pages()?;
2684        self.write_paged(paged)
2685    }
2686
2687    /// Writes one stripe whose pages are built, each column's parts contiguous on disk.
2688    fn write_stripe(&mut self, held: &[Part], encoded: Vec<ColumnStripe>) -> Result<()> {
2689        let width = self.table.fields.len();
2690        let parts = held.len();
2691        if encoded.len() != width {
2692            return Err(Error::internal("a stripe came to the writer with the wrong columns"));
2693        }
2694        let profile = self.profile.clone();
2695        if let Some(profile) = &profile {
2696            let rows = held.iter().map(|part| part.rows as u64).sum();
2697            let raw = held.iter().map(|part| part.footprint as u64).sum();
2698            let pages =
2699                encoded.iter().flat_map(|stripe| &stripe.pages).map(|page| page.len() as u64).sum();
2700            profile.moved(Stage::Pages, raw, pages, rows);
2701        }
2702        // Before a byte of the stripe is written, so that the blocks the stripe's pages were built
2703        // with, and any that were waiting on them, are let go of now rather than a stripe later.
2704        let timing = profile.as_deref().map(|profile| profile.span(Stage::Dictionary));
2705        let before = self.at;
2706        self.place_blocks()?;
2707        drop(timing);
2708        if let Some(profile) = &profile {
2709            profile.moved(Stage::Dictionary, 0, self.at - before, 0);
2710        }
2711        let timing = profile.as_deref().map(|profile| profile.span(Stage::Write));
2712        let before = self.at;
2713        let mut pages = Vec::with_capacity(width);
2714        let mut memberships = vec![None; width];
2715        let mut ranges = Vec::with_capacity(width);
2716        let mut index = Vec::with_capacity(width.saturating_mul(index_section(parts)?));
2717        for stripe in &encoded {
2718            let offset = self.at;
2719            let section = index.len();
2720            let mut length = 0_usize;
2721            for bytes in &stripe.pages {
2722                self.file.write_at(self.at + length as u64, bytes)?;
2723                put_u32(
2724                    &mut index,
2725                    u32::try_from(bytes.len()).map_err(|_| invalid("part length overflow"))?,
2726                );
2727                put_u64(&mut index, checksum(bytes));
2728                length = length
2729                    .checked_add(bytes.len())
2730                    .ok_or_else(|| invalid("column page length overflow"))?;
2731            }
2732            let hash = checksum(&index[section..]);
2733            put_u64(&mut index, hash);
2734            if length > MAX_PAGE {
2735                return Err(invalid("column page exceeds the configured bound"));
2736            }
2737            self.at = self
2738                .at
2739                .checked_add(length as u64)
2740                .ok_or_else(|| invalid("native file length overflow"))?;
2741            pages.push(Span {
2742                offset,
2743                length: u32::try_from(length).map_err(|_| invalid("page length overflow"))?,
2744            });
2745            ranges.push(merged_range(stripe.ranges.iter().cloned()));
2746        }
2747        for (membership, stripe) in memberships.iter_mut().zip(&encoded) {
2748            if stripe.codes.iter().all(Option::is_none) {
2749                continue;
2750            }
2751            let lists = stripe
2752                .codes
2753                .iter()
2754                .map(|codes| codes.clone().unwrap_or_default())
2755                .collect::<Vec<_>>();
2756            let bytes = encode_membership(&merged_codes(lists));
2757            let offset = self.at;
2758            self.put(&bytes)?;
2759            *membership = Some(Page {
2760                offset,
2761                length: u32::try_from(bytes.len())
2762                    .map_err(|_| invalid("membership page length overflow"))?,
2763                hash: checksum(&bytes),
2764            });
2765        }
2766        let mut sieves = vec![None; width];
2767        for (page, stripe) in sieves.iter_mut().zip(&encoded) {
2768            if stripe.sieves.iter().all(Option::is_none) {
2769                continue;
2770            }
2771            let bytes = encode_sieves(stripe.sieves.iter())?;
2772            let offset = self.at;
2773            self.put(&bytes)?;
2774            *page = Some(Page {
2775                offset,
2776                length: u32::try_from(bytes.len())
2777                    .map_err(|_| invalid("sieve page length overflow"))?,
2778                hash: checksum(&bytes),
2779            });
2780        }
2781        // A stripe of one part has the same rows in it as that part, so its own bounds are already
2782        // the part's and a page here would say what the directory says. Everywhere else the page is
2783        // written unless it comes to more than the column it indexes, which is the rule the sieves
2784        // go by and for the same reason: a reader reads this to decide whether to read the column,
2785        // so a page larger than the column has spent more than the read it is avoiding.
2786        let mut part_ranges = vec![None; width];
2787        if parts > 1 {
2788            for ((page, stripe), span) in part_ranges.iter_mut().zip(&encoded).zip(&pages) {
2789                let bytes = encode_part_ranges(&stripe.ranges)?;
2790                if bytes.len() >= span.length as usize {
2791                    continue;
2792                }
2793                let offset = self.at;
2794                self.put(&bytes)?;
2795                *page = Some(Page {
2796                    offset,
2797                    length: u32::try_from(bytes.len())
2798                        .map_err(|_| invalid("part range page length overflow"))?,
2799                    hash: checksum(&bytes),
2800                });
2801            }
2802        }
2803        let offset = self.at;
2804        self.put(&index)?;
2805        let index = Span {
2806            offset,
2807            length: u32::try_from(index.len())
2808                .map_err(|_| invalid("index page length overflow"))?,
2809        };
2810        let mut rows = 0_usize;
2811        let mut lengths = Vec::with_capacity(parts);
2812        let mut span = None;
2813        for part in held {
2814            rows = rows.checked_add(part.rows).ok_or_else(|| invalid("row count overflow"))?;
2815            lengths.push(u32::try_from(part.rows).map_err(|_| invalid("part row count overflow"))?);
2816            span = Some(span.map_or((part.order, part.order), |(first, _)| (first, part.order)));
2817        }
2818        self.order.push(span.ok_or_else(|| invalid("a stripe was flushed with no parts"))?);
2819        self.table.stripes.push(Stripe {
2820            rows,
2821            parts: lengths,
2822            index,
2823            pages,
2824            memberships: Pages::from_slots(memberships)?,
2825            sieves: Pages::from_slots(sieves)?,
2826            part_ranges: Pages::from_slots(part_ranges)?,
2827            zone: Zone::from_ranges(ranges),
2828        });
2829        drop(timing);
2830        if let Some(profile) = &profile {
2831            profile.moved(Stage::Write, 0, self.at - before, rows as u64);
2832        }
2833        Ok(())
2834    }
2835
2836    /// Finds exact heavy hitters without keeping a hash table for every numeric column while the
2837    /// load is live. The pages are already in the target file, so one column at a time uses a
2838    /// bounded Misra-Gries candidate table and then recounts only those candidates.
2839    ///
2840    /// The first of those passes also counts the column's distinct values exactly, up to the cap in
2841    /// [`distinct`], which is the number a string column gets from its dictionary. It comes back
2842    /// beside the summary because a column whose heavy hitters cannot be proved can still have been
2843    /// counted.
2844    ///
2845    /// The tables are keyed by a value's sixty four bits rather than by [`FrequencyValue`], and a
2846    /// null is counted beside them. Every integer type the format stores fits in those bits, so
2847    /// within one column two values share bits only if they are the same value, and a sixteen byte
2848    /// entry keeps the whole candidate table in the second level cache where the forty eight byte
2849    /// one did not. The null takes part in the candidate table exactly as a key would: it holds a
2850    /// place while its count is above zero, and it is decremented with the rest.
2851    ///
2852    /// `counted` is false for a column whose sketch says its distinct values are far past what the
2853    /// exact set holds. It still gets its frequencies, and a count only if it turns out to have
2854    /// fewer values than the candidate table, which is the count that costs nothing.
2855    fn numeric_frequency(
2856        &self,
2857        column: usize,
2858        counted: bool,
2859    ) -> Result<(Option<FrequencySummary>, Option<u64>)> {
2860        let signed = match self.table.fields[column].ty {
2861            LogicalType::TinyInt
2862            | LogicalType::SmallInt
2863            | LogicalType::Integer
2864            | LogicalType::BigInt
2865            | LogicalType::Date
2866            | LogicalType::Timestamp => true,
2867            LogicalType::UTinyInt
2868            | LogicalType::USmallInt
2869            | LogicalType::UInteger
2870            | LogicalType::UBigInt => false,
2871            _ => return Ok((None, None)),
2872        };
2873        let value_of = |bits: Option<u64>| match bits {
2874            None => FrequencyValue::Null,
2875            Some(bits) => integer_value(bits, signed),
2876        };
2877        // A column the writer's tally held whole has its exact counts already, gathered as the rows
2878        // went past, so the pages are not read back to count them again. On `hits` that is most of
2879        // the flag and enum columns. The tally only speaks for the whole column when it saw every
2880        // row, which is the same check the statistics make before they are written.
2881        let tallied = self
2882            .gathers
2883            .get(column)
2884            .and_then(Option::as_ref)
2885            .filter(|gather| gather.rows() == self.table.rows as u64)
2886            .and_then(stats::Gather::frequencies)
2887            .and_then(|(values, nulls)| {
2888                let entries = values
2889                    .iter()
2890                    .map(|(value, count)| {
2891                        let value = value_of(Some(frequency_bits(value)?));
2892                        Some(FrequencyEntry { value, count: *count })
2893                    })
2894                    .chain((nulls != 0).then_some(Some(FrequencyEntry {
2895                        value: FrequencyValue::Null,
2896                        count: nulls,
2897                    })))
2898                    .collect::<Option<Vec<_>>>()?;
2899                Some((entries, values.len() as u64))
2900            });
2901        // A column the sketch expects to fit the exact set is counted there, every value with the
2902        // rows holding it, which is its distinct count and its frequencies from one read of its
2903        // pages. Only a column past the set's cap goes through the candidate table.
2904        let exact = match (&tallied, counted) {
2905            (None, true) => self.exact_frequency(column, signed)?,
2906            _ => None,
2907        };
2908        let (mut entries, decrements, distinct_count) = match (tallied, exact) {
2909            (Some((entries, distinct)), _) => (entries, 0, Some(distinct)),
2910            (None, Some((Some(entries), distinct))) => (entries, 0, Some(distinct)),
2911            (None, Some((None, distinct))) => return Ok((None, Some(distinct))),
2912            (None, None) => {
2913                // Rows arrive a run of equal values at a time, because a sorted column is runs and
2914                // a flag column is mostly one value, so a run is counted and inserted once rather
2915                // than per row.
2916                let mut first = Candidates::default();
2917                let mut run = Run::default();
2918                self.visit_numeric(column, signed, |_, bits| {
2919                    if let Some((ended, times)) = run.push(bits) {
2920                        first.add(ended, times);
2921                    }
2922                })?;
2923                if let Some((bits, times)) = run.take() {
2924                    first.add(bits, times);
2925                }
2926                // Until a candidate is turned away the table holds every value the column has, so
2927                // its size is the count.
2928                let (nulls, decrements) = (first.nulls, first.decrements);
2929                let distinct_count = (decrements == 0).then_some(first.held as u64);
2930                let (exact, null_count) = if decrements == 0 {
2931                    let exact = first
2932                        .pairs()
2933                        .map(|(bits, count)| (bits, u64::from(count)))
2934                        .collect::<FrequencyMap<_>>();
2935                    (exact, (nulls != 0).then_some(u64::from(nulls)))
2936                } else {
2937                    let mut lower = first.pairs().map(|(_, count)| count).collect::<Vec<_>>();
2938                    if nulls != 0 {
2939                        lower.push(nulls);
2940                    }
2941                    lower.sort_unstable_by(|left, right| right.cmp(left));
2942                    if lower.len() < FREQUENCY_BUILD_RANK
2943                        || u64::from(lower[FREQUENCY_BUILD_RANK - 1]) <= decrements
2944                    {
2945                        return Ok((None, distinct_count));
2946                    }
2947                    // Counted beside the slot each candidate sits in, since the table is not
2948                    // changed again and a lookup in it is the one probe the first pass made.
2949                    let mut recounts = vec![0_u64; first.slots.len()];
2950                    let mut null_count = (nulls != 0).then_some(0_u64);
2951                    let mut recount = |bits: Option<u64>, times: u32| {
2952                        let held = match bits {
2953                            Some(bits) => first.position(bits).map(|at| &mut recounts[at]),
2954                            None => null_count.as_mut(),
2955                        };
2956                        if let Some(count) = held {
2957                            *count = count.saturating_add(u64::from(times));
2958                        }
2959                    };
2960                    let mut run = Run::default();
2961                    self.visit_numeric(column, signed, |_, bits| {
2962                        if let Some((bits, times)) = run.push(bits) {
2963                            recount(bits, times);
2964                        }
2965                    })?;
2966                    if let Some((bits, times)) = run.take() {
2967                        recount(bits, times);
2968                    }
2969                    let exact = first
2970                        .slots
2971                        .iter()
2972                        .zip(&recounts)
2973                        .filter(|(slot, _)| slot.count != 0)
2974                        .map(|(slot, &count)| (slot.bits, count))
2975                        .collect::<FrequencyMap<_>>();
2976                    (exact, null_count)
2977                };
2978                let entries = exact
2979                    .into_iter()
2980                    .map(|(bits, count)| FrequencyEntry { value: value_of(Some(bits)), count })
2981                    .chain(
2982                        null_count
2983                            .map(|count| FrequencyEntry { value: FrequencyValue::Null, count }),
2984                    )
2985                    .collect::<Vec<_>>();
2986                (entries, decrements, distinct_count)
2987            }
2988        };
2989        let mut omitted_max = keep_most_frequent(&mut entries).max(decrements);
2990        // A complete value-to-count table is also the result of grouping this column.
2991        // Keep up to two leading frequencies for selectivity and equality predicates,
2992        // but leave multi-value grouped counts to the encoded rows at query time.
2993        if omitted_max == 0 && entries.len() > 1 {
2994            let retained = entries.len().saturating_sub(1).min(2);
2995            omitted_max = entries[retained].count;
2996            entries.truncate(retained);
2997        }
2998        let kept_rows = entries.iter().try_fold(0_u64, |total, entry| {
2999            total.checked_add(entry.count).filter(|&total| total <= FREQUENCY_ORDINALS as u64)
3000        });
3001        let mut ordinals = Vec::new();
3002        let mut ordinal_entries = Vec::new();
3003        if let Some(kept_rows) = kept_rows {
3004            let mut kept = FrequencyMap::default();
3005            let mut null_kept = None;
3006            for (at, entry) in entries.iter().enumerate() {
3007                let at = u16::try_from(at)
3008                    .map_err(|_| invalid("too many retained frequency entries"))?;
3009                match entry.value {
3010                    FrequencyValue::Integer(value) => {
3011                        kept.insert(value as u64, at);
3012                    }
3013                    FrequencyValue::Null => null_kept = Some(at),
3014                    FrequencyValue::Code(_) => {}
3015                }
3016            }
3017            ordinals.reserve(usize::try_from(kept_rows).unwrap_or(FREQUENCY_ORDINALS));
3018            ordinal_entries.reserve(usize::try_from(kept_rows).unwrap_or(FREQUENCY_ORDINALS));
3019            self.visit_numeric(column, signed, |ordinal, bits| {
3020                let held = match bits {
3021                    Some(bits) => kept.get(&bits).copied(),
3022                    None => null_kept,
3023                };
3024                if let Some(entry) = held {
3025                    ordinals.push(ordinal);
3026                    ordinal_entries.push(entry);
3027                }
3028            })?;
3029        }
3030        Ok((
3031            Some(FrequencySummary { entries, omitted_max, ordinals, ordinal_entries }),
3032            distinct_count,
3033        ))
3034    }
3035
3036    /// Counts every value of an integer column and the rows holding it, and hands back the
3037    /// frequency entries worth keeping beside the distinct count, or nothing for a column with more
3038    /// values than [`distinct::ExactCounts`] keeps.
3039    ///
3040    /// The entries are `None` for a column with no value common enough to be worth a synopsis. The
3041    /// rule is the one the candidate table applied. A column with more values than that table holds
3042    /// keeps its frequencies only if its tenth commonest value is held by more rows than a
3043    /// Misra-Gries table of [`FREQUENCY_CANDIDATES`] could have decremented it by, which is its rows
3044    /// over one more than the candidates. The counts kept are exact either way, so the largest one
3045    /// left out is exact too and not the table's bound on it.
3046    ///
3047    /// Only the commonest entries and the ones tied with the first left out are built, since on a
3048    /// column of a million values the rest are thrown away the moment they are ranked.
3049    fn exact_frequency(
3050        &self,
3051        column: usize,
3052        signed: bool,
3053    ) -> Result<Option<(Option<Vec<FrequencyEntry>>, u64)>> {
3054        let mut set = distinct::ExactCounts::new();
3055        let mut nulls = 0_u64;
3056        let mut run = Run::default();
3057        let mut add = |bits: Option<u64>, times: u32| match bits {
3058            Some(bits) => set.insert(bits, times),
3059            None => nulls += u64::from(times),
3060        };
3061        self.visit_numeric(column, signed, |_, bits| {
3062            if let Some((bits, times)) = run.push(bits) {
3063                add(bits, times);
3064            }
3065        })?;
3066        if let Some((bits, times)) = run.take() {
3067            add(bits, times);
3068        }
3069        let Some(distinct) = set.count() else {
3070            return Ok(None);
3071        };
3072        // The commonest counts, one more than the entries kept so that the first left out is here.
3073        let mut top = std::collections::BinaryHeap::with_capacity(FREQUENCY_ENTRIES + 2);
3074        let mut rank = |count: u64| {
3075            if top.len() <= FREQUENCY_ENTRIES {
3076                top.push(Reverse(count));
3077            } else if top.peek().is_some_and(|&Reverse(least)| count > least) {
3078                top.pop();
3079                top.push(Reverse(count));
3080            }
3081        };
3082        set.visit(|_, count| rank(count));
3083        if nulls != 0 {
3084            rank(nulls);
3085        }
3086        let top = top.into_sorted_vec();
3087        let values = distinct + u64::from(nulls != 0);
3088        if values > FREQUENCY_CANDIDATES as u64 {
3089            let bound = self.table.rows as u64 / (FREQUENCY_CANDIDATES as u64 + 1);
3090            if top.get(FREQUENCY_BUILD_RANK - 1).is_none_or(|&Reverse(count)| count <= bound) {
3091                return Ok(Some((None, distinct)));
3092            }
3093        }
3094        let least = top.get(FREQUENCY_ENTRIES).map_or(0, |&Reverse(count)| count);
3095        let mut entries = Vec::with_capacity(FREQUENCY_ENTRIES + 1);
3096        set.visit(|bits, count| {
3097            if count >= least {
3098                entries.push(FrequencyEntry { value: integer_value(bits, signed), count });
3099            }
3100        });
3101        if nulls != 0 && nulls >= least {
3102            entries.push(FrequencyEntry { value: FrequencyValue::Null, count: nulls });
3103        }
3104        Ok(Some((Some(entries), distinct)))
3105    }
3106
3107    /// Hands every row of an integer column to `visit` as its ordinal and its sixty four bits, or
3108    /// `None` for a null.
3109    ///
3110    /// `signed` says which of the two readings the column has. A packed unsigned column would come
3111    /// back from `signed_block` as a base plus a code in `i64`, which wraps for a value past the top
3112    /// of `BIGINT`, so only a signed column takes the block path.
3113    fn visit_numeric(
3114        &self,
3115        column: usize,
3116        signed: bool,
3117        mut visit: impl FnMut(u64, Option<u64>),
3118    ) -> Result<()> {
3119        let ty = &self.table.fields[column].ty;
3120        let mut start = 0_u64;
3121        let mut block = Vec::new();
3122        for stripe in &self.table.stripes {
3123            let spans = read_index(&self.file, stripe, column)?;
3124            let page = stripe.pages[column];
3125            let mut bytes = vec![0; page.length as usize];
3126            read_at(&self.file, page.offset, &mut bytes)?;
3127            for (span, &rows) in spans.iter().zip(&stripe.parts) {
3128                let part = part_bytes(&bytes, *span)?;
3129                if checksum(part) != span.hash {
3130                    return Err(invalid("column page checksum differs while building frequencies"));
3131                }
3132                let rows = rows as usize;
3133                let vector = decode(ty, rows, part, None)?;
3134                // Every signed layout a numeric column decodes to, which is every column of `hits`,
3135                // comes out as one run of `i64` and is walked as a slice. The row path below is for
3136                // the unsigned types and anything else that cannot be handed over that way.
3137                if signed && vector.signed_block(&mut block) && block.len() == rows {
3138                    if vector.none_null() {
3139                        for (row, &value) in block.iter().enumerate() {
3140                            visit(start.saturating_add(row as u64), Some(value as u64));
3141                        }
3142                    } else {
3143                        for (row, &value) in block.iter().enumerate() {
3144                            let bits = (!vector.is_null_at(row)).then_some(value as u64);
3145                            visit(start.saturating_add(row as u64), bits);
3146                        }
3147                    }
3148                    start = start.saturating_add(rows as u64);
3149                    continue;
3150                }
3151                // row at a time: frequency construction visits decoded values to update bounded candidates.
3152                for row in 0..rows {
3153                    let bits = if vector.is_null_at(row) {
3154                        None
3155                    } else {
3156                        // An unsigned column has no signed reading, and the documented fallback is
3157                        // the value itself. Every width the format stores fits in sixty four bits,
3158                        // so nothing is lost on the way through.
3159                        let widened = match vector.signed_at(row) {
3160                            Some(value) => Some(value as u64),
3161                            None => match vector.value_at(row) {
3162                                Value::UTinyInt(value) => Some(u64::from(value)),
3163                                Value::USmallInt(value) => Some(u64::from(value)),
3164                                Value::UInteger(value) => Some(u64::from(value)),
3165                                Value::UBigInt(value) => Some(value),
3166                                _ => None,
3167                            },
3168                        };
3169                        Some(widened.ok_or_else(|| {
3170                            invalid("numeric frequency page did not contain an integer value")
3171                        })?)
3172                    };
3173                    visit(start.saturating_add(row as u64), bits);
3174                }
3175                start = start.saturating_add(rows as u64);
3176            }
3177        }
3178        Ok(())
3179    }
3180
3181    /// The columns that get numeric frequencies, which are the integer, date and timestamp ones.
3182    fn numeric_columns(&self) -> Vec<usize> {
3183        self.table
3184            .fields
3185            .iter()
3186            .enumerate()
3187            .filter_map(|(column, field)| {
3188                matches!(
3189                    field.ty,
3190                    LogicalType::TinyInt
3191                        | LogicalType::SmallInt
3192                        | LogicalType::Integer
3193                        | LogicalType::BigInt
3194                        | LogicalType::UTinyInt
3195                        | LogicalType::USmallInt
3196                        | LogicalType::UInteger
3197                        | LogicalType::UBigInt
3198                        | LogicalType::Date
3199                        | LogicalType::Timestamp
3200                )
3201                .then_some(column)
3202            })
3203            .collect()
3204    }
3205
3206    /// Reads one stable dictionary code column only at sorted table-wide row ordinals.
3207    #[allow(dead_code)]
3208    fn stable_codes_at(&self, column: usize, ordinals: &[u64]) -> Result<Option<Vec<Option<u32>>>> {
3209        if self.dictionaries.get(column).and_then(Option::as_ref).is_none() {
3210            return Ok(None);
3211        }
3212        if ordinals.windows(2).any(|pair| pair[0] >= pair[1]) {
3213            return Err(invalid("frequency ordinals are not sorted and unique"));
3214        }
3215        let mut out = Vec::with_capacity(ordinals.len());
3216        let mut wanted = 0;
3217        let mut stripe_start = 0_u64;
3218        for stripe in &self.table.stripes {
3219            let stripe_end = stripe_start.saturating_add(stripe.rows as u64);
3220            if wanted == ordinals.len() || ordinals[wanted] >= stripe_end {
3221                stripe_start = stripe_end;
3222                continue;
3223            }
3224            let spans = read_index(&self.file, stripe, column)?;
3225            let page = stripe.pages[column];
3226            let mut bytes = vec![0; page.length as usize];
3227            read_at(&self.file, page.offset, &mut bytes)?;
3228            let mut part_start = stripe_start;
3229            for (span, &rows) in spans.iter().zip(&stripe.parts) {
3230                let part_end = part_start.saturating_add(u64::from(rows));
3231                if wanted < ordinals.len() && ordinals[wanted] < part_end {
3232                    let part = part_bytes(&bytes, *span)?;
3233                    if checksum(part) != span.hash {
3234                        return Err(invalid(
3235                            "column page checksum differs while building pair frequencies",
3236                        ));
3237                    }
3238                    let upto = ordinals.partition_point(|&ordinal| ordinal < part_end);
3239                    let positions = ordinals[wanted..upto]
3240                        .iter()
3241                        .map(|&ordinal| {
3242                            usize::try_from(ordinal.saturating_sub(part_start))
3243                                .map_err(|_| invalid("frequency row offset does not fit in memory"))
3244                        })
3245                        .collect::<Result<Vec<_>>>()?;
3246                    if !decode_selected_stable_codes(rows as usize, part, &positions, &mut out)? {
3247                        return Ok(None);
3248                    }
3249                    wanted = upto;
3250                }
3251                part_start = part_end;
3252            }
3253            stripe_start = stripe_end;
3254        }
3255        if wanted != ordinals.len() {
3256            return Err(invalid("frequency ordinal is outside the table"));
3257        }
3258        Ok(Some(out))
3259    }
3260
3261    /// Derives bounded two-key leaders from numeric anchor ordinals and stable string codes.
3262    #[allow(dead_code)]
3263    fn pair_frequencies(
3264        &self,
3265        frequencies: &[Option<Frequencies>],
3266    ) -> Result<Vec<PairFrequencySummary>> {
3267        let anchors = frequencies
3268            .iter()
3269            .enumerate()
3270            .filter_map(|(column, summary)| {
3271                // A writer holds every synopsis it counted, so there is nothing stored to skip.
3272                match summary {
3273                    Some(Frequencies::Held(summary)) => Some(summary),
3274                    _ => None,
3275                }
3276                .filter(|summary| {
3277                    !summary.ordinals.is_empty()
3278                        && summary.ordinal_entries.len() == summary.ordinals.len()
3279                })
3280                .cloned()
3281                .map(|summary| (column, summary))
3282            })
3283            .collect::<Vec<_>>();
3284        let strings = self
3285            .dictionaries
3286            .iter()
3287            .enumerate()
3288            .filter_map(|(column, dictionary)| dictionary.as_ref().map(|_| column))
3289            .collect::<Vec<_>>();
3290        let mut summaries = Vec::new();
3291        for (first, anchors) in anchors {
3292            for &second in &strings {
3293                if summaries.len() == MAX_PAIR_FREQUENCIES {
3294                    return Ok(summaries);
3295                }
3296                let Some(codes) = self.stable_codes_at(second, &anchors.ordinals)? else {
3297                    continue;
3298                };
3299                if codes.len() != anchors.ordinal_entries.len() {
3300                    return Err(invalid("pair frequency columns have different lengths"));
3301                }
3302                let mut counts = HashMap::<(u16, Option<u32>), u64>::new();
3303                for (&anchor, code) in anchors.ordinal_entries.iter().zip(codes) {
3304                    *counts.entry((anchor, code)).or_default() += 1;
3305                }
3306                let mut entries = counts
3307                    .into_iter()
3308                    .map(|((first_entry, second), count)| PairFrequencyEntry {
3309                        first_entry,
3310                        second,
3311                        count,
3312                    })
3313                    .collect::<Vec<_>>();
3314                entries.sort_unstable_by(|left, right| {
3315                    right
3316                        .count
3317                        .cmp(&left.count)
3318                        .then_with(|| left.first_entry.cmp(&right.first_entry))
3319                        .then_with(|| left.second.cmp(&right.second))
3320                });
3321                let pair_omitted = entries.get(FREQUENCY_ENTRIES).map_or(0, |entry| entry.count);
3322                entries.truncate(FREQUENCY_ENTRIES);
3323                summaries.push(PairFrequencySummary {
3324                    first: u16::try_from(first)
3325                        .map_err(|_| invalid("pair frequency column index overflows"))?,
3326                    second: u16::try_from(second)
3327                        .map_err(|_| invalid("pair frequency column index overflows"))?,
3328                    entries,
3329                    omitted_max: anchors.omitted_max.max(pair_omitted),
3330                });
3331            }
3332        }
3333        Ok(summaries)
3334    }
3335
3336    /// Writes the directory of the table this writer is on and says where it went.
3337    ///
3338    /// Everything [`Writer::finish`] used to do except the two writes that publish. Pulling it out
3339    /// is what lets a second table follow a first: the bytes of a closed table are complete and
3340    /// addressable while nothing yet points at them, and the pointer is the last write of the
3341    /// commit.
3342    ///
3343    /// # Errors
3344    ///
3345    /// If directory encoding or writing fails.
3346    fn close(&mut self) -> Result<Entry> {
3347        self.reclaim()?;
3348        self.flush_pending()?;
3349        // The rest of a table is its statistics, its dictionaries and its directory. The dictionary
3350        // work is charged as its own stage, because ranking a global dictionary can be most of what
3351        // this costs, and the rest as publish.
3352        let profile = self.profile.clone();
3353        let timing = profile.as_deref().map(|profile| profile.span(Stage::Publish));
3354        let before = self.at;
3355        let mut stripes = std::mem::take(&mut self.order)
3356            .into_iter()
3357            .zip(std::mem::take(&mut self.table.stripes))
3358            .collect::<Vec<_>>();
3359        stripes.sort_by_key(|(order, _)| order.0);
3360        let mut previous: Option<(u64, u64)> = None;
3361        for ((first, last), _) in &stripes {
3362            if previous.is_some_and(|previous| previous >= *first) {
3363                return Err(invalid("chunks did not arrive in source order"));
3364            }
3365            previous = Some(*last);
3366        }
3367        self.table.stripes = stripes.into_iter().map(|(_, stripe)| stripe).collect();
3368        drop(timing);
3369        let timing = profile.as_deref().map(|profile| profile.span(Stage::Dictionary));
3370        let placing = self.at;
3371        finish_dictionaries(&mut self.dictionaries)?;
3372        self.place_blocks()?;
3373        for dictionary in self.dictionaries.iter_mut().flatten() {
3374            dictionary.release_lookup();
3375            dictionary.recharge(profile.as_deref());
3376        }
3377        let (numeric, closed) = self.close_columns()?;
3378        let (frequencies, distincts): (Vec<Option<FrequencySummary>>, Vec<_>) =
3379            numeric.into_iter().unzip();
3380        let frequencies =
3381            frequencies.into_iter().map(|held| held.map(Frequencies::Held)).collect::<Vec<_>>();
3382        // Pair leaders are query results, not reusable column statistics.
3383        let pairs = Vec::new();
3384        self.table.frequencies = frequencies;
3385        self.table.distincts = distincts;
3386        self.table.pair_frequencies = pairs;
3387        if let Some(profile) = &profile {
3388            profile.release(self.dictionaries.iter().flatten().map(|held| held.charged).sum());
3389        }
3390        self.table.demoted = self
3391            .dictionaries
3392            .iter()
3393            .map(|dictionary| dictionary.as_ref().is_some_and(|held| held.demoted))
3394            .collect();
3395        if !self.table.demoted.contains(&true) {
3396            self.table.demoted = Vec::new();
3397        }
3398        self.dictionaries = Vec::new();
3399        self.table.dictionary_payloads = vec![0; self.table.fields.len()];
3400        self.table.frequency_texts = vec![Vec::new(); self.table.fields.len()];
3401        self.table.host_groups = None;
3402        for (index, closed) in closed.into_iter().enumerate() {
3403            let Some(closed) = closed else { continue };
3404            let ClosedDictionary { distinct, frequencies, texts, hosts, encoded, payload } = closed;
3405            self.table.distincts[index] = distinct;
3406            self.table.frequencies[index] = frequencies.map(Frequencies::Held);
3407            self.table.frequency_texts[index] = texts;
3408            if hosts.is_some() {
3409                self.table.host_groups = hosts;
3410            }
3411            let offset = self.at;
3412            self.put(&encoded.index)?;
3413            self.put(&encoded.ranks)?;
3414            self.put(&encoded.grams)?;
3415            self.table.dictionary_payloads[index] = payload;
3416            let length = encoded
3417                .index
3418                .len()
3419                .checked_add(encoded.ranks.len())
3420                .and_then(|len| len.checked_add(encoded.grams.len()))
3421                .ok_or_else(|| invalid("dictionary page length overflow"))?;
3422            self.table.dictionaries[index] = Some(Page {
3423                offset,
3424                length: u32::try_from(length)
3425                    .map_err(|_| invalid("dictionary page length overflow"))?,
3426                hash: checksum(&encoded.index),
3427            });
3428        }
3429        drop(timing);
3430        let timing = profile.as_deref().map(|profile| profile.span(Stage::Publish));
3431        let placed = self.at - placing;
3432        self.write_stats()?;
3433        let directory = encode_directory(&self.table)?;
3434        if directory.len() > MAX_DIRECTORY {
3435            return Err(invalid("directory exceeds the configured bound"));
3436        }
3437        let offset = self.at;
3438        self.put(&directory)?;
3439        drop(timing);
3440        if let Some(profile) = &profile {
3441            profile.moved(Stage::Dictionary, 0, placed, 0);
3442            profile.moved(Stage::Publish, 0, self.at - before - placed, 0);
3443        }
3444        Ok(Entry {
3445            name: self.table.name.clone(),
3446            fields: self.table.fields.clone(),
3447            rows: self.table.rows,
3448            nonzero: vec![None; self.table.fields.len()],
3449            aggregates: table_aggregate_sums(&self.table),
3450            distincts: self.table.distincts.clone(),
3451            extremes: table_integer_extremes(&self.table),
3452            frequencies: table_complete_numeric_frequencies(&self.table),
3453            directory: Page {
3454                offset,
3455                length: u32::try_from(directory.len())
3456                    .map_err(|_| invalid("directory length overflow"))?,
3457                hash: checksum(&directory),
3458            },
3459        })
3460    }
3461
3462    /// Every numeric column's frequencies and every global dictionary's page and statistics, by
3463    /// column, as many columns at a time as [`CLOSE_BYTES`] allows.
3464    ///
3465    /// The two kinds read what is already written and write nothing, so they share one set of
3466    /// threads. Each was most of a second on `hits` with the other waiting for it, and neither keeps
3467    /// every core busy on its own. The most expensive column that fits is the one taken next, so
3468    /// the long ones start first and the short ones fill in behind them. A column that does not fit
3469    /// waits for one that is closing to finish, unless nothing is closing, in which case it goes
3470    /// alone.
3471    ///
3472    /// A numeric column is charged the exact distinct set its sketch says it will need, and one the
3473    /// sketch puts far past what that set can hold does not build it, because the set would fill,
3474    /// give up and have held 512 MiB for nothing. A column with no sketch is charged the whole set.
3475    /// Each job charges itself as its own span, publish for the numeric ones and dictionary for the
3476    /// rest, because it runs on a thread of its own and a span on this one would see the wall time
3477    /// and none of the CPU.
3478    #[allow(clippy::type_complexity)]
3479    fn close_columns(
3480        &self,
3481    ) -> Result<(Vec<(Option<FrequencySummary>, Option<u64>)>, Vec<Option<ClosedDictionary>>)> {
3482        let numeric = self.numeric_columns().into_iter().map(|column| {
3483            let estimate =
3484                self.gathers.get(column).and_then(Option::as_ref).and_then(stats::Gather::distinct);
3485            let counted = !estimate.is_some_and(distinct::beyond);
3486            let set =
3487                if counted { distinct::bytes_for(estimate.unwrap_or(f64::INFINITY)) } else { 0 };
3488            let cost = self.table.rows.saturating_mul(weight(&self.table.fields[column].ty));
3489            (Closing::Numeric { column, counted }, NUMERIC_CLOSE_BYTES + set, cost)
3490        });
3491        let dictionaries =
3492            self.dictionaries.iter().enumerate().filter_map(|(index, dictionary)| {
3493                let dictionary = dictionary.as_ref()?;
3494                let bytes = dictionary.closing_bytes();
3495                Some((Closing::Dictionary { index, dictionary }, bytes, bytes))
3496            });
3497        let mut jobs = numeric.chain(dictionaries).collect::<Vec<_>>();
3498        jobs.sort_by_key(|&(_, _, cost)| cost);
3499        let columns = self.table.fields.len();
3500        let mut frequencies = vec![(None, None); columns];
3501        let mut closed = (0..columns).map(|_| None).collect::<Vec<_>>();
3502        let profile = self.profile.as_deref();
3503        let run = |job: Closing<'_>, bytes: usize| -> Result<Closed> {
3504            let _holding = profile.map(|profile| profile.holding(bytes as u64));
3505            match job {
3506                Closing::Numeric { column, counted } => {
3507                    let _timing = profile.map(|profile| profile.span(Stage::Publish));
3508                    Ok(Closed::Numeric(column, self.numeric_frequency(column, counted)?))
3509                }
3510                Closing::Dictionary { index, dictionary } => {
3511                    let _timing = profile.map(|profile| profile.span(Stage::Dictionary));
3512                    Ok(Closed::Dictionary(index, self.close_dictionary(index, dictionary)?))
3513                }
3514            }
3515        };
3516        let workers = close_workers().min(jobs.len());
3517        let pieces = if workers <= 1 {
3518            jobs.into_iter().map(|(job, bytes, _)| run(job, bytes)).collect::<Result<Vec<_>>>()?
3519        } else {
3520            // The columns not taken yet, cheapest first, and the bytes the ones closing now hold.
3521            let state = Mutex::new((jobs, 0_usize));
3522            let finished = Condvar::new();
3523            std::thread::scope(|scope| {
3524                (0..workers)
3525                    .map(|_| {
3526                        scope.spawn(|| {
3527                            let mut mine = Vec::new();
3528                            loop {
3529                                let mut held = state.lock().map_err(|_| {
3530                                    Error::internal("a native close worker panicked")
3531                                })?;
3532                                let (job, bytes) = loop {
3533                                    let (jobs, busy) = &mut *held;
3534                                    if jobs.is_empty() {
3535                                        return Ok(mine);
3536                                    }
3537                                    let fits = jobs.iter().rposition(|&(_, bytes, _)| {
3538                                        *busy == 0 || busy.saturating_add(bytes) <= CLOSE_BYTES
3539                                    });
3540                                    if let Some(at) = fits {
3541                                        let (job, bytes, _) = jobs.remove(at);
3542                                        *busy += bytes;
3543                                        break (job, bytes);
3544                                    }
3545                                    held = finished.wait(held).map_err(|_| {
3546                                        Error::internal("a native close worker panicked")
3547                                    })?;
3548                                };
3549                                drop(held);
3550                                // Given back on the way out whether the close worked, failed or
3551                                // panicked, so that a worker waiting for room is never left waiting.
3552                                let _room = Room { state: &state, finished: &finished, bytes };
3553                                mine.push(run(job, bytes)?);
3554                            }
3555                        })
3556                    })
3557                    .collect::<Vec<_>>()
3558                    .into_iter()
3559                    .map(|handle| {
3560                        handle
3561                            .join()
3562                            .map_err(|_| Error::internal("a native close worker panicked"))?
3563                    })
3564                    .collect::<Result<Vec<_>>>()
3565            })?
3566            .into_iter()
3567            .flatten()
3568            .collect()
3569        };
3570        for piece in pieces {
3571            match piece {
3572                Closed::Numeric(column, summary) => frequencies[column] = summary,
3573                Closed::Dictionary(index, one) => closed[index] = Some(one),
3574            }
3575        }
3576        Ok((frequencies, closed))
3577    }
3578
3579    /// One global dictionary's page and statistics, built from what is already in the file.
3580    ///
3581    /// Nothing is written here, so that [`Self::close`] can run this beside the numeric frequencies
3582    /// and put the pages down afterwards in column order, which is where they always went. The
3583    /// column's values are decoded in here and dropped before it returns, and
3584    /// [`Self::close_columns`] decides how many columns are in here at once.
3585    fn close_dictionary(
3586        &self,
3587        _index: usize,
3588        dictionary: &GlobalDictionary,
3589    ) -> Result<ClosedDictionary> {
3590        let (order, flat, bases) = dictionary.ranked_with_values(Some(&*self.file))?;
3591        // A code nothing counted is a code no non-null row of this column holds, which is the
3592        // empty string a null was written as and nothing else, because a code is only ever made by
3593        // a row asking for one. A demoted dictionary counted the stripes before its demotion and
3594        // none after, so it has no count or frequency of the column to give.
3595        let (distinct, frequencies, texts) = if dictionary.demoted {
3596            (None, None, Vec::new())
3597        } else {
3598            let distinct = dictionary.counts.iter().filter(|count| **count != 0).count() as u64;
3599            let (frequencies, texts) = code_frequency(dictionary, &flat, &bases)?;
3600            (Some(distinct), Some(frequencies), texts)
3601        };
3602        // Deriving a fixed SQL host expression at load time materializes its answer.
3603        let hosts = None;
3604        drop(flat);
3605        drop(bases);
3606        let encoded = encode_global_dictionary(dictionary, &order, &dictionary.placed, true)?;
3607        let payload = dictionary
3608            .placed
3609            .iter()
3610            .try_fold(0_u64, |sum, place| sum.checked_add(place.length))
3611            .ok_or_else(|| invalid("global dictionary payload overflow"))?;
3612        Ok(ClosedDictionary { distinct, frequencies, texts, hosts, encoded, payload })
3613    }
3614
3615    /// Writes the statistics sections for the table being closed, as far as the budget reaches.
3616    ///
3617    /// Called from [`Self::close`] after the last stripe and after the dictionaries, which is the
3618    /// first moment the table's column bytes are final and the last moment before the directory is
3619    /// encoded. Both halves matter: the budget is a share of the column bytes, and a section that
3620    /// went in after the directory would be a section the directory does not name.
3621    ///
3622    /// Nothing here can fail the write. A column whose gather came back blind gets no sections, a
3623    /// column the budget could not reach gets none, and section 3.1 says both of those plan the way
3624    /// they planned before statistics existed. The two errors that are returned are an encode
3625    /// failure and a section count past the bound, and neither is a thing a column can cause.
3626    fn write_stats(&mut self) -> Result<()> {
3627        let gathers = std::mem::take(&mut self.gathers);
3628        let rows = self.table.rows as u64;
3629        let mut payloads = Vec::new();
3630        for (column, gather) in gathers.into_iter().enumerate() {
3631            let Some(gather) = gather else { continue };
3632            // A gather that saw a different number of rows than the table committed is a gather
3633            // that missed some, and a distinct count over some of a column is the one error an
3634            // estimator cannot see coming. This has no way of happening today, since a table is
3635            // written once and every chunk goes through `flush_pending`, and that is exactly why it
3636            // is worth a line: it stays true only while that stays true.
3637            if gather.rows() != rows {
3638                continue;
3639            }
3640            let Some(stats) = gather.finish() else { continue };
3641            let mut summary = Vec::new();
3642            stats.summary.encode(&mut summary)?;
3643            let mut sketches = Vec::new();
3644            stats.sketches.encode(&mut sketches)?;
3645            payloads.push((column, summary, sketches));
3646        }
3647        if payloads.is_empty() {
3648            return Ok(());
3649        }
3650        let costs = payloads
3651            .iter()
3652            .map(|(_, summary, sketches)| summary.len() + sketches.len())
3653            .collect::<Vec<_>>();
3654        let allowance = stats::allowance(stats::column_bytes(&self.table), stats::BUDGET_SHARE);
3655        // Nothing is spent yet. A table this writer is closing is one it wrote from nothing, so the
3656        // only statistics sections it can have are the ones about to go in.
3657        let keep = stats::within(&costs, allowance, 0);
3658        for ((column, summary, sketches), _) in
3659            payloads.iter().zip(&keep).filter(|&(_, &keep)| keep)
3660        {
3661            let id = u64::try_from(*column).map_err(|_| invalid("column index overflow"))?;
3662            for (kind, bytes, header_bytes) in [
3663                // A summary is a header the whole way down: there is nothing behind it a reader
3664                // could decide not to read.
3665                (*section::SUMMARY, summary, summary.len() as u32),
3666                (*section::SKETCHES, sketches, rudb_stats::sketches::HEADER_BYTES),
3667            ] {
3668                let written = write_section(
3669                    &*self.file,
3670                    &mut self.at,
3671                    &section::Attachment { kind, id, flags: 0, header_bytes, bytes },
3672                    self.generation,
3673                )?;
3674                self.table.sections.push(written);
3675            }
3676        }
3677        if self.table.sections.len() > MAX_SECTIONS {
3678            return Err(invalid("the table would name more sections than the bound allows"));
3679        }
3680        Ok(())
3681    }
3682
3683    /// Commits every table this writer has written and syncs the file before publishing its header
3684    /// slot.
3685    ///
3686    /// The table handed back is the one the writer was on, which is the last of them. Callers that
3687    /// wrote several already know the others, since they named them.
3688    ///
3689    /// # Errors
3690    ///
3691    /// If directory encoding, writing, or syncing fails.
3692    pub fn finish(mut self) -> Result<Table> {
3693        let entry = self.close()?;
3694        let profile = self.profile.take();
3695        let _timing = profile.as_deref().map(|profile| profile.span(Stage::Publish));
3696        let mut tables = std::mem::take(&mut self.closed);
3697        tables.push(entry);
3698        let catalog = encode_catalog(&tables, &self.views)?;
3699        if catalog.len() > MAX_DIRECTORY {
3700            return Err(invalid("catalog exceeds the configured bound"));
3701        }
3702        let offset = self.at;
3703        self.put(&catalog)?;
3704        if let Some(profile) = &profile {
3705            profile.moved(Stage::Publish, 0, catalog.len() as u64, 0);
3706        }
3707        // Every page and every table directory is on the disk before anything points at them. The
3708        // slot write below is what makes this generation the one a reader picks, so the order of
3709        // these two syncs is the whole of the commit.
3710        synced(&*self.file, profile.as_deref())?;
3711        let slot = Slot {
3712            offset,
3713            length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
3714            generation: self.generation,
3715            hash: checksum(&catalog),
3716        };
3717        // The one write that is not an append, and the last one. It goes back over the slot in the
3718        // header, so it names its offset rather than going through `put`, and `at` does not move.
3719        // Which of the two slots it is alternates with the generation, so the one naming the
3720        // generation before this is still intact and still valid until this write lands.
3721        self.file.write_at(slot_offset(self.generation), &slot.bytes())?;
3722        synced(&*self.file, profile.as_deref())?;
3723        Ok(self.table)
3724    }
3725
3726    /// Commits a generation that changes the views and leaves every table exactly where it is.
3727    ///
3728    /// There was no way to do this before views existed, because everything that could change the
3729    /// catalog also wrote a table, so the only way to say something new about a file was to go
3730    /// through a table. A view is the first thing that can change on its own. Without this, adding
3731    /// a view to a database with eight tables in it would rewrite all eight, since the append path
3732    /// needs a table to append and the fallback is the whole file.
3733    ///
3734    /// It is the same commit as [`Writer::finish`] with nothing appended before it. The table
3735    /// entries are carried forward by directory pointer the way an append carries them, the new
3736    /// catalog goes on the end, and the slot write at the end is what publishes it.
3737    ///
3738    /// # Errors
3739    ///
3740    /// If the file has no valid committed directory, is not this build's format, or cannot be
3741    /// written.
3742    pub fn restate(path: impl AsRef<Path>, views: &[ViewEntry]) -> Result<()> {
3743        let file = RealFilesystem::new().open(path.as_ref(), OpenMode::ReadWrite)?;
3744        let size = file.len()?;
3745        let (slot, bytes, _) = committed_slot(&*file, size)?;
3746        let (closed, _) = decode_catalog(&bytes, size)?;
3747        let generation = slot
3748            .generation
3749            .checked_add(1)
3750            .ok_or_else(|| invalid("native file generation overflow"))?;
3751        let catalog = encode_catalog(&closed, views)?;
3752        if catalog.len() > MAX_DIRECTORY {
3753            return Err(invalid("catalog exceeds the configured bound"));
3754        }
3755        file.write_at(size, &catalog)?;
3756        file.sync()?;
3757        let slot = Slot {
3758            offset: size,
3759            length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
3760            generation,
3761            hash: checksum(&catalog),
3762        };
3763        file.write_at(slot_offset(generation), &slot.bytes())?;
3764        file.sync()?;
3765        Ok(())
3766    }
3767
3768    /// Adds exact count, sum, distinct, bound, and bounded frequency certificates to an older file without
3769    /// rewriting table pages. The old slot remains readable until the new catalog is synced.
3770    pub fn certify_summaries(path: impl AsRef<Path>) -> Result<()> {
3771        let path = path.as_ref();
3772        let (_, size, slot, bytes, _) = slot_bytes(path)?;
3773        let (mut entries, views) = decode_catalog(&bytes, size)?;
3774        let native = Catalog::open(path)?;
3775        for entry in &mut entries {
3776            let reader = native.table(&entry.name)?;
3777            entry.nonzero.fill(None);
3778            entry.aggregates = reader_aggregate_sums(&reader)?;
3779            entry.distincts = (0..entry.fields.len())
3780                .map(|column| reader.distinct_values(column))
3781                .collect::<Result<Vec<_>>>()?;
3782            entry.extremes = reader_integer_extremes(&reader)?;
3783            entry.frequencies = reader_complete_numeric_frequencies(&reader)?;
3784        }
3785        let generation = slot
3786            .generation
3787            .checked_add(1)
3788            .ok_or_else(|| invalid("native file generation overflow"))?;
3789        let catalog = encode_catalog(&entries, &views)?;
3790        if catalog.len() > MAX_DIRECTORY {
3791            return Err(invalid("catalog exceeds the configured bound"));
3792        }
3793        let file = RealFilesystem::new().open(path, OpenMode::ReadWrite)?;
3794        file.write_at(size, &catalog)?;
3795        file.sync()?;
3796        let slot = Slot {
3797            offset: size,
3798            length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
3799            generation,
3800            hash: checksum(&catalog),
3801        };
3802        file.write_at(slot_offset(generation), &slot.bytes())?;
3803        file.sync()?;
3804        Ok(())
3805    }
3806
3807    /// The earlier name for [`Self::certify_summaries`].
3808    pub fn certify_counts(path: impl AsRef<Path>) -> Result<()> {
3809        Self::certify_summaries(path)
3810    }
3811}
3812
3813/// Appends one run of bytes at `at` and moves it past them, answering where they went.
3814///
3815/// The append half of [`attach`], which cannot use [`Writer::put`] because it is not writing a
3816/// table. Every byte a section costs goes through here, so the offsets in an extent table come
3817/// from one place.
3818fn append(file: &dyn rudb_io::File, at: &mut u64, bytes: &[u8]) -> Result<u64> {
3819    let offset = *at;
3820    file.write_at(offset, bytes)?;
3821    *at =
3822        at.checked_add(bytes.len() as u64).ok_or_else(|| invalid("native file length overflow"))?;
3823    Ok(offset)
3824}
3825
3826/// Writes one attachment's payload as extents and returns the entry that names it.
3827///
3828/// The split is by bytes, and the `first` of each extent is therefore a byte count. A section kind
3829/// whose extents should break on a row boundary instead will want to hand its extents over already
3830/// split; nothing needs that yet, and guessing at the shape of it now would be guessing.
3831fn write_section(
3832    file: &dyn rudb_io::File,
3833    at: &mut u64,
3834    one: &section::Attachment<'_>,
3835    generation: u64,
3836) -> Result<Section> {
3837    // A payload of nothing is the exception, and it is not a special case so much as a different
3838    // reading of the same field: an entry with no bytes has no header to be longer than them, and
3839    // `header_bytes` is what the structure would have cost. See `Section::refused`.
3840    if !one.bytes.is_empty() && one.header_bytes as usize > one.bytes.len() {
3841        return Err(invalid("a section's header is longer than its payload"));
3842    }
3843    let mut extents = Vec::new();
3844    let mut first = 0_u64;
3845    let extent_size =
3846        if one.kind == *section::SORTED_PROJECTION || one.kind == *section::RUN_PROJECTION {
3847            1 << 19
3848        } else {
3849            section::MAX_EXTENT as usize
3850        };
3851    for chunk in one.bytes.chunks(extent_size) {
3852        let offset = append(file, at, chunk)?;
3853        extents.push(section::Extent {
3854            offset,
3855            length: u32::try_from(chunk.len()).map_err(|_| invalid("extent length overflow"))?,
3856            hash: checksum(chunk),
3857            first,
3858        });
3859        first += chunk.len() as u64;
3860    }
3861    let mut table = Vec::with_capacity(extents.len() * section::EXTENT_BYTES);
3862    section::encode_extents(&extents, &mut table)?;
3863    // A payload of nothing is a section of no extents and no extent table, and its `extent_page`
3864    // is zero rather than the end of the file. Section 3.7 wants that entry to exist: it is how a
3865    // relationship that did not fit the budget is recorded as not built rather than forgotten.
3866    let extent_page = if table.is_empty() { 0 } else { append(file, at, &table)? };
3867    Ok(Section {
3868        kind: one.kind,
3869        id: one.id,
3870        generation,
3871        extents: u32::try_from(extents.len()).map_err(|_| invalid("too many extents"))?,
3872        extent_page,
3873        extent_bytes: u32::try_from(table.len()).map_err(|_| invalid("extent table overflow"))?,
3874        hash: checksum(&table),
3875        flags: one.flags,
3876        header_bytes: one.header_bytes,
3877    })
3878}
3879
3880/// Attaches graph sections to a table already committed in a file, without rewriting a page.
3881///
3882/// This is the second pass spec/graph/03-the-file-format.md section 3.8 asks for. A key map has to
3883/// exist before the link that uses it can be built, and it is built by reading the key column back,
3884/// so the structures of a table cannot be written during the load that wrote the table. They are
3885/// written afterwards, by this, and the file in between the two is a correct file that answers
3886/// every query more slowly.
3887///
3888/// Nothing is overwritten. The payloads, the extent tables, the new directory for this table and
3889/// the new catalog all go on the end of the file past the committed generation, and the last write
3890/// is the header slot, exactly as [`Writer::finish`] does it. So a crash anywhere in here leaves
3891/// the generation before it intact, and leaves unreferenced trailing bytes that the next commit
3892/// writes past.
3893///
3894/// An attachment replaces any section of the same kind and id, and every other section is carried
3895/// through untouched, including one whose kind this build does not know. The table's own generation
3896/// is carried through too, because attaching a section moves no row: see [`Table::generation`].
3897///
3898/// # Errors
3899///
3900/// If the file has no valid committed directory, is an older format than this build writes, holds
3901/// no table of that name, names a section whose payload cannot be written, or would end up naming
3902/// more sections than the format allows.
3903pub fn attach(
3904    path: impl AsRef<Path>,
3905    table: &str,
3906    attachments: &[section::Attachment<'_>],
3907) -> Result<Table> {
3908    let file = RealFilesystem::new().open(path.as_ref(), OpenMode::ReadWrite)?;
3909    let file = &*file;
3910    let size = file.len()?;
3911    let (slot, bytes, _) = committed_slot(file, size)?;
3912    let (mut entries, views) = decode_catalog(&bytes, size)?;
3913    let at = entries
3914        .iter()
3915        .position(|entry| entry.name == table)
3916        .ok_or_else(|| invalid(&format!("the file holds no table called {table}")))?;
3917    let mut version = [0; 4];
3918    read_at(file, 8, &mut version)?;
3919    let version = u32::from_le_bytes(version);
3920    // Readable is not the same as writable. A format 22 file has no section table, and giving its
3921    // directory one without moving the number in its header would leave a file that claims to be
3922    // format 22 and is not, which is worse than refusing. Rewriting it with this build is the
3923    // answer, and the format is at 0.3.x, so nobody has one of these that this project did not
3924    // just make.
3925    if version != FORMAT {
3926        return Err(invalid(&format!(
3927            "the file is format {version} and a graph section needs format {FORMAT}, so it has \
3928             to be written again"
3929        )));
3930    }
3931    let mut directory = vec![0; entries[at].directory.length as usize];
3932    read_at(file, entries[at].directory.offset, &mut directory)?;
3933    if checksum(&directory) != entries[at].directory.hash {
3934        return Err(invalid(&format!("the directory of table {table} does not checksum")));
3935    }
3936    let mut held = decode_directory(&directory, size)?;
3937    let mut cursor = size;
3938    for one in attachments {
3939        let written = write_section(file, &mut cursor, one, held.generation)?;
3940        held.sections.retain(|old| !(old.kind == one.kind && old.id == one.id));
3941        held.sections.push(written);
3942    }
3943    if held.sections.len() > MAX_SECTIONS {
3944        return Err(invalid("the table would name more sections than the bound allows"));
3945    }
3946    let encoded = encode_directory(&held)?;
3947    if encoded.len() > MAX_DIRECTORY {
3948        return Err(invalid("directory exceeds the configured bound"));
3949    }
3950    let offset = append(file, &mut cursor, &encoded)?;
3951    entries[at].directory = Page {
3952        offset,
3953        length: u32::try_from(encoded.len()).map_err(|_| invalid("directory length overflow"))?,
3954        hash: checksum(&encoded),
3955    };
3956    // The views the file already had, written back unchanged. Attaching a section to a table says
3957    // nothing about a view and must not drop one.
3958    let catalog = encode_catalog(&entries, &views)?;
3959    if catalog.len() > MAX_DIRECTORY {
3960        return Err(invalid("catalog exceeds the configured bound"));
3961    }
3962    let offset = append(file, &mut cursor, &catalog)?;
3963    file.sync()?;
3964    let generation =
3965        slot.generation.checked_add(1).ok_or_else(|| invalid("native file generation overflow"))?;
3966    let committed = Slot {
3967        offset,
3968        length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
3969        generation,
3970        hash: checksum(&catalog),
3971    };
3972    file.write_at(slot_offset(generation), &committed.bytes())?;
3973    file.sync()?;
3974    Ok(held)
3975}
3976
3977/// One column's frequency synopsis as values with their row counts, shared by every clone of a
3978/// reader.
3979type Synopsis = Arc<Vec<(Value, u64)>>;
3980
3981/// Reads committed native column pages without holding the table in memory.
3982#[derive(Debug, Clone)]
3983pub struct Reader {
3984    file: Arc<File>,
3985    table: Arc<Table>,
3986    dictionaries: Arc<Vec<OnceLock<Arc<Vector>>>>,
3987    /// Held while a global dictionary is being opened, one per column.
3988    ///
3989    /// The [`OnceLock`] above says whether one has been opened, which is the question a reader that
3990    /// already has it needs answered and is free. It does not say whether one is being opened, and
3991    /// the difference matters because every worker of a scan wants the same dictionary at the same
3992    /// moment. Without this they all miss, all read the page, all verify it and all decode it, and
3993    /// all but one throw the answer away. ClickBench 38 reads the URL dictionary, which is 515,958
3994    /// entries, and was paying for it twice.
3995    loading: Arc<Vec<Mutex<()>>>,
3996    /// Each column's frequency synopsis as values, the first time anything asks for it. See
3997    /// [`Reader::decode_frequencies`].
3998    frequency_values: Arc<Vec<OnceLock<Synopsis>>>,
3999    /// Stored frequency sections are decoded once per open table. A small directory can hold the
4000    /// summary inline, but a larger one otherwise rereads and decodes the same section on every
4001    /// plan and every summary-backed aggregate.
4002    frequency_summaries: Arc<Vec<OnceLock<Arc<FrequencySummary>>>>,
4003    /// How many global dictionaries have been opened. A scan of a dictionary column should open its
4004    /// dictionary once however many workers it has, and the test that says so is the only thing
4005    /// keeping it that way.
4006    opened: Arc<AtomicUsize>,
4007    /// The membership sieves of one stripe of one column, by column and then by stripe, read the
4008    /// first time a probe asks about them. A query filters on one or two columns and never looks at
4009    /// the rest, so reading these at open would be the whole index for the sake of a fraction of it.
4010    sieves: Arc<Vec<Vec<SieveSlot>>>,
4011    /// The per part ranges of one stripe of one column, by column and then by stripe, read the
4012    /// first time something compares that column and kept after that.
4013    part_ranges: Arc<Vec<Vec<RangeSlot>>>,
4014    /// Which stripe and which part of it every part of the table is, by table wide part number.
4015    places: Arc<Vec<Place>>,
4016    cache: Arc<Shelf>,
4017    /// Where the pages above are counted against the database's budget. See [`PagePool`].
4018    pool: PagePool,
4019    /// How many whole stripe pages have been read, which is what the sharing above is judged on. A
4020    /// scan of a column should read each of its stripes once however many workers it has.
4021    pages: Arc<AtomicUsize>,
4022    /// How many index sections have been read. A scan of a column should read each of its stripes
4023    /// once here too, and the test that says so is the only thing keeping it that way.
4024    indexes: Arc<AtomicUsize>,
4025    /// The file's size when it was opened, for [`Reader::layout`].
4026    size: u64,
4027    /// The committed directory's size, for [`Reader::layout`].
4028    directory: u64,
4029    /// What opening the file cost, which is a number rather than a claim.
4030    opening: Opening,
4031}
4032
4033/// What [`Reader::open`] read before it returned.
4034///
4035/// `spec/stats/04-in-memory.md` section 4.2 says opening a table reads the header and the directory
4036/// and nothing else, and once that document's statistics are in the file the tempting change is to
4037/// load a column summary or two on the way past, because they are small and the next query will
4038/// want them. A hundred milliseconds of that is a hundred milliseconds nobody asked for, and an
4039/// embedded database is opened by processes that are about to run one trivial query.
4040///
4041/// So the claim gets a number. Both of these are fixed by the schema and the stripe count and are
4042/// independent of how many rows the file holds, and the test that says so is what stops the
4043/// tempting change from landing quietly.
4044#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
4045pub struct Opening {
4046    /// How many times the file was read. The header, then each directory slot that looked valid
4047    /// enough to check, so three at the most.
4048    pub reads: u32,
4049    /// How many bytes those reads asked for.
4050    pub bytes: u64,
4051}
4052
4053/// What a reader has read, while it was being opened and since.
4054#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
4055pub struct Reads {
4056    /// What opening cost, before any query had been planned.
4057    pub opening: Opening,
4058    /// Whole stripe pages read since.
4059    pub pages: usize,
4060    /// Index sections read since.
4061    pub indexes: usize,
4062    /// Global dictionaries opened since. One per dictionary column that a query touched, however
4063    /// many workers touched it, which is a claim only a test can keep true.
4064    pub dictionaries: usize,
4065}
4066
4067/// Where one table wide part number lands.
4068#[derive(Debug, Clone, Copy)]
4069struct Place {
4070    stripe: u32,
4071    part: u32,
4072    rows: u32,
4073}
4074
4075/// One part's bytes inside one column page.
4076#[derive(Debug, Clone, Copy)]
4077struct PartSpan {
4078    start: usize,
4079    length: usize,
4080    hash: u64,
4081}
4082
4083/// What a reader holds for one stripe of one column.
4084///
4085/// The index is small and is loaded whether the caller wants the whole page or one part of it. The
4086/// page is loaded only by a scan, because a sparse fetch that wants a thousand rows out of sixty
4087/// four thousand would be reading sixty four times what it uses.
4088#[derive(Debug, Clone)]
4089struct CachedColumn {
4090    stripe: usize,
4091    index: Arc<Vec<PartSpan>>,
4092    page: Option<Arc<HeldPage>>,
4093}
4094
4095/// One stripe's page of one column, with which of its parts have already matched their checksums.
4096///
4097/// The bytes never change once they are read, so a part that matched once matches for as long as
4098/// the page is held. Hashing it again on every read was 3.5% of a `GROUP BY CounterID` over the
4099/// held pages of the ClickBench sample, run seventy times in one process. A part read without its
4100/// page is still checked every time, since those bytes come fresh off the file.
4101#[derive(Debug)]
4102struct HeldPage {
4103    bytes: Vec<u8>,
4104    checked: Vec<AtomicBool>,
4105}
4106
4107impl HeldPage {
4108    /// The bytes of part `part`, checked against `span` the first time anyone asks for them.
4109    fn part(&self, part: usize, span: PartSpan) -> Result<&[u8]> {
4110        let bytes = part_bytes(&self.bytes, span)?;
4111        let checked = self.checked.get(part).ok_or_else(|| invalid("part index out of range"))?;
4112        if !checked.load(Atomic::Relaxed) {
4113            verify_part(bytes, span)?;
4114            checked.store(true, Atomic::Relaxed);
4115        }
4116        Ok(bytes)
4117    }
4118}
4119
4120/// Checks one part's bytes against the hash its index carries for them.
4121fn verify_part(bytes: &[u8], span: PartSpan) -> Result<()> {
4122    let got = checksum(bytes);
4123    if got != span.hash {
4124        return Err(invalid(&format!(
4125            "column page checksum differs, part at {}+{} bytes, wanted {:016x} and got {got:016x}",
4126            span.start, span.length, span.hash,
4127        )));
4128    }
4129    Ok(())
4130}
4131
4132/// One column's stripes a reader holds, and which of them somebody is reading right now.
4133///
4134/// The pages are one slot per stripe of the table rather than a list of the ones being kept, so
4135/// finding a page is an index and not a walk. That matters because the walk happened under the
4136/// lock, once per part per column, and a scan that gives a whole stripe to each of thirty two
4137/// workers keeps enough pages that walking them was the longest thing the lock was held for. The
4138/// slots cost a pointer per stripe per column, which on the ClickBench file is eight kilobytes
4139/// against the forty megabytes of pages they point at. `order` is which of them are filled, oldest
4140/// first, because that is the one thing the slots cannot say by themselves.
4141///
4142/// `loading` is what keeps a scan from reading the same page once per worker. It is a list and not
4143/// a set because it holds at most one stripe per worker on the column and is walked far less often
4144/// than a hash of it would be built.
4145///
4146/// `index` is every index this reader has ever read for the column, one slot per stripe, and it is
4147/// never evicted. An index is a few hundred bytes and a page is a quarter of a megabyte, so the two
4148/// do not belong under the same budget. Riding in the page cache meant a worker that came back to a
4149/// stripe after its page had been evicted read the index again with it, which on the full
4150/// ClickBench file was about thirteen hundred reads out of a hundred and fourteen thousand.
4151///
4152/// `seen` is which stripes have had their page read before, and `passing` is the pages read for the
4153/// first time that are still held, oldest first. A page goes into the pool the second time it is
4154/// read and not the first, which is the rule [`NativeText`] follows for its decoded blocks. A
4155/// process that runs one statement, which is how a script or a benchmark uses the engine, reads
4156/// each page once, and with no memory limit the pool kept every one of them to the end: ClickBench
4157/// q33 held all of `WatchID` and `ClientIP` at its peak for a second scan that never came. The
4158/// first read now keeps a page only while it is among the column's floor of newest ones, and a
4159/// session that scans the table again pays one more read of each page and keeps it from then on.
4160#[derive(Debug, Default)]
4161struct Cached {
4162    pages: Vec<Option<Resident>>,
4163    loading: Vec<usize>,
4164    index: Vec<Option<Arc<Vec<PartSpan>>>>,
4165    seen: Vec<bool>,
4166    passing: VecDeque<usize>,
4167}
4168
4169/// One page a reader holds, and whether anyone has read it since the pool last looked.
4170#[derive(Debug, Clone)]
4171struct Resident {
4172    page: Arc<HeldPage>,
4173    used: Arc<AtomicBool>,
4174}
4175
4176/// Every column's pages of one reader, with how many each column holds and the floor under that.
4177#[derive(Debug)]
4178struct Shelf {
4179    columns: Vec<Mutex<Cached>>,
4180    /// How many pages each column holds right now. Counted outside the column locks so that the
4181    /// pool can tell whether a column is at its floor without taking a lock it might be under.
4182    held: Vec<AtomicUsize>,
4183    /// How many stripes of one column are kept whatever the budget says. See
4184    /// [`CACHED_STRIPES_PER_COLUMN`] for what sets it and [`Reader::keep_stripes`] for who raises it.
4185    kept: AtomicUsize,
4186}
4187
4188/// The pages every reader of one database keeps, under one budget in bytes.
4189///
4190/// A reader lives as long as the database does, so the pages it holds are what the next query finds
4191/// already in memory. They used to be four stripes a column, oldest out first, which on TPC-H SF1
4192/// meant every query read every page of lineitem off the file again and paid the system call for
4193/// it. Keeping every page there costs 38 MB and took a third of the system time off the suite.
4194///
4195/// So the question is no longer how many stripes a column keeps but how many bytes the database
4196/// does, and one budget answers it for every reader at once. A table nobody queries gives its pages
4197/// up to one that is being queried, which a count per column cannot do.
4198///
4199/// Pages leave by the clock. Each has a bit a read sets, and when the pool is over budget it walks
4200/// from the oldest: a page with the bit set loses the bit and goes round again, and a page without
4201/// it goes. That keeps what is read over and over and lets a page one scan read once go first.
4202///
4203/// The old count is still a floor. A column never gives up a page while it holds four or fewer,
4204/// because a scan whose workers evict each other's pages reads a quarter of a megabyte for every
4205/// part it takes, and a budget of zero is the cache as it was before the pool existed.
4206#[derive(Debug, Clone, Default)]
4207pub struct PagePool {
4208    ring: Arc<Mutex<Ring>>,
4209    budget: Arc<AtomicUsize>,
4210}
4211
4212#[derive(Debug, Default)]
4213struct Ring {
4214    held: VecDeque<Held>,
4215    bytes: usize,
4216}
4217
4218/// One page in the pool, pointing back at the reader that holds it.
4219///
4220/// Weak, because a reader that has gone, which every reader does at a checkpoint, should take its
4221/// pages with it and not have them kept alive by the pool.
4222#[derive(Debug)]
4223struct Held {
4224    shelf: Weak<Shelf>,
4225    column: usize,
4226    stripe: usize,
4227    bytes: usize,
4228    used: Arc<AtomicBool>,
4229}
4230
4231impl PagePool {
4232    /// A pool that keeps up to `budget` bytes of pages beyond each column's floor.
4233    #[must_use]
4234    pub fn new(budget: usize) -> Self {
4235        let pool = Self::default();
4236        pool.budget.store(budget, Atomic::Relaxed);
4237        pool
4238    }
4239
4240    /// The bytes of pages the pool is counting now.
4241    ///
4242    /// # Panics
4243    ///
4244    /// If the pool's lock is poisoned, which takes a panic while it was held.
4245    #[must_use]
4246    pub fn bytes(&self) -> usize {
4247        self.ring.lock().map_or(0, |ring| ring.bytes)
4248    }
4249
4250    /// Counts a page a reader has just taken in, and lets pages go until the pool is back under its
4251    /// budget or it has looked at every page once.
4252    ///
4253    /// Called with no column lock held. The pages that go are chosen under the pool's lock and
4254    /// dropped under their column's lock afterwards, so no thread ever holds both.
4255    fn admit(&self, held: Held) {
4256        let budget = self.budget.load(Atomic::Relaxed);
4257        let mut gone = Vec::new();
4258        {
4259            let Ok(mut ring) = self.ring.lock() else { return };
4260            ring.bytes += held.bytes;
4261            ring.held.push_back(held);
4262            // One lap and no more. A page read since the last pass loses its bit on this one and
4263            // can only go on a later one, which is the second chance the clock is named for.
4264            let mut looked = 0;
4265            let limit = ring.held.len();
4266            while ring.bytes > budget && looked < limit {
4267                looked += 1;
4268                let Some(entry) = ring.held.pop_front() else { break };
4269                let Some(shelf) = entry.shelf.upgrade() else {
4270                    ring.bytes -= entry.bytes;
4271                    continue;
4272                };
4273                if entry.used.swap(false, Atomic::Relaxed) {
4274                    ring.held.push_back(entry);
4275                    continue;
4276                }
4277                let count = &shelf.held[entry.column];
4278                if count.load(Atomic::Relaxed) <= shelf.kept.load(Atomic::Relaxed).max(1) {
4279                    ring.held.push_back(entry);
4280                    continue;
4281                }
4282                count.fetch_sub(1, Atomic::Relaxed);
4283                ring.bytes -= entry.bytes;
4284                gone.push((shelf, entry));
4285            }
4286            // A reader that has gone leaves its entries behind, and with a budget nobody reaches
4287            // they would pile up one checkpoint after another. The front is where the oldest are.
4288            while ring.held.front().is_some_and(|entry| entry.shelf.strong_count() == 0) {
4289                if let Some(entry) = ring.held.pop_front() {
4290                    ring.bytes -= entry.bytes;
4291                }
4292            }
4293        }
4294        for (shelf, entry) in gone {
4295            let Ok(mut cached) = shelf.columns[entry.column].lock() else { continue };
4296            if let Some(slot) = cached.pages.get_mut(entry.stripe) {
4297                if slot.as_ref().is_some_and(|slot| Arc::ptr_eq(&slot.used, &entry.used)) {
4298                    *slot = None;
4299                }
4300            }
4301        }
4302    }
4303}
4304
4305/// Stripes of one column a reader keeps the bytes of, when nobody has asked for more.
4306///
4307/// This has to hold at least as many stripes as a column has workers in it at once, or the workers
4308/// evict each other's pages and read them again. Four is what a scan that hands parts out in order
4309/// needs, because then every worker is within a few parts of every other and at most a couple of
4310/// stripes are open at a time. A scan that hands a whole stripe to each worker has one stripe open
4311/// per worker for the length of that stripe, and it says so with [`Reader::keep_stripes`] rather
4312/// than paying for sixteen slots on every table that is read one part at a time.
4313///
4314/// It multiplies by the page size, which is a quarter of a megabyte for a four byte column, and by
4315/// the number of columns a query touches.
4316const CACHED_STRIPES_PER_COLUMN: usize = 4;
4317
4318/// The sieves of one stripe of one column, once somebody has asked for them.
4319type SieveSlot = OnceLock<Arc<Vec<Option<Sieve>>>>;
4320
4321type RangeSlot = OnceLock<Arc<Vec<Range>>>;
4322
4323#[derive(Debug)]
4324struct NativeText {
4325    file: Arc<File>,
4326    /// How many values the dictionary holds.
4327    values: usize,
4328    /// Where each value ends inside its payload block, packed at `offset_bits` in runs of
4329    /// [`TEXT_OFFSET_RUN`].
4330    ///
4331    /// Ends rather than starts, because then a block of 1,024 values is 1,024 numbers rather than
4332    /// 1,025: the start of a value is the end of the one before it, and the first value of a block
4333    /// starts at zero by construction. Relative to the block rather than to the payload, because a
4334    /// reader decodes a whole block and slices it, so an offset into the payload is a number it
4335    /// would have to subtract a base from anyway.
4336    ///
4337    /// The vector is the index as it was read, so the offsets start after the header, and
4338    /// [`Self::packed`] is where they are read from.
4339    offsets: Vec<u8>,
4340    /// Bits one offset is packed at, which is what the largest block of this column spans and is the
4341    /// same for every block of it.
4342    offset_bits: usize,
4343    /// The same ends unpacked, built once enough readers have asked for one at a time.
4344    ///
4345    /// Reading one offset out of the packed form costs about fifty instructions: a division to find
4346    /// the run, a bounds check to slice it, a shift to reach the bit the value starts at and a
4347    /// narrowing on the way out. That is the right price for a reader that wants a handful. It is
4348    /// the wrong price for `STRLEN` over a column, which asks for one per row and nothing else, and
4349    /// where a million of them was a third of ClickBench 28.
4350    ///
4351    /// With the ends unpacked every read is a load, and a vector of lengths is one loop over them.
4352    /// The table is built only once the reads say it will be used, which is what
4353    /// [`Self::ends_worth_unpacking`] decides and [`Self::ends_asked`] counts towards, because a
4354    /// table built for a reader that wanted three values is four bytes a value spent on nothing.
4355    value_ends: OnceLock<Option<Vec<u32>>>,
4356    /// The length of every value, worked out of [`Self::value_ends`] the first time a vector of
4357    /// lengths is asked for.
4358    ///
4359    /// A length out of the ends is two loads, a test for whether the value opens its block and a
4360    /// check that it does not end before it starts, which came to thirteen instructions a row on
4361    /// ClickBench 28. Out of this it is one load. The order is checked once for the whole table
4362    /// while it is built, and a column that fails it gets no table and goes on reading the ends,
4363    /// which is where the error is reported. Two bytes a value where every value is short enough,
4364    /// four otherwise, and only for a column something has asked the length of a vector at a time.
4365    value_lens: OnceLock<Option<Lengths>>,
4366    /// How many single offset reads have come in while the table is not built.
4367    ///
4368    /// Relaxed, and read only against a threshold, so two threads racing here means the table is
4369    /// built one read early or one read late. Counting stops the moment the table exists, because
4370    /// [`OnceLock::get`] settles it before this is touched.
4371    ends_asked: AtomicUsize,
4372    /// How many entries the sorted order has, which is the value count.
4373    ranks: usize,
4374    /// Where the sorted order starts in the file. It is read a block at a time and only when
4375    /// something searches it, so a query that never compares this column against a literal never
4376    /// touches it at all.
4377    rank_at: u64,
4378    /// Where each block of the sorted order ends, as a byte offset from `rank_at`. A block is packed
4379    /// at whatever width its own heads need, so unlike the entries it replaced its length is not
4380    /// arithmetic on the block number.
4381    rank_ends: Vec<u64>,
4382    rank_hashes: Vec<u64>,
4383    rank_blocks: Vec<OnceLock<Result<Vec<u8>>>>,
4384    /// Bits one code is packed at, which is what the value count needs and is the same for every
4385    /// block of the column.
4386    code_bits: usize,
4387    /// The sorted order turned round, built the first time a reader asks for it.
4388    ///
4389    /// Four bytes per value against the four the offsets already hold, so a column that has this is
4390    /// carrying half again what it carried before rather than something of a new order. It is built
4391    /// only when something asks, which is a grouped min or max over this column and nothing else,
4392    /// and that reader was going to read the payload of this column once per row otherwise.
4393    code_ranks: OnceLock<Option<Vec<u32>>>,
4394    /// Where each block of the payload starts in the file, and how many stored bytes it is.
4395    ///
4396    /// Absolute rather than an offset from a base the blocks share, because a block is written the
4397    /// moment it fills and what comes after it in the file is whatever the load wrote next. A file
4398    /// old enough to have them back to back is read into these same two lists by adding the base to
4399    /// the ends it carries, so nothing below here knows which kind of file it came from.
4400    starts: Vec<u64>,
4401    lengths: Vec<u64>,
4402    hashes: Vec<u64>,
4403    /// Conservative four-byte substring signatures, read only by a compatible LIKE filter.
4404    grams: Option<NativeGrams>,
4405    /// The payload, read and decoded a block at a time and kept after that.
4406    blocks: Vec<OnceLock<Result<Vec<u8>>>>,
4407    /// The length in characters of every value of a block, worked out the first time `length` asks
4408    /// for a value in that block.
4409    ///
4410    /// Kept instead of the block it was counted out of. `length` reads every row of a column, and
4411    /// reading the bytes through [`Self::payload_block`] kept every block it touched, which is every
4412    /// distinct value of the column decoded: seven string columns of ClickBench held 13.9 GB to
4413    /// answer seven `max(length(...))`. The counts are four bytes a value, so the same scan keeps
4414    /// the counts and decodes each block once, the same number of times it did before.
4415    char_lens: Vec<OnceLock<Box<[u32]>>>,
4416    /// How many decoded payload bytes this column keeps before a sweep stops keeping what it reads.
4417    /// [`TEXT_KEEP_BUDGET`] everywhere but in the test of the ceiling.
4418    keep_budget: usize,
4419    /// Roughly how many decoded payload bytes are being kept, which is what [`TEXT_KEEP_BUDGET`]
4420    /// is measured against.
4421    ///
4422    /// Roughly, because two threads that keep the same block at the same time both add its length
4423    /// while [`OnceLock`] keeps one of the two. That makes the count read high and the budget bind
4424    /// a little early, which is the harmless direction, and it costs one relaxed add a block rather
4425    /// than a lock on the path every scan of a string column goes through.
4426    payload_kept: AtomicUsize,
4427    /// Which payload blocks a sweep has decoded before, one flag a block.
4428    ///
4429    /// A sweep keeps a block the second time it decodes it and not the first. A process that runs
4430    /// one statement, which is how a benchmark or a script uses the engine, sweeps each block once
4431    /// and so keeps nothing: on ten million rows a `URL LIKE` held 396 MB with every block kept and
4432    /// 97 MB with none, for the same processor time. A session that asks again pays the decode one
4433    /// more time and reads kept blocks from then on, under the same [`TEXT_KEEP_BUDGET`].
4434    swept: Vec<AtomicBool>,
4435    /// How many blocks [`TextSource::visit_at`] has decoded and dropped because the column was
4436    /// already holding its [`TEXT_KEEP_BUDGET`].
4437    ///
4438    /// A sweep reads the dictionary in order and touches a block once, so dropping what it reads
4439    /// past the budget costs one decode a block and bounds the column. A visit reads a vector of
4440    /// codes, and the codes of a scan land all over the dictionary: on ten million rows of
4441    /// ClickBench each vector of two thousand `URL`s touches about a hundred and forty of its two
4442    /// and a half thousand blocks, and so does the next one. A cache holding a tenth of the column
4443    /// still misses half of those, and dropping every block past the budget would decode the
4444    /// column hundreds of times over to answer one `lower(URL)`. So a visit drops past the budget
4445    /// only until it has dropped as many blocks as the column has, which is what a read whose codes
4446    /// are few or clustered never reaches, and keeps what it reads after that, the way a row at a
4447    /// time read always did. That bounds what a visit can cost over the old read at one more decode
4448    /// of the column.
4449    visit_dropped: AtomicUsize,
4450    /// The boundaries this dictionary has already been searched for, by the value searched for.
4451    ///
4452    /// A search is the expensive thing this type does. It settles a probe on the stored head where
4453    /// it can and reads a value where it cannot, and reading a value decodes the payload block it
4454    /// sits in, so one search can cost several blocks. The thing that makes remembering worth it is
4455    /// that the same search comes back: a top N asks once a chunk whether anything left can beat its
4456    /// worst candidate, and the worst candidate settles long before the chunks run out.
4457    ///
4458    /// Shared across the instances of a scan rather than kept per instance, because each of them has
4459    /// its own worst candidate and all of them are searching the same dictionary. One lock per chunk
4460    /// is nothing next to a probe of a file.
4461    ///
4462    /// Bounded by [`TEXT_SEARCH_MEMO`] and emptied rather than evicted when it is full. What fills
4463    /// it is a top N improving its bound, which happens a few dozen times and then stops, so the
4464    /// bound is there for the filter that searches for a different literal every chunk rather than
4465    /// for anything this is meant to help.
4466    searched: Mutex<HashMap<Vec<u8>, (usize, bool)>>,
4467}
4468
4469#[derive(Debug)]
4470struct NativeGrams {
4471    start: u64,
4472    length: usize,
4473    /// How long one block's signature is.
4474    width: usize,
4475    hash: u64,
4476    /// For each literal asked about lately, whether each block might hold it.
4477    ///
4478    /// The answer for every block at once, worked out by one pass over the signatures a window at a
4479    /// time, rather than the signatures read in and kept. On ClickBench `URL` they are 21 MB for
4480    /// ten million rows and a verdict is 2,650 flags, and a filter asks the same question of every
4481    /// block, so the pass is paid once and what stays resident is the flags.
4482    verdicts: Mutex<Vec<Verdict>>,
4483}
4484
4485/// A literal and whether each block might hold it.
4486type Verdict = (Vec<u8>, Arc<[bool]>);
4487
4488/// How many literals a column remembers the verdicts of.
4489const GRAM_VERDICTS: usize = 8;
4490
4491impl NativeGrams {
4492    /// Whether each block might hold `literal`, remembered or worked out now.
4493    ///
4494    /// The lock is held over the pass so that the threads of one scan, which all ask about the
4495    /// same literal at the start, read the signatures once between them.
4496    fn verdicts(&self, file: &File, literal: &[u8]) -> Result<Arc<[bool]>> {
4497        let mut held = self.verdicts.lock().map_err(|_| invalid("a poisoned signature verdict"))?;
4498        if let Some((_, verdict)) = held.iter().find(|(asked, _)| asked == literal) {
4499            return Ok(Arc::clone(verdict));
4500        }
4501        let wanted = literal.windows(4).map(|gram| gram_bits(gram, self.width)).collect::<Vec<_>>();
4502        let mut verdict = Vec::with_capacity(self.length / self.width);
4503        let window = GRAM_WINDOW / self.width * self.width;
4504        let hash = walk_checksummed(file, self.start, self.length, window, |bytes| {
4505            verdict.extend(bytes.chunks(self.width).map(|bits| {
4506                wanted
4507                    .iter()
4508                    .flatten()
4509                    .all(|&bit| bits.get(bit / 8).is_some_and(|byte| byte & (1 << (bit % 8)) != 0))
4510            }));
4511            Ok(())
4512        })?;
4513        if hash != self.hash {
4514            return Err(invalid("global dictionary substring signatures checksum differs"));
4515        }
4516        let verdict: Arc<[bool]> = verdict.into();
4517        if held.len() >= GRAM_VERDICTS {
4518            held.remove(0);
4519        }
4520        held.push((literal.to_vec(), Arc::clone(&verdict)));
4521        Ok(verdict)
4522    }
4523
4524    fn footprint(&self) -> usize {
4525        self.verdicts.lock().map_or(0, |held| {
4526            held.iter().map(|(asked, verdict)| asked.capacity() + verdict.len()).sum()
4527        })
4528    }
4529}
4530
4531/// How many searched for values a column's dictionary remembers the boundary of.
4532///
4533/// See [`NativeText::searched`]. Small because the case it is for repeats one value, not because a
4534/// larger one would be wrong.
4535const TEXT_SEARCH_MEMO: usize = 64;
4536
4537/// How many values of a dictionary go in one block of the payload.
4538///
4539/// The block is the unit the string cascade encodes, the unit a checksum covers, and the unit a
4540/// reader has to decode to get at a single value, so it is the one number the payload format turns
4541/// on. Blocking by values rather than by bytes is what keeps a value out of two blocks at once: the
4542/// block holding a code is `code / TEXT_PAYLOAD_VALUES` and nothing has to be stitched.
4543///
4544/// A probe on the five ClickBench columns that have a dictionary worth the name, written up on
4545/// #347, measured the ratio and the decode speed at 128, 256, 512, 1,024 and 4,096 values. Both get
4546/// better all the way up, because front coding and the LZ matcher have more to look back at and
4547/// because the per chunk setup is spread over more values. What stops it is the point read: a query
4548/// that wants ten values has to decode ten blocks, so the block is what a lookup costs. At 1,024
4549/// values a block is between 67 KB and 394 KB decoded across those five columns, and the ratios are
4550/// 2.3 to 4.5. Going up to 4,096 buys two to six percent more and makes a block as much as 1.5 MB.
4551/// Going down to 512 gives up five to nine percent.
4552const TEXT_PAYLOAD_VALUES: usize = 1024;
4553
4554/// Eight KiB per payload block, which is what makes a four-byte substring a useful negative test on
4555/// a column of URLs.
4556///
4557/// Two KiB was the first answer and on ClickBench `URL` it proved almost nothing. A block of 1,024
4558/// sorted URLs holds about seventeen thousand distinct four-byte grams, and at two bits each that
4559/// set nine in ten of the sixteen thousand bits there were, so `LIKE '%google%'` passed most blocks
4560/// it had no match in and decoded them. At eight KiB four bits in ten are set, and of the 2,650
4561/// blocks of `URL` in ten million rows a needle that is in none of them passes 36. The signatures
4562/// are not read into memory, see [`NativeGrams::verdicts`], so the width costs file and not
4563/// resident memory.
4564const TEXT_GRAM_BYTES: usize = 8192;
4565
4566/// The signature width of a format 28 file, which is still read.
4567const NARROW_GRAM_BYTES: usize = 2048;
4568
4569/// How much of a column's signatures a verdict reads at a time.
4570const GRAM_WINDOW: usize = 256 << 10;
4571
4572/// A fast mixing step for exactly four bytes, shared by load and query, into a signature of
4573/// `width` bytes.
4574fn gram_bits(bytes: &[u8], width: usize) -> [usize; 2] {
4575    let original = u32::from_le_bytes(bytes.try_into().expect("a four-byte gram"));
4576    let mut first = original ^ (original >> 16);
4577    first = first.wrapping_mul(0x7feb_352d);
4578    first ^= first >> 15;
4579    let mut second = original ^ (original >> 17);
4580    second = second.wrapping_mul(0x846c_a68b);
4581    second ^= second >> 16;
4582    let mask = width * 8 - 1;
4583    [(first as usize) & mask, (second as usize) & mask]
4584}
4585
4586/// How many decoded payload bytes one dictionary keeps before a sweep stops keeping what it reads.
4587///
4588/// A sweep of the whole dictionary decodes every block whatever it does, and the only question is
4589/// whether it hangs on to them. Keeping all of them is 4.2 GB on ClickBench `URL` at a hundred
4590/// million rows, which is what #997 was right to stop. Keeping none of them means the next query
4591/// asking the same thing decodes all of it again, and on the same column at a million rows that
4592/// took a `LIKE` from 2.7 ms to 16.2 ms, because the decode used to be paid once by a session and
4593/// is now paid by every statement in it. Neither end is the answer. A bound is.
4594///
4595/// So a sweep keeps what it decodes until the column is holding this much and decodes without
4596/// keeping after that. At a million rows the five ClickBench string columns decode to between 8 MB
4597/// and 85 MB, so they sit inside it and a repeated `LIKE` reads a decoded block rather than a
4598/// stored one. At a hundred million rows `URL` fills it and the rest of that column is read and
4599/// dropped, which is the old cost on the part that does not fit and none of the old footprint.
4600///
4601/// Two hundred and fifty six megabytes a column is a number and not a policy, and the policy is
4602/// what should replace it: this wants to be a buffer pool over the whole database, sized against
4603/// the memory limit the session was given, with the blocks of every column competing for it and the
4604/// least useful one evicted. That is F2 work. What is here is the part of it that can be written
4605/// without an eviction order, which is a ceiling.
4606const TEXT_KEEP_BUDGET: usize = 256 * 1024 * 1024;
4607
4608/// The length of every value of a column, as narrow as the longest of them allows.
4609///
4610/// The table is read at the codes a vector holds, which on a column the size of ClickBench `URL`
4611/// land all over it, so what a length costs is whether its line is in cache. Half a million URLs
4612/// are two megabytes at four bytes a length and one at two, which is the difference between the
4613/// table sitting in the second level cache or not.
4614#[derive(Debug)]
4615enum Lengths {
4616    /// Every length fits in sixteen bits.
4617    Narrow(Vec<u16>),
4618    /// Some value is longer than that.
4619    Wide(Vec<u32>),
4620}
4621
4622impl Lengths {
4623    /// The lengths at `indices`, appended to `into`, and zero for a position past the end, which
4624    /// is what a row at a time read says.
4625    fn extend_at(&self, indices: &[u32], into: &mut Vec<i64>) {
4626        match self {
4627            Lengths::Narrow(lens) => into.extend(
4628                indices
4629                    .iter()
4630                    .map(|&index| lens.get(index as usize).map_or(0, |&len| i64::from(len))),
4631            ),
4632            Lengths::Wide(lens) => into.extend(
4633                indices
4634                    .iter()
4635                    .map(|&index| lens.get(index as usize).map_or(0, |&len| i64::from(len))),
4636            ),
4637        }
4638    }
4639
4640    /// The bytes the table holds on to.
4641    fn footprint(&self) -> usize {
4642        match self {
4643            Lengths::Narrow(lens) => lens.capacity() * size_of::<u16>(),
4644            Lengths::Wide(lens) => lens.capacity() * size_of::<u32>(),
4645        }
4646    }
4647}
4648
4649/// The length of every value out of where each one ends inside its payload block, or `None` for
4650/// ends that go backwards somewhere inside a block.
4651///
4652/// A value that opens a block starts at zero and every other one starts where the value before it
4653/// ends, so a block is a run of differences.
4654///
4655/// Built at two bytes a length straight away, and built again at four only when some value turns
4656/// out too long for that, which is rare enough that the second pass is not worth avoiding.
4657fn lengths_of(ends: &[u32]) -> Option<Lengths> {
4658    match lengths_as::<u16>(ends)? {
4659        Some(narrow) => Some(Lengths::Narrow(narrow)),
4660        None => lengths_as::<u32>(ends)?.map(Lengths::Wide),
4661    }
4662}
4663
4664/// [`lengths_of`] at one width: `None` for ends that go backwards, and `Some(None)` for a length
4665/// that does not fit in `T`.
4666fn lengths_as<T: TryFrom<u32>>(ends: &[u32]) -> Option<Option<Vec<T>>> {
4667    let mut lens = Vec::with_capacity(ends.len());
4668    for block in ends.chunks(TEXT_PAYLOAD_VALUES) {
4669        let mut start = 0;
4670        for &end in block {
4671            let Ok(len) = T::try_from(end.checked_sub(start)?) else {
4672                return Some(None);
4673            };
4674            lens.push(len);
4675            start = end;
4676        }
4677    }
4678    Some(Some(lens))
4679}
4680
4681/// How many offsets go in one packed run.
4682///
4683/// A payload block holds 1,024 values and `bitpack::pack_tail` takes fewer than 1,024 at a time,
4684/// since a whole unit of that many belongs in the transposed layout instead. So the offsets of a
4685/// block go in two runs. Five hundred and twelve values at any width is a whole number of bytes, so
4686/// a run starts where a multiply says it does and nothing is padded.
4687const TEXT_OFFSET_RUN: usize = 512;
4688
4689/// Bytes at the front of a global dictionary index: the value count, the values a payload block
4690/// holds, the block count and the bits an offset is packed at.
4691const DICTIONARY_HEADER: usize = 16;
4692
4693/// Set beside the offset width in the fourth word of a global dictionary index, meaning each
4694/// payload block says where in the file it starts and how long it is, rather than sitting directly
4695/// behind the block before it.
4696///
4697/// In that word rather than in a word of its own because the width is at most 32 and lives in a
4698/// `u32`, so the top of it has never been anything. A build old enough not to know the flag reads
4699/// the file's format before it reads any of this and refuses it there, and if it somehow did get
4700/// here it would find an offset width of two billion and say so.
4701///
4702/// The point of the flag is that a block written the moment it fills does not know what will be
4703/// written after it, so the payload of a column cannot be one run of bytes unless the whole column
4704/// is held until the file is closed. That is the memory the load cannot afford. What it costs is
4705/// eight bytes a block, against the block being a thousand values.
4706const DICTIONARY_SCATTERED: u32 = 1 << 31;
4707/// The dictionary index carries one four-byte substring signature per payload block.
4708const DICTIONARY_GRAMS: u32 = 1 << 30;
4709/// Each signature is [`TEXT_GRAM_BYTES`] long rather than the [`NARROW_GRAM_BYTES`] a format 28
4710/// file wrote.
4711const DICTIONARY_WIDE_GRAMS: u32 = 1 << 29;
4712/// Every flag the width word of a dictionary can carry above the offset width.
4713const DICTIONARY_FLAGS: u32 = DICTIONARY_SCATTERED | DICTIONARY_GRAMS | DICTIONARY_WIDE_GRAMS;
4714
4715/// How many entries of a dictionary's sorted order sit in one block that is read and checked as a
4716/// unit.
4717///
4718/// Five hundred and twelve entries is between two and three kilobytes on the ClickBench string
4719/// columns, which is well under a page. A binary search over half a million entries makes nineteen
4720/// probes, and the first ten land in ten different blocks while the last nine land in the one block
4721/// that holds the answer, so the whole search reads about thirty kilobytes of a megabyte of order. A
4722/// smaller block would save a little on the early probes, cost a checksum and an end list four times
4723/// as long, and give the heads less to share a base with. A larger one would read more than it uses
4724/// on every probe.
4725const TEXT_RANK_BLOCK: usize = 512;
4726
4727/// Bytes at the front of a rank block, which is the base of its heads and the width they are packed
4728/// at.
4729///
4730/// An entry used to be twelve bytes flat, eight for the head and four for the code, and on the five
4731/// ClickBench columns that have a dictionary worth the name that was 744 MB of a 12.2 GB file. Both
4732/// halves of it are nearly empty. The heads are the first eight bytes of the values in sorted order,
4733/// so a block of five hundred and twelve of them spans a tiny slice of the column, and on a column of
4734/// URLs they are all `http://w` and the block holds one distinct head. The codes are positions in a
4735/// dictionary of eighteen million, which is twenty five bits and not thirty two.
4736///
4737/// So a block now writes the smallest head in it, the bits the largest is above that, and the heads
4738/// and the codes packed at the width each needs. A block where every head agrees costs nine bytes
4739/// and the codes.
4740const RANK_BLOCK_HEADER: usize = size_of::<u64>() + 1;
4741
4742impl NativeText {
4743    /// One block of the payload, read and decoded the first time anything asks for a value in it.
4744    ///
4745    /// The bytes handed back are the values of the block laid end to end, which is what the offsets
4746    /// describe, so a caller slices it with the offsets it already has. Where the block sits in the
4747    /// file is the only thing the caller cannot work out for itself, because the stored form is
4748    /// shorter than the decoded one and by a different amount in every block.
4749    fn payload_block(&self, block: usize) -> Result<Option<&[u8]>> {
4750        let Some(slot) = self.blocks.get(block) else { return Ok(None) };
4751        let bytes = slot.get_or_init(|| self.decode_block(block)).as_ref().map_err(Clone::clone)?;
4752        Ok(Some(bytes.as_slice()))
4753    }
4754
4755    /// The character length of every value in one block, counted the first time it is asked for.
4756    ///
4757    /// The block is read out of [`Self::blocks`] where something already kept it and decoded and
4758    /// dropped where nothing did, so counting never adds a block to what this column holds. Two
4759    /// threads asking for the same block at once both count it and one of the two answers is kept,
4760    /// which costs a decode and is cheaper than a lock on every lookup.
4761    fn block_chars(&self, block: usize) -> Result<&[u32]> {
4762        let slot = self
4763            .char_lens
4764            .get(block)
4765            .ok_or_else(|| invalid("a block past the global dictionary"))?;
4766        if let Some(lens) = slot.get() {
4767            return Ok(lens);
4768        }
4769        let decoded;
4770        let bytes: &[u8] = match self.blocks.get(block).and_then(OnceLock::get) {
4771            Some(Ok(kept)) => kept,
4772            _ => {
4773                decoded = self.decode_block(block)?;
4774                &decoded
4775            }
4776        };
4777        let first = block * TEXT_PAYLOAD_VALUES;
4778        let last = (first + TEXT_PAYLOAD_VALUES).min(self.values);
4779        let ends = self.ends_within(first, last)?;
4780        if ends.len() != last - first {
4781            return Err(invalid("global dictionary offsets are short"));
4782        }
4783        let mut lens = Vec::with_capacity(ends.len());
4784        let mut start = u64::from(self.start_within(first)?);
4785        for &end in &ends {
4786            let value = usize::try_from(start)
4787                .ok()
4788                .zip(usize::try_from(end).ok())
4789                .and_then(|(from, to)| bytes.get(from..to))
4790                .ok_or_else(|| invalid("global dictionary value is past its block"))?;
4791            // A continuation byte of UTF-8 is `0b10xx_xxxx` and every other byte starts a
4792            // character, so the bytes that are not continuations are the characters.
4793            let characters = value.iter().filter(|byte| (**byte as i8) >= -0x40).count();
4794            lens.push(u32::try_from(characters).unwrap_or(u32::MAX));
4795            start = end;
4796        }
4797        Ok(slot.get_or_init(|| lens.into_boxed_slice()))
4798    }
4799
4800    /// Reads and decodes one block of the payload, without deciding who keeps it.
4801    ///
4802    /// [`Self::payload_block`] keeps it forever, which is what a point read wants and what a walk
4803    /// of the whole dictionary must not do. Both call this and they differ in nothing else.
4804    fn decode_block(&self, block: usize) -> Result<Vec<u8>> {
4805        let len = self.lengths[block];
4806        let mut stored = vec![
4807            0;
4808            usize::try_from(len).map_err(|_| invalid(
4809                "global dictionary block does not fit in memory"
4810            ))?
4811        ];
4812        read_at(&self.file, self.starts[block], &mut stored)?;
4813        if checksum(&stored) != self.hashes[block] {
4814            return Err(invalid("global dictionary payload checksum differs"));
4815        }
4816        let first = block * TEXT_PAYLOAD_VALUES;
4817        let last = (first + TEXT_PAYLOAD_VALUES).min(self.values);
4818        let want = self.end_within(last - 1)? as usize;
4819        let values = string::decode_flat(&stored)?;
4820        if values.len() != last - first {
4821            return Err(invalid("global dictionary block holds the wrong value count"));
4822        }
4823        let bytes = values.into_bytes();
4824        if bytes.len() != want {
4825            return Err(invalid("global dictionary block decodes to the wrong length"));
4826        }
4827        Ok(bytes)
4828    }
4829
4830    /// The block holding a value that a read hands over on loan, kept or decoded for the call.
4831    ///
4832    /// A block something already kept is read where it is. One nothing kept is kept the second
4833    /// time a loaned read decodes it while the column is holding less than [`Self::keep_budget`],
4834    /// and decoded into `decoded` and dropped with it otherwise, which is the policy
4835    /// [`TextSource::sweep`] explains. `scattered` is a read by code rather than in order, which
4836    /// stops dropping once it has dropped a column's worth of blocks, for the reason
4837    /// [`Self::visit_dropped`] gives.
4838    fn loaned_block<'a>(
4839        &'a self,
4840        block: usize,
4841        decoded: &'a mut Vec<u8>,
4842        scattered: bool,
4843    ) -> Result<&'a [u8]> {
4844        let kept = self.blocks.get(block).and_then(OnceLock::get);
4845        if let Some(Ok(kept)) = kept {
4846            return Ok(kept);
4847        }
4848        let again = kept.is_none()
4849            && self.swept.get(block).is_some_and(|swept| swept.swap(true, Atomic::Relaxed));
4850        let keep = again
4851            && (self.payload_kept.load(Atomic::Relaxed) < self.keep_budget
4852                || (scattered && self.visit_dropped.load(Atomic::Relaxed) >= self.blocks.len()));
4853        if keep {
4854            let kept = self
4855                .payload_block(block)?
4856                .ok_or_else(|| invalid("global dictionary block is past the payload"))?;
4857            self.payload_kept.fetch_add(kept.len(), Atomic::Relaxed);
4858            return Ok(kept);
4859        }
4860        *decoded = self.decode_block(block)?;
4861        if scattered && again {
4862            self.visit_dropped.fetch_add(1, Atomic::Relaxed);
4863        }
4864        Ok(decoded)
4865    }
4866
4867    /// How many single offset reads make [`Self::value_ends`] worth building.
4868    ///
4869    /// As many reads as the dictionary has values. Building the table costs about thirty
4870    /// instructions a value once the fresh pages it lands in are counted, and a read out of it saves
4871    /// about thirty five, so it repays itself after roughly one read per value. The reads so far are
4872    /// the only guess there is at the reads to come, and waiting until they match the size of the
4873    /// dictionary is betting that a column read that much will be read that much again.
4874    ///
4875    /// A sixteenth was the first answer, from counting the unpacking alone at three instructions a
4876    /// value. ClickBench 38 showed what that missed: it reads about twenty thousand titles a
4877    /// statement out of a dictionary of three hundred and fifty thousand, crossed a sixteenth in its
4878    /// second statement and was two percent slower for a table it did not read enough to repay. A
4879    /// scan asking for the length of every row crosses it part way through its first statement on
4880    /// ClickBench, where a string column has about two rows for every value, and a filter that keeps
4881    /// a few thousand rows never does. The floor is there
4882    /// because a short dictionary would otherwise build a table for a handful of reads.
4883    fn ends_worth_unpacking(&self) -> usize {
4884        self.values.max(TEXT_PAYLOAD_VALUES)
4885    }
4886
4887    /// The unpacked ends, if they are built or if this read is the one that makes them worth it.
4888    fn value_ends(&self) -> Option<&[u32]> {
4889        if let Some(built) = self.value_ends.get() {
4890            return built.as_deref();
4891        }
4892        if self.ends_asked.fetch_add(1, Atomic::Relaxed) < self.ends_worth_unpacking() {
4893            return None;
4894        }
4895        self.value_ends.get_or_init(|| self.unpack_ends()).as_deref()
4896    }
4897
4898    /// Every end of the column, a run at a time.
4899    ///
4900    /// `None` rather than an error on anything wrong, because this is a cache in front of a reader
4901    /// that answers the same question. A column whose offsets are short or whose ends do not fit in
4902    /// four bytes gets no table and the same error it would have got, from the read that wanted it.
4903    fn unpack_ends(&self) -> Option<Vec<u32>> {
4904        let mut ends = vec![0u32; self.values];
4905        for (run, into) in ends.chunks_mut(TEXT_OFFSET_RUN).enumerate() {
4906            let bytes = self.packed().get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)?;
4907            bitpack::unpack_tail_into(bytes, self.offset_bits, into, |bits| {
4908                u32::try_from(bits).unwrap_or(u32::MAX)
4909            })
4910            .ok()?;
4911        }
4912        // An end that did not fit was stored as the sentinel, and a real one cannot reach it because
4913        // a payload block is far smaller than four gigabytes. So the column keeps the packed reader.
4914        if ends.contains(&u32::MAX) { None } else { Some(ends) }
4915    }
4916
4917    /// The packed offsets, which is the index past its header.
4918    fn packed(&self) -> &[u8] {
4919        self.offsets.get(DICTIONARY_HEADER..).unwrap_or_default()
4920    }
4921
4922    /// Where the value at `index` ends inside its payload block.
4923    fn end_within(&self, index: usize) -> Result<u32> {
4924        if let Some(ends) = self.value_ends() {
4925            return ends
4926                .get(index)
4927                .copied()
4928                .ok_or_else(|| invalid("global dictionary offsets are short"));
4929        }
4930        let run = index / TEXT_OFFSET_RUN;
4931        let bytes = self
4932            .packed()
4933            .get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)
4934            .ok_or_else(|| invalid("global dictionary offsets are short"))?;
4935        let end = bitpack::tail_at(bytes, self.offset_bits, index % TEXT_OFFSET_RUN)
4936            .map_err(|_| invalid("global dictionary offsets are short"))?;
4937        u32::try_from(end).map_err(|_| invalid("global dictionary offset is past the payload"))
4938    }
4939
4940    /// Where every value in `first..last` ends inside its payload block, in one pass over the runs.
4941    ///
4942    /// [`Self::end_within`] answers for one value and pays for it twice over: it shifts a window to
4943    /// the bit the value starts at, and the copy that fills that window is a length the compiler does
4944    /// not know, so it is a call to `memcpy` rather than a load. A sweep asked for two of those per
4945    /// value, one for the end and one for the start that is the end before it, and on the ClickBench
4946    /// `URL` dictionary of eighteen million that was most of the half second a `LIKE` over it took.
4947    ///
4948    /// [`bitpack::unpack_tail_into`] walks the run instead, which makes the window a fixed width and
4949    /// so an unaligned load, and reads the bit position off a counter. A run is five hundred and
4950    /// twelve values and a block is two of them, so a block of a thousand and twenty four values
4951    /// costs two calls here and nothing per value.
4952    ///
4953    /// The answer is written straight into the result. A run that is wanted from its first value,
4954    /// which is every run but the one the sweep starts in, unpacks into its own window of the result
4955    /// and is never copied. Only a run joined part way through needs the scratch buffer, and there is
4956    /// at most one of those per sweep, so the buffer is allocated the first time one turns up.
4957    fn ends_within(&self, first: usize, last: usize) -> Result<Vec<u64>> {
4958        let mut ends = vec![0u64; last.saturating_sub(first)];
4959        let mut scratch = Vec::new();
4960        let mut at = first;
4961        while at < last {
4962            let run = at / TEXT_OFFSET_RUN;
4963            let stop = ((run + 1) * TEXT_OFFSET_RUN).min(last);
4964            let held = self.values.saturating_sub(run * TEXT_OFFSET_RUN).min(TEXT_OFFSET_RUN);
4965            let bytes = self
4966                .packed()
4967                .get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)
4968                .ok_or_else(|| invalid("global dictionary offsets are short"))?;
4969            let from = at % TEXT_OFFSET_RUN;
4970            let upto = stop - run * TEXT_OFFSET_RUN;
4971            if upto > held || bytes.len() < bitpack::tail_len(held, self.offset_bits) {
4972                return Err(invalid("global dictionary offsets are short"));
4973            }
4974            let into = &mut ends[at - first..stop - first];
4975            if from == 0 {
4976                bitpack::unpack_tail_into(bytes, self.offset_bits, into, |bits| bits)
4977                    .map_err(|_| invalid("global dictionary offsets are short"))?;
4978            } else {
4979                scratch.resize(held, 0);
4980                bitpack::unpack_tail_into(bytes, self.offset_bits, &mut scratch, |bits| bits)
4981                    .map_err(|_| invalid("global dictionary offsets are short"))?;
4982                into.copy_from_slice(&scratch[from..upto]);
4983            }
4984            at = stop;
4985        }
4986        Ok(ends)
4987    }
4988
4989    /// Where the value at `index` starts inside its payload block, which is where the value before
4990    /// it ended unless it is the first of the block.
4991    fn start_within(&self, index: usize) -> Result<u32> {
4992        if index % TEXT_PAYLOAD_VALUES == 0 { Ok(0) } else { self.end_within(index - 1) }
4993    }
4994
4995    /// Where the value at `index` starts and ends inside its payload block.
4996    ///
4997    /// The two offsets sit next to each other in the same run unless the value opens one, and a run
4998    /// of seventeen bit offsets, which is what a block of a thousand strings needs, puts a pair of
4999    /// them inside one eight byte load. So the common case reads the packed bytes once rather than
5000    /// twice and does the bounds arithmetic once. This is asked once per string a text column hands
5001    /// out, and on ClickBench 27 the two reads together were a quarter of the query.
5002    fn span_within(&self, index: usize) -> Result<(u32, u32)> {
5003        if let Some(ends) = self.value_ends() {
5004            let end =
5005                *ends.get(index).ok_or_else(|| invalid("global dictionary offsets are short"))?;
5006            // The value before it in the same block, and zero where there is no value before it.
5007            // `index` is inside the table, so the one under it is too.
5008            let start = if index % TEXT_PAYLOAD_VALUES == 0 { 0 } else { ends[index - 1] };
5009            if start > end {
5010                return Err(invalid("global dictionary value ends before it starts"));
5011            }
5012            return Ok((start, end));
5013        }
5014        let within = index % TEXT_OFFSET_RUN;
5015        let (start, end) = if within == 0 {
5016            (self.start_within(index)?, self.end_within(index)?)
5017        } else {
5018            let run = index / TEXT_OFFSET_RUN;
5019            let bytes = self
5020                .packed()
5021                .get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)
5022                .ok_or_else(|| invalid("global dictionary offsets are short"))?;
5023            let (start, end) = bitpack::tail_pair(bytes, self.offset_bits, within)
5024                .map_err(|_| invalid("global dictionary offsets are short"))?;
5025            let ends = u32::try_from(end)
5026                .map_err(|_| invalid("global dictionary offset is past the payload"))?;
5027            let starts = u32::try_from(start)
5028                .map_err(|_| invalid("global dictionary offset is past the payload"))?;
5029            (starts, ends)
5030        };
5031        if start > end {
5032            return Err(invalid("global dictionary value ends before it starts"));
5033        }
5034        Ok((start, end))
5035    }
5036
5037    /// The block of the sorted order that holds `rank`, and where in it that rank sits.
5038    ///
5039    /// The block is read from the file and checked against the hash the index carries for it the
5040    /// first time anything asks, and kept after that, the same way a payload block is. A search
5041    /// makes about as many probes as the order has bits, so the whole search reads a handful of
5042    /// these and never the rest.
5043    fn rank_parts(&self, rank: usize) -> Result<(&[u8], usize)> {
5044        let slot = self
5045            .rank_blocks
5046            .get(rank / TEXT_RANK_BLOCK)
5047            .ok_or_else(|| invalid("global dictionary rank is past the order"))?;
5048        let block = slot
5049            .get_or_init(|| {
5050                let mut bytes = Vec::new();
5051                self.read_rank_block(rank / TEXT_RANK_BLOCK, &mut bytes)?;
5052                Ok(bytes)
5053            })
5054            .as_ref()
5055            .map_err(Clone::clone)?;
5056        Ok((block.as_slice(), rank % TEXT_RANK_BLOCK))
5057    }
5058
5059    /// Reads block `which` of the sorted order into `bytes`, checked against the hash the index
5060    /// carries for it.
5061    fn read_rank_block(&self, which: usize, bytes: &mut Vec<u8>) -> Result<()> {
5062        let start = if which == 0 { 0 } else { self.rank_ends[which - 1] };
5063        let end = self.rank_ends[which];
5064        bytes.clear();
5065        bytes.resize((end - start) as usize, 0);
5066        read_at(&self.file, self.rank_at + start, bytes)?;
5067        let expected = self
5068            .rank_hashes
5069            .get(which)
5070            .ok_or_else(|| invalid("global dictionary rank block has no checksum"))?;
5071        if checksum(bytes) != *expected {
5072            return Err(invalid("global dictionary rank checksum differs"));
5073        }
5074        Ok(())
5075    }
5076
5077    /// The first eight bytes of the value at `rank`, as the integer a comparison reads.
5078    fn head_at(&self, rank: usize) -> Result<u64> {
5079        let (block, within) = self.rank_parts(rank)?;
5080        let (base, width, packed) = rank_heads(block)?;
5081        let above = bitpack::tail_at(packed, width, within)
5082            .map_err(|_| invalid("global dictionary rank block is short of heads"))?;
5083        Ok(base.wrapping_add(above))
5084    }
5085
5086    /// The packed codes of one rank block, which follow the heads on the next byte boundary.
5087    fn rank_codes<'block>(&self, block: &'block [u8], count: usize) -> Result<&'block [u8]> {
5088        let (_, width, packed) = rank_heads(block)?;
5089        packed
5090            .get(bitpack::tail_len(count, width)..)
5091            .ok_or_else(|| invalid("global dictionary rank block is short of codes"))
5092    }
5093
5094    /// How many entries the block holding `rank` has, which is a full block except at the end.
5095    fn rank_block_len(&self, rank: usize) -> usize {
5096        let first = rank / TEXT_RANK_BLOCK * TEXT_RANK_BLOCK;
5097        TEXT_RANK_BLOCK.min(self.ranks - first)
5098    }
5099}
5100
5101/// The base, the width and the packed bytes of one rank block's heads.
5102fn rank_heads(block: &[u8]) -> Result<(u64, usize, &[u8])> {
5103    let header = block
5104        .get(..RANK_BLOCK_HEADER)
5105        .ok_or_else(|| invalid("global dictionary rank block is short"))?;
5106    let base = u64::from_le_bytes(header[..8].try_into().expect("eight bytes"));
5107    let width = header[8] as usize;
5108    if width > 64 {
5109        return Err(invalid("global dictionary rank block packs heads past a word"));
5110    }
5111    Ok((base, width, &block[RANK_BLOCK_HEADER..]))
5112}
5113
5114/// Bits one offset of a dictionary takes, which is what its widest payload block spans.
5115///
5116/// One width for the whole column rather than one a block. A block is 1,024 values of the same
5117/// column, so the blocks of a column are within a factor of two of each other on every ClickBench
5118/// string column, and a width a block would save a fraction of a bit and cost a byte a block plus
5119/// the arithmetic that finds where a block starts.
5120fn offset_width(ends: &[u32]) -> usize {
5121    // The ends are already relative to the block the value is in, so the last end of a block is that
5122    // block's total and the largest end anywhere is the widest block. There is no subtraction left
5123    // to do and no need to walk the blocks to find where one starts.
5124    let span = ends.iter().copied().max().unwrap_or(0);
5125    (u32::BITS - span.leading_zeros()) as usize
5126}
5127
5128/// How many bytes `values` offsets take at `bits`, which is what the reader has to know before it
5129/// has read any of them.
5130fn offset_bytes(values: usize, bits: usize) -> usize {
5131    let full = values / TEXT_OFFSET_RUN;
5132    let rest = values % TEXT_OFFSET_RUN;
5133    full * TEXT_OFFSET_RUN / 8 * bits + bitpack::tail_len(rest, bits)
5134}
5135
5136/// The end of every value within its payload block, packed a run at a time.
5137/// A run never straddles a block, because [`TEXT_OFFSET_RUN`] divides [`TEXT_PAYLOAD_VALUES`], which
5138/// is what lets this be a walk of the ends rather than arithmetic against a per block base.
5139fn encode_offsets(ends: &[u32], bits: usize, out: &mut Vec<u8>) -> Result<()> {
5140    let mut run = Vec::with_capacity(TEXT_OFFSET_RUN);
5141    for chunk in ends.chunks(TEXT_OFFSET_RUN) {
5142        run.clear();
5143        run.extend(chunk.iter().map(|&end| u64::from(end)));
5144        bitpack::pack_tail(&run, bits, out)
5145            .map_err(|_| invalid("global dictionary offsets do not pack"))?;
5146    }
5147    Ok(())
5148}
5149
5150/// How many bits a code of a dictionary of `values` entries takes.
5151fn code_width(values: usize) -> usize {
5152    match u64::try_from(values).unwrap_or(u64::MAX) {
5153        0 | 1 => 0,
5154        last => (u64::BITS - (last - 1).leading_zeros()) as usize,
5155    }
5156}
5157
5158impl TextSource for NativeText {
5159    fn len(&self) -> usize {
5160        self.values
5161    }
5162
5163    fn might_contain(&self, first: usize, literal: &[u8]) -> Result<bool> {
5164        let Some(grams) = &self.grams else { return Ok(true) };
5165        if literal.len() < 4 || first >= self.values {
5166            return Ok(true);
5167        }
5168        let verdict = grams.verdicts(&self.file, literal)?;
5169        Ok(verdict.get(first / TEXT_PAYLOAD_VALUES).copied().unwrap_or(true))
5170    }
5171
5172    fn bytes_at(&self, index: usize) -> Result<Option<&[u8]>> {
5173        if index >= self.values {
5174            return Ok(None);
5175        }
5176        let (start, end) = self.span_within(index)?;
5177        if start == end {
5178            return Ok(Some(&[]));
5179        }
5180        // A block holds a fixed number of values rather than a fixed number of bytes, so the value
5181        // is in one block and the offsets already say where in it.
5182        let block = index / TEXT_PAYLOAD_VALUES;
5183        let Some(bytes) = self.payload_block(block)? else { return Ok(None) };
5184        Ok(bytes.get(start as usize..end as usize))
5185    }
5186
5187    fn bytes_len_at(&self, index: usize) -> Result<Option<usize>> {
5188        if index >= self.values {
5189            return Ok(None);
5190        }
5191        let (start, end) = self.span_within(index)?;
5192        Ok(Some((end - start) as usize))
5193    }
5194
5195    /// Every length out of the unpacked ends in one loop, which is the point of having them.
5196    ///
5197    /// The whole run of positions counts towards [`Self::ends_worth_unpacking`] at once, because a
5198    /// caller asking for a vector of lengths has said how many it wants, and a vector of them is
5199    /// usually enough on its own. Until the table is worth building this is the row at a time read,
5200    /// the same as the default.
5201    fn bytes_lens_at(&self, indices: &[u32], into: &mut Vec<i64>) -> Result<()> {
5202        self.ends_asked.fetch_add(indices.len(), Atomic::Relaxed);
5203        into.reserve(indices.len());
5204        let Some(ends) = self.value_ends() else {
5205            for &index in indices {
5206                into.push(
5207                    self.bytes_len_at(index as usize)?
5208                        .map_or(0, |len| i64::try_from(len).unwrap_or(i64::MAX)),
5209                );
5210            }
5211            return Ok(());
5212        };
5213        if let Some(lens) = self.value_lens.get_or_init(|| lengths_of(ends)) {
5214            lens.extend_at(indices, into);
5215            return Ok(());
5216        }
5217        for &index in indices {
5218            let index = index as usize;
5219            // Past the end is no value and so no length, which is what a row at a time read says.
5220            let Some(&end) = ends.get(index) else {
5221                into.push(0);
5222                continue;
5223            };
5224            let start = if index % TEXT_PAYLOAD_VALUES == 0 { 0 } else { ends[index - 1] };
5225            if start > end {
5226                return Err(invalid("global dictionary value ends before it starts"));
5227            }
5228            into.push(i64::from(end - start));
5229        }
5230        Ok(())
5231    }
5232
5233    /// Every length in characters out of the counts kept a block at a time, which is what keeps a
5234    /// scan of `length` from holding the column decoded. See [`NativeText::char_lens`].
5235    fn chars_lens_at(&self, indices: &[u32], into: &mut Vec<i64>) -> Result<()> {
5236        into.reserve(indices.len());
5237        for &index in indices {
5238            let index = index as usize;
5239            // Past the end is no value and so no length, which is what a row at a time read says.
5240            if index >= self.values {
5241                into.push(0);
5242                continue;
5243            }
5244            let lens = self.block_chars(index / TEXT_PAYLOAD_VALUES)?;
5245            let len = lens
5246                .get(index % TEXT_PAYLOAD_VALUES)
5247                .ok_or_else(|| invalid("global dictionary block holds the wrong value count"))?;
5248            into.push(i64::from(*len));
5249        }
5250        Ok(())
5251    }
5252
5253    /// The rest of the block holding `first`, decoded into a buffer that may die with the call.
5254    ///
5255    /// A block is the unit this format decodes, so a walk that wants every value is going to decode
5256    /// every block whatever it does. The question is whether it keeps them, and both answers are
5257    /// wrong on their own. [`Self::payload_block`] keeps every block it is asked for, so a reader
5258    /// that walked the whole dictionary through `bytes_at` ended up holding the whole dictionary
5259    /// decoded, 4.2 GB on ClickBench `URL`. Keeping none of them makes the next statement asking
5260    /// the same question decode all of it again, which on the same column at a million rows is a
5261    /// `LIKE` going from 2.7 ms to 16.2 ms.
5262    ///
5263    /// So a sweep keeps what it decodes for the second time while the column is under
5264    /// [`TEXT_KEEP_BUDGET`] and drops it after that. A block already in hand is used where it is there and costs nothing either way.
5265    fn sweep(
5266        &self,
5267        first: usize,
5268        limit: usize,
5269        body: &mut dyn FnMut(usize, &[u8]) -> Result<()>,
5270    ) -> Result<usize> {
5271        let limit = limit.min(self.values);
5272        if first >= limit {
5273            return Ok(first);
5274        }
5275        let block = first / TEXT_PAYLOAD_VALUES;
5276        let last = ((block + 1) * TEXT_PAYLOAD_VALUES).min(limit);
5277        let mut decoded = Vec::new();
5278        let bytes = self.loaned_block(block, &mut decoded, false)?;
5279        let ends = self.ends_within(first, last)?;
5280        if ends.len() != last - first {
5281            return Err(invalid("global dictionary offsets are short"));
5282        }
5283        let mut start = u64::from(self.start_within(first)?);
5284        // row at a time: the caller is handed one value after another, and what it does with one is
5285        // its own business, so there is no shape here for anything but a walk.
5286        for (index, &end) in (first..last).zip(&ends) {
5287            let value = usize::try_from(start)
5288                .ok()
5289                .zip(usize::try_from(end).ok())
5290                .and_then(|(from, to)| bytes.get(from..to))
5291                .ok_or_else(|| invalid("global dictionary value is past its block"))?;
5292            body(index, value)?;
5293            start = end;
5294        }
5295        Ok(last)
5296    }
5297
5298    /// The values at `indices` a block at a time, each block read once for the call.
5299    ///
5300    /// The positions are put in code order first, because the codes of a vector are in row order
5301    /// and land all over the dictionary, and read in that order each block a vector touches would
5302    /// be looked up once for every row in it. Whether a block is kept is
5303    /// [`NativeText::loaned_block`]'s decision, which keeps at most the budget of this column
5304    /// until the reads have shown they come back to the same blocks too often for dropping them to
5305    /// be cheap.
5306    fn visit_at(
5307        &self,
5308        indices: &[u32],
5309        body: &mut dyn FnMut(usize, &[u8]) -> Result<()>,
5310    ) -> Result<()> {
5311        let mut order = (0..indices.len()).collect::<Vec<_>>();
5312        order.sort_unstable_by_key(|&at| indices[at]);
5313        let block_of = |at: usize| {
5314            let index = indices[at] as usize;
5315            (index < self.values).then_some(index / TEXT_PAYLOAD_VALUES)
5316        };
5317        let mut decoded = Vec::new();
5318        let mut run = 0;
5319        while run < order.len() {
5320            let Some(block) = block_of(order[run]) else {
5321                // Past the end is no value, and every position after this one is past it too.
5322                for &at in &order[run..] {
5323                    body(at, &[])?;
5324                }
5325                break;
5326            };
5327            let upto = run + order[run..].partition_point(|&at| block_of(at) == Some(block));
5328            let bytes = self.loaned_block(block, &mut decoded, true)?;
5329            for &at in &order[run..upto] {
5330                let (start, end) = self.span_within(indices[at] as usize)?;
5331                let value = bytes
5332                    .get(start as usize..end as usize)
5333                    .ok_or_else(|| invalid("global dictionary value is past its block"))?;
5334                body(at, value)?;
5335            }
5336            run = upto;
5337        }
5338        Ok(())
5339    }
5340
5341    /// Each block the indices land in, decoded once and dropped, or read where it is already kept.
5342    ///
5343    /// Never kept, unlike [`Self::sweep`] under its budget, because a scattered read is a one off:
5344    /// a synopsis turned into values is turned once and remembered by the reader as values, a few
5345    /// kilobytes, where the blocks it went through are megabytes nobody asks for again.
5346    fn visit(
5347        &self,
5348        indices: &[usize],
5349        body: &mut dyn FnMut(usize, &[u8]) -> Result<()>,
5350    ) -> Result<()> {
5351        let mut at = 0;
5352        while at < indices.len() {
5353            let block = indices[at] / TEXT_PAYLOAD_VALUES;
5354            let upto =
5355                at + indices[at..].partition_point(|&index| index / TEXT_PAYLOAD_VALUES == block);
5356            let wanted = &indices[at..upto];
5357            if wanted.iter().any(|&index| index >= self.values) {
5358                return Err(invalid("a visited value is past the global dictionary"));
5359            }
5360            let decoded;
5361            let bytes: &[u8] = match self.blocks.get(block).and_then(OnceLock::get) {
5362                Some(Ok(kept)) => kept,
5363                _ => {
5364                    decoded = self.decode_block(block)?;
5365                    &decoded
5366                }
5367            };
5368            for (offset, &index) in wanted.iter().enumerate() {
5369                let (start, end) = self.span_within(index)?;
5370                let value = bytes
5371                    .get(start as usize..end as usize)
5372                    .ok_or_else(|| invalid("global dictionary value is past its block"))?;
5373                body(at + offset, value)?;
5374            }
5375            at = upto;
5376        }
5377        Ok(())
5378    }
5379
5380    fn ranks(&self) -> Option<usize> {
5381        (self.ranks > 0).then_some(self.ranks)
5382    }
5383
5384    /// The boundary for `wanted`, out of [`Self::searched`] where it is there and put there where
5385    /// it is not.
5386    ///
5387    /// The lock is held over the search rather than dropped and taken again, so that two threads
5388    /// asking for the same value at the same time do the work once between them. That is the shape
5389    /// the scan actually arrives in: sixteen instances of a top N, all reading the same column, all
5390    /// improving their bound over the same early chunks.
5391    fn below(&self, ranks: usize, wanted: &[u8]) -> Result<(usize, bool)> {
5392        let mut memo = self.searched.lock().map_err(|_| invalid("a poisoned dictionary search"))?;
5393        if let Some(&answer) = memo.get(wanted) {
5394            return Ok(answer);
5395        }
5396        let answer = search_below(self, ranks, wanted)?;
5397        if memo.len() >= TEXT_SEARCH_MEMO {
5398            memo.clear();
5399        }
5400        memo.insert(wanted.to_vec(), answer);
5401        Ok(answer)
5402    }
5403
5404    fn compare_rank(&self, rank: usize, wanted: &[u8]) -> Result<Ordering> {
5405        // The head settles the probe unless the two values start with the same eight bytes, and
5406        // only then is a value read. On a column of URLs that is the difference between a search
5407        // that touches one block of the payload and a search that touches nineteen of them.
5408        let settled = self.head_at(rank)?.cmp(&head(wanted));
5409        if settled != Ordering::Equal {
5410            return Ok(settled);
5411        }
5412        let code = self.code_at_rank(rank)?;
5413        let bytes = self
5414            .bytes_at(code as usize)?
5415            .ok_or_else(|| invalid("global dictionary order names a code it does not have"))?;
5416        Ok(bytes.cmp(wanted))
5417    }
5418
5419    fn code_at_rank(&self, rank: usize) -> Result<u32> {
5420        let (block, within) = self.rank_parts(rank)?;
5421        let codes = self.rank_codes(block, self.rank_block_len(rank))?;
5422        let code = bitpack::tail_at(codes, self.code_bits, within)
5423            .map_err(|_| invalid("global dictionary rank block is short of codes"))?;
5424        let code = u32::try_from(code)
5425            .map_err(|_| invalid("global dictionary order names a code it does not have"))?;
5426        if code as usize >= self.len() {
5427            return Err(invalid("global dictionary order names a code it does not have"));
5428        }
5429        Ok(code)
5430    }
5431
5432    fn code_ranks(&self) -> Option<&[u32]> {
5433        // The order is a permutation of the positions, so inverting it needs every position to be
5434        // named exactly once. Anything else and the slice would have holes, and a caller indexing
5435        // it by a code would read a rank that belongs to nothing.
5436        if self.ranks == 0 || self.ranks != self.len() {
5437            return None;
5438        }
5439        self.code_ranks
5440            .get_or_init(|| {
5441                let mut ranks = vec![u32::MAX; self.ranks];
5442                // A block at a time rather than a rank at a time, because reading it per rank pays
5443                // for the bounds check, the division and the lock on every one of them.
5444                //
5445                // A block nothing has read yet is read into one buffer that is reused, rather than
5446                // through `rank_parts`, which would keep every block of the order once this is
5447                // done with it. The inverse is all anything wants after this, and on the `Referer`
5448                // column of the ClickBench file the blocks are tens of megabytes held for nothing.
5449                let mut scratch = Vec::new();
5450                let mut codes = vec![0u64; TEXT_RANK_BLOCK];
5451                for first in (0..self.ranks).step_by(TEXT_RANK_BLOCK) {
5452                    let which = first / TEXT_RANK_BLOCK;
5453                    let block = match self.rank_blocks.get(which)?.get() {
5454                        Some(kept) => kept.as_ref().ok()?.as_slice(),
5455                        None => {
5456                            self.read_rank_block(which, &mut scratch).ok()?;
5457                            scratch.as_slice()
5458                        }
5459                    };
5460                    let count = self.rank_block_len(first);
5461                    let packed = self.rank_codes(block, count).ok()?;
5462                    let codes = codes.get_mut(..count)?;
5463                    bitpack::unpack_tail_into(packed, self.code_bits, codes, |bits| bits).ok()?;
5464                    for (within, &code) in codes.iter().enumerate() {
5465                        let code = usize::try_from(code).ok()?;
5466                        *ranks.get_mut(code)? = u32::try_from(first + within).ok()?;
5467                    }
5468                }
5469                if ranks.contains(&u32::MAX) {
5470                    return None;
5471                }
5472                Some(ranks)
5473            })
5474            .as_deref()
5475    }
5476
5477    fn footprint(&self) -> usize {
5478        self.offsets.capacity()
5479            + self
5480                .value_ends
5481                .get()
5482                .and_then(Option::as_ref)
5483                .map_or(0, |ends| ends.capacity() * size_of::<u32>())
5484            + self.value_lens.get().and_then(Option::as_ref).map_or(0, Lengths::footprint)
5485            + self
5486                .code_ranks
5487                .get()
5488                .and_then(Option::as_ref)
5489                .map_or(0, |ranks| ranks.capacity() * size_of::<u32>())
5490            + self.rank_hashes.capacity() * size_of::<u64>()
5491            + self.rank_ends.capacity() * size_of::<u64>()
5492            + self.rank_blocks.capacity() * size_of::<OnceLock<Result<Vec<u8>>>>()
5493            + self
5494                .rank_blocks
5495                .iter()
5496                .filter_map(OnceLock::get)
5497                .filter_map(|result| result.as_ref().ok())
5498                .map(Vec::capacity)
5499                .sum::<usize>()
5500            + self.blocks.capacity() * size_of::<OnceLock<Result<Vec<u8>>>>()
5501            + self.char_lens.capacity() * size_of::<OnceLock<Box<[u32]>>>()
5502            + self
5503                .char_lens
5504                .iter()
5505                .filter_map(OnceLock::get)
5506                .map(|lens| lens.len() * size_of::<u32>())
5507                .sum::<usize>()
5508            + self.hashes.capacity() * size_of::<u64>()
5509            + self.starts.capacity() * size_of::<u64>()
5510            + self.lengths.capacity() * size_of::<u64>()
5511            + self.grams.as_ref().map_or(0, NativeGrams::footprint)
5512            + self
5513                .blocks
5514                .iter()
5515                .filter_map(OnceLock::get)
5516                .filter_map(|result| result.as_ref().ok())
5517                .map(Vec::capacity)
5518                .sum::<usize>()
5519    }
5520}
5521
5522/// Every table wide part number in order, with the stripe it belongs to.
5523fn places(table: &Table) -> Result<Vec<Place>> {
5524    let mut places = Vec::with_capacity(table.stripes.len().saturating_mul(STRIPE_PARTS));
5525    for (at, stripe) in table.stripes.iter().enumerate() {
5526        let index = u32::try_from(at).map_err(|_| invalid("too many stripes"))?;
5527        for (part, &rows) in stripe.parts.iter().enumerate() {
5528            places.push(Place {
5529                stripe: index,
5530                part: u32::try_from(part).map_err(|_| invalid("too many parts in a stripe"))?,
5531                rows,
5532            });
5533        }
5534    }
5535    Ok(places)
5536}
5537
5538/// Reads one column's section of a stripe's index page.
5539///
5540/// The section carries its own checksum, so a reader that wants one column out of a hundred and
5541/// five preads a few hundred bytes and still knows that what it got is what was written.
5542fn read_index<F: Positional + ?Sized>(
5543    file: &F,
5544    stripe: &Stripe,
5545    column: usize,
5546) -> Result<Vec<PartSpan>> {
5547    let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
5548    read_index_span(file, stripe.index, *page, stripe.parts.len(), column)
5549}
5550
5551fn read_index_span<F: Positional + ?Sized>(
5552    file: &F,
5553    index: Span,
5554    page: Span,
5555    parts: usize,
5556    column: usize,
5557) -> Result<Vec<PartSpan>> {
5558    let section = index_section(parts)?;
5559    let at = column.checked_mul(section).ok_or_else(|| invalid("index page offset overflow"))?;
5560    let end = at.checked_add(section).ok_or_else(|| invalid("index page offset overflow"))?;
5561    if end > index.length as usize {
5562        return Err(invalid("index page is shorter than its columns"));
5563    }
5564    let mut bytes = vec![0; section];
5565    let offset =
5566        index.offset.checked_add(at as u64).ok_or_else(|| invalid("index page offset overflow"))?;
5567    read_at(file, offset, &mut bytes)?;
5568    let entries = section - size_of::<u64>();
5569    let stored = u64::from_le_bytes(bytes[entries..].try_into().expect("eight bytes"));
5570    if checksum(&bytes[..entries]) != stored {
5571        // With where it was read from, because the two ways this fires look identical from the
5572        // message alone: a file somebody damaged, and a file we wrote to the wrong offset.
5573        return Err(invalid(&format!(
5574            "index page section checksum differs, column {column} of {parts} parts at {offset}, \
5575             wanted {stored:016x} and got {:016x}",
5576            checksum(&bytes[..entries]),
5577        )));
5578    }
5579    let mut spans = Vec::with_capacity(parts);
5580    let mut start = 0_usize;
5581    for part in 0..parts {
5582        let at = part * INDEX_ENTRY;
5583        let length = u32::from_le_bytes(bytes[at..at + 4].try_into().expect("four bytes")) as usize;
5584        let hash = u64::from_le_bytes(bytes[at + 4..at + 12].try_into().expect("eight bytes"));
5585        spans.push(PartSpan { start, length, hash });
5586        start = start.checked_add(length).ok_or_else(|| invalid("column page length overflow"))?;
5587    }
5588    if start != page.length as usize {
5589        return Err(invalid("column page length differs from its index"));
5590    }
5591    Ok(spans)
5592}
5593
5594/// One part's bytes out of a whole column page.
5595fn part_bytes(page: &[u8], span: PartSpan) -> Result<&[u8]> {
5596    let end = span.start.checked_add(span.length).ok_or_else(|| invalid("part range overflow"))?;
5597    page.get(span.start..end).ok_or_else(|| invalid("part exceeds its column page"))
5598}
5599
5600/// Puts one stripe of one column in the cache, and hands back the page for the pool to count when
5601/// it is a page the column did not already hold.
5602///
5603/// The index goes in its own slot and stays. Only the page is under the budget, and the pool is
5604/// what enforces it, once the caller has let go of the column's lock.
5605fn remember(cached: &mut Cached, held: &CachedColumn) -> Option<(usize, Arc<AtomicBool>)> {
5606    if let Some(slot) = cached.index.get_mut(held.stripe) {
5607        if slot.is_none() {
5608            *slot = Some(Arc::clone(&held.index));
5609        }
5610    }
5611    let page = held.page.clone()?;
5612    let slot = cached.pages.get_mut(held.stripe)?;
5613    if slot.is_some() {
5614        return None;
5615    }
5616    let bytes = page.bytes.len();
5617    // Set, so that the page a worker has just paid to read is not the one the pass it pays for
5618    // lets go of before the worker has read a part out of it.
5619    let used = Arc::new(AtomicBool::new(true));
5620    *slot = Some(Resident { page, used: Arc::clone(&used) });
5621    Some((bytes, used))
5622}
5623
5624/// Every table a native file holds, without the directory of any of them.
5625///
5626/// This is what opening a database reads. It is the small level of the directory, so the cost is
5627/// proportional to how many tables there are rather than to how much data they hold, and a session
5628/// that touches two tables of eight decodes two table directories.
5629///
5630/// The file handle is shared with every reader this hands out. Eight tables in one file is one open
5631/// file descriptor, not eight, which is the other thing one file buys over a file per table.
5632#[derive(Debug, Clone)]
5633pub struct Catalog {
5634    file: Arc<File>,
5635    size: u64,
5636    entries: Arc<Vec<Entry>>,
5637    /// The views the file holds, whole, since a view has no second level to read later.
5638    views: Arc<Vec<ViewEntry>>,
5639    opening: Opening,
5640    /// Where every reader this hands out counts its pages.
5641    pool: PagePool,
5642}
5643
5644/// Signed integer sums and non-null counts for selected columns, plus total table rows.
5645#[derive(Debug, Clone, PartialEq, Eq)]
5646pub struct CertifiedSums {
5647    pub columns: Vec<(i128, u64)>,
5648    pub rows: u64,
5649}
5650
5651/// Exact ends of an integer or date column, including a certified all-null column.
5652#[derive(Debug, Clone, Copy, PartialEq, Eq)]
5653pub enum IntegerExtremes {
5654    Null,
5655    Values { low: i128, high: i128 },
5656}
5657
5658/// A complete numeric value-to-row-count synopsis; `None` represents SQL NULL.
5659pub type NumericFrequencies = Vec<(Option<i128>, u64)>;
5660
5661impl Catalog {
5662    /// Reads the highest valid catalog slot and nothing under it.
5663    ///
5664    /// The readers it hands out keep pages in a pool of their own with no budget, so each column
5665    /// holds its floor of four stripes and no more. A database opens with [`Catalog::open_in`].
5666    ///
5667    /// # Errors
5668    ///
5669    /// If the file has no valid committed catalog or a catalog pointer is out of bounds.
5670    pub fn open(path: impl AsRef<Path>) -> Result<Self> {
5671        Self::open_in(path, &PagePool::default())
5672    }
5673
5674    /// The same, with every reader it hands out keeping its pages in `pool`.
5675    ///
5676    /// # Errors
5677    ///
5678    /// If the file has no valid committed catalog or a catalog pointer is out of bounds.
5679    pub fn open_in(path: impl AsRef<Path>, pool: &PagePool) -> Result<Self> {
5680        let (file, size, _, bytes, opening) = slot_bytes(path)?;
5681        let (entries, views) = decode_catalog(&bytes, size)?;
5682        Ok(Self {
5683            file: Arc::new(file),
5684            size,
5685            entries: Arc::new(entries),
5686            views: Arc::new(views),
5687            opening,
5688            pool: pool.clone(),
5689        })
5690    }
5691
5692    /// The tables in the file, in the order they were written.
5693    pub fn names(&self) -> impl ExactSizeIterator<Item = &str> {
5694        self.entries.iter().map(|entry| entry.name.as_str())
5695    }
5696
5697    /// The same tables with how many rows each of them holds.
5698    ///
5699    /// The names alone answer which tables the file has, which is what a checkpoint needs to know.
5700    /// A load asks a second question: whether a table already in the file is really in the way of
5701    /// the one it wants to write. A table with no rows is not, because it has no pages the next
5702    /// generation would have to carry, so the count has to come out of the catalog beside the name.
5703    pub fn rows(&self) -> impl ExactSizeIterator<Item = (&str, usize)> {
5704        self.entries.iter().map(|entry| (entry.name.as_str(), entry.rows))
5705    }
5706
5707    /// The views in the file, in the order they were written.
5708    ///
5709    /// Whole, unlike [`Catalog::names`], which hands back names and makes the caller ask for a table
5710    /// by one. A view is a few strings and a column list and it was all read at open, so there is
5711    /// nothing left to go and fetch and no reason to make the caller ask twice.
5712    pub fn views(&self) -> impl ExactSizeIterator<Item = &ViewEntry> {
5713        self.views.iter()
5714    }
5715
5716    /// How many tables the file holds.
5717    #[must_use]
5718    pub fn len(&self) -> usize {
5719        self.entries.len()
5720    }
5721
5722    /// Whether the file holds no table at all, which is what [`Writer::empty`] writes and what a
5723    /// database somebody dropped the last table out of comes back as.
5724    #[must_use]
5725    pub fn is_empty(&self) -> bool {
5726        self.entries.is_empty()
5727    }
5728
5729    /// Opens one table by name, decoding its directory now.
5730    ///
5731    /// # Errors
5732    ///
5733    /// If there is no table by that name, or its directory is torn or points outside the file.
5734    pub fn table(&self, name: &str) -> Result<Reader> {
5735        let entry = self
5736            .entries
5737            .iter()
5738            .find(|entry| entry.name == name)
5739            .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
5740        // Checked and then decoded a window at a time, so that the directory's own bytes are never
5741        // all in memory beside the table they decode into. It is read twice, and the second read
5742        // comes out of the page cache the first one filled.
5743        let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
5744        if file_checksum(&self.file, offset, length)? != entry.directory.hash {
5745            return Err(invalid(&format!("the directory of table {name} does not checksum")));
5746        }
5747        let mut opening = self.opening;
5748        opening.reads += 1;
5749        opening.bytes += u64::from(entry.directory.length);
5750        Reader::build(
5751            Arc::clone(&self.file),
5752            self.size,
5753            read_directory(Cursor::over(&self.file, offset, length), self.size, Some(offset))?,
5754            u64::from(entry.directory.length),
5755            opening,
5756            self.pool.clone(),
5757        )
5758    }
5759
5760    /// Counts one signed integer column from its encoded parts without building metadata for
5761    /// unrelated columns. The counts are computed from row encodings when this is called.
5762    /// Nullable and non-cascade parts use the ordinary decoder for that part.
5763    ///
5764    /// # Errors
5765    ///
5766    /// If the directory, selected page index, checksum, or encoded integer is invalid.
5767    pub fn integer_tally(&self, name: &str, column: usize) -> Result<Option<Vec<(i64, u64)>>> {
5768        let mut counts = BTreeMap::<i64, u64>::new();
5769        let Some(()) = self.integer_fold(name, column, |value, count| {
5770            let held = counts.entry(value).or_default();
5771            *held = held.checked_add(count).ok_or_else(|| invalid("integer count overflow"))?;
5772            Ok(())
5773        })?
5774        else {
5775            return Ok(None);
5776        };
5777        Ok(Some(counts.into_iter().collect()))
5778    }
5779
5780    /// Visits a signed integer column's row values without building per-part or table-wide count
5781    /// maps. The caller combines the emitted counts for its query at runtime.
5782    ///
5783    /// # Errors
5784    ///
5785    /// If the selected file data is invalid or the callback rejects a count.
5786    pub fn integer_fold(
5787        &self,
5788        name: &str,
5789        column: usize,
5790        mut emit: impl FnMut(i64, u64) -> Result<()>,
5791    ) -> Result<Option<()>> {
5792        let entry = self
5793            .entries
5794            .iter()
5795            .find(|entry| entry.name == name)
5796            .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
5797        let field =
5798            entry.fields.get(column).ok_or_else(|| invalid("integer column index out of range"))?;
5799        if !signed_integer(&field.ty) {
5800            return Ok(None);
5801        }
5802        let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
5803        if file_checksum(&self.file, offset, length)? != entry.directory.hash {
5804            return Err(invalid(&format!("the directory of table {name} does not checksum")));
5805        }
5806        quick_integer_fold(
5807            &self.file,
5808            Cursor::over(&self.file, offset, length),
5809            entry,
5810            self.size,
5811            column,
5812            &mut emit,
5813        )?;
5814        Ok(Some(()))
5815    }
5816
5817    /// Counts non-null, nonzero values from generic column frequencies when complete. For an
5818    /// older file or a partial catalog synopsis, reads the validated native directory without
5819    /// building a reader for every stripe. Returns `None` when the bounded frequency synopsis
5820    /// cannot prove the count, so callers can use the ordinary query path.
5821    pub fn nonzero_count(&self, name: &str, column: usize) -> Result<Option<u64>> {
5822        let entry = self
5823            .entries
5824            .iter()
5825            .find(|entry| entry.name == name)
5826            .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
5827        let Some(field) = entry.fields.get(column) else {
5828            return Err(invalid("frequency column index out of range"));
5829        };
5830        if !matches!(
5831            field.ty,
5832            LogicalType::TinyInt
5833                | LogicalType::SmallInt
5834                | LogicalType::Integer
5835                | LogicalType::BigInt
5836                | LogicalType::UTinyInt
5837                | LogicalType::USmallInt
5838                | LogicalType::UInteger
5839                | LogicalType::UBigInt
5840        ) {
5841            return Ok(None);
5842        }
5843        let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
5844        if file_checksum(&self.file, offset, length)? != entry.directory.hash {
5845            return Err(invalid(&format!("the directory of table {name} does not checksum")));
5846        }
5847        if let Some(Some(frequencies)) = entry.frequencies.get(column) {
5848            return frequencies
5849                .iter()
5850                .filter(|(value, _)| value.is_some_and(|value| value != 0))
5851                .try_fold(0_u64, |total, (_, count)| total.checked_add(*count))
5852                .map(Some)
5853                .ok_or_else(|| invalid("numeric frequency count overflow"));
5854        }
5855        quick_nonzero(
5856            Cursor::over(&self.file, offset, length),
5857            &entry.name,
5858            &entry.fields,
5859            entry.rows,
5860            column,
5861        )
5862    }
5863
5864    /// Exact signed-integer sums and non-null counts from the small catalog. The table directory
5865    /// checksum is still checked once before any certificate can answer a query.
5866    pub fn aggregate_sums(&self, name: &str, columns: &[usize]) -> Result<Option<CertifiedSums>> {
5867        let entry = self
5868            .entries
5869            .iter()
5870            .find(|entry| entry.name == name)
5871            .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
5872        let mut sums = Vec::with_capacity(columns.len());
5873        for &column in columns {
5874            let Some(field) = entry.fields.get(column) else {
5875                return Err(invalid("aggregate column index out of range"));
5876            };
5877            if !signed_integer(&field.ty) {
5878                return Ok(None);
5879            }
5880            let Some(sum) = entry.aggregates[column] else {
5881                return Ok(None);
5882            };
5883            sums.push(sum);
5884        }
5885        let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
5886        if file_checksum(&self.file, offset, length)? != entry.directory.hash {
5887            return Err(invalid(&format!("the directory of table {name} does not checksum")));
5888        }
5889        Ok(Some(CertifiedSums { columns: sums, rows: entry.rows as u64 }))
5890    }
5891
5892    /// Exact non-null distinct count from the small catalog, after checking the table directory.
5893    pub fn distinct_count(&self, name: &str, column: usize) -> Result<Option<u64>> {
5894        let entry = self
5895            .entries
5896            .iter()
5897            .find(|entry| entry.name == name)
5898            .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
5899        let Some(count) = entry.distincts.get(column).copied() else {
5900            return Err(invalid("distinct column index out of range"));
5901        };
5902        let Some(count) = count else { return Ok(None) };
5903        let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
5904        if file_checksum(&self.file, offset, length)? != entry.directory.hash {
5905            return Err(invalid(&format!("the directory of table {name} does not checksum")));
5906        }
5907        Ok(Some(count))
5908    }
5909
5910    /// Exact integer or date ends from the small catalog after checking the table directory.
5911    pub fn integer_extremes(&self, name: &str, column: usize) -> Result<Option<IntegerExtremes>> {
5912        let entry = self
5913            .entries
5914            .iter()
5915            .find(|entry| entry.name == name)
5916            .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
5917        let Some(extremes) = entry.extremes.get(column).copied() else {
5918            return Err(invalid("extremes column index out of range"));
5919        };
5920        let Some(extremes) = extremes else { return Ok(None) };
5921        let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
5922        if file_checksum(&self.file, offset, length)? != entry.directory.hash {
5923            return Err(invalid(&format!("the directory of table {name} does not checksum")));
5924        }
5925        Ok(Some(match extremes {
5926            None => IntegerExtremes::Null,
5927            Some((low, high)) => IntegerExtremes::Values { low, high },
5928        }))
5929    }
5930
5931    /// Complete numeric frequencies from the small catalog, after checking the table directory.
5932    pub fn exact_numeric_frequencies(
5933        &self,
5934        name: &str,
5935        column: usize,
5936    ) -> Result<Option<NumericFrequencies>> {
5937        let entry = self
5938            .entries
5939            .iter()
5940            .find(|entry| entry.name == name)
5941            .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
5942        let Some(frequencies) = entry.frequencies.get(column).cloned() else {
5943            return Err(invalid("numeric frequency column index out of range"));
5944        };
5945        let Some(frequencies) = frequencies else { return Ok(None) };
5946        let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
5947        if file_checksum(&self.file, offset, length)? != entry.directory.hash {
5948            return Err(invalid(&format!("the directory of table {name} does not checksum")));
5949        }
5950        Ok(Some(frequencies))
5951    }
5952
5953    /// The schema copied into the small file catalog, available without opening the table directory.
5954    pub fn table_fields(&self, name: &str) -> Option<&[Field]> {
5955        self.entries.iter().find(|entry| entry.name == name).map(|entry| entry.fields.as_slice())
5956    }
5957}
5958
5959/// Where the slot naming `generation` goes, which is the one the generation before it did not use.
5960///
5961/// Generation 1 takes the slot at 16, so a file written once is byte for byte the file this wrote
5962/// before there was a second generation to write.
5963fn slot_offset(generation: u64) -> u64 {
5964    16 + (generation - 1) % 2 * SLOT_BYTES as u64
5965}
5966
5967/// The header and the bytes the highest valid slot points at.
5968///
5969/// Both levels of the directory are reached this way, so the magic check, the version check and the
5970/// choice between the two slots live here rather than being written out twice.
5971fn slot_bytes(path: impl AsRef<Path>) -> Result<(File, u64, Slot, Vec<u8>, Opening)> {
5972    let file = File::open(path).map_err(io)?;
5973    let size = file.metadata().map_err(io)?.len();
5974    let (slot, bytes, opening) = committed_slot(&file, size)?;
5975    Ok((file, size, slot, bytes, opening))
5976}
5977
5978/// The committed slot of a file that is `size` bytes long, and the catalog it points at.
5979///
5980/// The half of [`slot_bytes`] that does not care how the file was opened. A reader comes here with
5981/// the `std::fs::File` it goes on to share between its threads, and a writer with the `rudb_io`
5982/// file it is about to append to.
5983fn committed_slot<F: Positional + ?Sized>(file: &F, size: u64) -> Result<(Slot, Vec<u8>, Opening)> {
5984    if size < HEADER {
5985        return Err(invalid("file is shorter than its header"));
5986    }
5987    let mut header = [0; HEADER as usize];
5988    read_at(file, 0, &mut header)?;
5989    let mut opening = Opening { reads: 1, bytes: HEADER };
5990    let version = u32::from_le_bytes([header[8], header[9], header[10], header[11]]);
5991    // The two halves are worth telling apart. A wrong magic is a file that was never ours and
5992    // the answer is to look at the path. A wrong version is our own file from another build,
5993    // and the number this build wants is the only thing that tells the reader whether to
5994    // rebuild the file or to go back to the binary that wrote it.
5995    if &header[..8] != MAGIC {
5996        return Err(invalid("the header does not begin with a rudb native magic"));
5997    }
5998    if !READABLE.contains(&version) {
5999        return Err(invalid(&format!(
6000            "the file is format {version} and this build reads format {FORMAT}, so it has to \
6001                 be written again"
6002        )));
6003    }
6004    let mut selected = None;
6005    for start in [16, 16 + SLOT_BYTES] {
6006        let slot = Slot::read(&header[start..start + SLOT_BYTES]);
6007        if slot.generation == 0 || slot.length == 0 || slot.length as usize > MAX_DIRECTORY {
6008            continue;
6009        }
6010        let Some(end) = slot.offset.checked_add(u64::from(slot.length)) else { continue };
6011        if slot.offset < HEADER || end > size {
6012            continue;
6013        }
6014        let mut bytes = vec![0; slot.length as usize];
6015        read_at(file, slot.offset, &mut bytes)?;
6016        opening.reads += 1;
6017        opening.bytes += u64::from(slot.length);
6018        if checksum(&bytes) == slot.hash
6019            && selected
6020                .as_ref()
6021                .is_none_or(|(old, _): &(Slot, Vec<u8>)| old.generation < slot.generation)
6022        {
6023            selected = Some((slot, bytes));
6024        }
6025    }
6026    let (slot, bytes) = selected.ok_or_else(|| invalid("no committed directory slot is valid"))?;
6027    Ok((slot, bytes, opening))
6028}
6029
6030impl Reader {
6031    /// Opens a file that holds exactly one table.
6032    ///
6033    /// # Errors
6034    ///
6035    /// If the file has no valid committed directory, a directory pointer is out of bounds, or the
6036    /// file holds more than one table, which is a file that has to be opened by name.
6037    pub fn open(path: impl AsRef<Path>) -> Result<Self> {
6038        let catalog = Catalog::open(path)?;
6039        let mut names = catalog.names();
6040        let name = names.next().ok_or_else(|| invalid("the file holds no table"))?.to_string();
6041        if names.next().is_some() {
6042            return Err(invalid(
6043                "the file holds more than one table, so it has to be opened by name",
6044            ));
6045        }
6046        catalog.table(&name)
6047    }
6048
6049    /// Builds a reader over one decoded table directory.
6050    fn build(
6051        file: Arc<File>,
6052        size: u64,
6053        table: Table,
6054        directory: u64,
6055        opening: Opening,
6056        pool: PagePool,
6057    ) -> Result<Self> {
6058        let places = places(&table)?;
6059        let dictionaries = (0..table.fields.len()).map(|_| OnceLock::new()).collect();
6060        let table_fields = table.fields.len();
6061        let stripes = table.stripes.len();
6062        let columns = (0..table.fields.len())
6063            .map(|_| {
6064                Mutex::new(Cached {
6065                    pages: (0..stripes).map(|_| None).collect(),
6066                    index: (0..stripes).map(|_| None).collect(),
6067                    seen: vec![false; stripes],
6068                    ..Cached::default()
6069                })
6070            })
6071            .collect::<Vec<_>>();
6072        let cache = Shelf {
6073            columns,
6074            held: (0..table_fields).map(|_| AtomicUsize::new(0)).collect(),
6075            kept: AtomicUsize::new(CACHED_STRIPES_PER_COLUMN),
6076        };
6077        let sieves: Vec<Vec<SieveSlot>> = (0..table.fields.len())
6078            .map(|_| table.stripes.iter().map(|_| OnceLock::new()).collect())
6079            .collect();
6080        let part_ranges: Vec<Vec<RangeSlot>> = (0..table.fields.len())
6081            .map(|_| table.stripes.iter().map(|_| OnceLock::new()).collect())
6082            .collect();
6083        Ok(Self {
6084            file,
6085            table: Arc::new(table),
6086            dictionaries: Arc::new(dictionaries),
6087            loading: Arc::new((0..table_fields).map(|_| Mutex::new(())).collect()),
6088            frequency_values: Arc::new((0..table_fields).map(|_| OnceLock::new()).collect()),
6089            frequency_summaries: Arc::new((0..table_fields).map(|_| OnceLock::new()).collect()),
6090            opened: Arc::new(AtomicUsize::new(0)),
6091            sieves: Arc::new(sieves),
6092            part_ranges: Arc::new(part_ranges),
6093            places: Arc::new(places),
6094            cache: Arc::new(cache),
6095            pool,
6096            pages: Arc::new(AtomicUsize::new(0)),
6097            indexes: Arc::new(AtomicUsize::new(0)),
6098            size,
6099            directory,
6100            opening,
6101        })
6102    }
6103
6104    /// What this reader has read so far, and what opening it cost.
6105    ///
6106    /// Public because the claim of `spec/stats/04-in-memory.md` section 4.2 is about this number
6107    /// and a claim nobody can check is a comment. A caller that wants to know whether opening a
6108    /// file touched the data asks here, and gets an answer that does not depend on what the page
6109    /// cache happened to hold.
6110    #[must_use]
6111    pub fn reads(&self) -> Reads {
6112        Reads {
6113            opening: self.opening,
6114            pages: self.pages.load(Atomic::Relaxed),
6115            indexes: self.indexes.load(Atomic::Relaxed),
6116            dictionaries: self.opened.load(Atomic::Relaxed),
6117        }
6118    }
6119
6120    /// Where the file's bytes went, from the directory alone.
6121    ///
6122    /// No page is read, so this costs the same on a 45 GB table as on an empty one. See [`Layout`]
6123    /// for what is charged where and for why the three things that are not columns stay separate.
6124    #[must_use]
6125    pub fn layout(&self) -> Layout {
6126        let table = &self.table;
6127        let stripes = table.stripes.as_slice();
6128        let columns = table
6129            .fields
6130            .iter()
6131            .enumerate()
6132            .map(|(at, field)| ColumnLayout {
6133                name: field.name.clone(),
6134                kind: field.ty.to_string(),
6135                pages: sum(stripes.iter().map(|stripe| span_bytes(&stripe.pages, at))),
6136                memberships: sum(stripes.iter().map(|stripe| stripe.memberships.bytes(at))),
6137                sieves: sum(stripes.iter().map(|stripe| stripe.sieves.bytes(at))),
6138                part_ranges: sum(stripes.iter().map(|stripe| stripe.part_ranges.bytes(at))),
6139                dictionary: dictionary_bytes(table, at),
6140            })
6141            .collect();
6142        Layout {
6143            file: self.size,
6144            rows: table.rows,
6145            stripes: stripes.len(),
6146            parts: self.places.len(),
6147            columns,
6148            indexes: sum(stripes.iter().map(|stripe| u64::from(stripe.index.length))),
6149            directory: self.directory,
6150            header: HEADER,
6151        }
6152    }
6153
6154    /// What every part of one column is stored as, which is what `pragma_storage_info` reports.
6155    ///
6156    /// Unlike [`Self::layout`] this reads the data, because the encoder's choice is in the page and
6157    /// nowhere else. The directory says how many bytes a column took and says nothing about what
6158    /// shape they are in, and the shape is the question worth asking: the same rows in a different
6159    /// order come back bit packed on one file and plain on another, and that is the difference a
6160    /// clustered load makes to a scan.
6161    ///
6162    /// One read per stripe rather than one per part. A part is a few kilobytes out of a page that
6163    /// is a quarter of a megabyte, so asking part by part would read the same page sixty four
6164    /// times. Nothing is put in the page cache, because a caller asking what a file looks like is
6165    /// not about to scan it and evicting the pages a real query wants would be a poor trade.
6166    ///
6167    /// # Errors
6168    ///
6169    /// If the column is outside the schema, or a page, index section or checksum is invalid.
6170    pub fn stored(&self, column: usize) -> Result<Vec<StoredPart>> {
6171        let field = self
6172            .table
6173            .fields
6174            .get(column)
6175            .ok_or_else(|| invalid("stored column index out of range"))?;
6176        let mut stored = Vec::with_capacity(self.places.len());
6177        let mut row = 0;
6178        for (at, stripe) in self.table.stripes.iter().enumerate() {
6179            let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
6180            let index = read_index(&self.file, stripe, column)?;
6181            let mut bytes = vec![0; page.length as usize];
6182            read_at(&self.file, page.offset, &mut bytes)?;
6183            let ranges = self.stripe_part_ranges(at, column);
6184            for (part, &rows) in stripe.parts.iter().enumerate() {
6185                let span = *index.get(part).ok_or_else(|| invalid("part index out of range"))?;
6186                let held = part_bytes(&bytes, span)?;
6187                let range = ranges.and_then(|held| held.get(part));
6188                stored.push(StoredPart {
6189                    stripe: at,
6190                    part,
6191                    row,
6192                    rows: rows as usize,
6193                    encoding: page_encoding(&field.ty, rows as usize, held),
6194                    bytes: span.length as u64,
6195                    page: page.offset,
6196                    offset: span.start as u64,
6197                    low: range
6198                        .and_then(|range| range.low.clone())
6199                        .and_then(|bound| bound.into_value(&field.ty)),
6200                    high: range
6201                        .and_then(|range| range.high.clone())
6202                        .and_then(|bound| bound.into_value(&field.ty)),
6203                    nulls: range.map(|range| range.nulls),
6204                });
6205                row += rows as usize;
6206            }
6207        }
6208        Ok(stored)
6209    }
6210
6211    /// How many parts the table has, which is how many chunks a scan of it reads.
6212    #[must_use]
6213    pub fn parts(&self) -> usize {
6214        self.places.len()
6215    }
6216
6217    /// The parts of each stripe, in table wide part numbers.
6218    ///
6219    /// A scan that wants one worker to own the page it reads hands work out in these runs. The
6220    /// stripes are contiguous in part numbering and all but the last hold sixty four parts, but a
6221    /// stripe can be flushed early when rows arrive out of order, so the runs are read off the
6222    /// directory rather than worked out from a constant.
6223    #[must_use]
6224    pub fn stripe_parts(&self) -> Vec<std::ops::Range<usize>> {
6225        let mut runs = Vec::with_capacity(self.table.stripes.len());
6226        let mut start = 0;
6227        for stripe in &self.table.stripes {
6228            let end = start + stripe.parts.len();
6229            runs.push(start..end);
6230            start = end;
6231        }
6232        runs
6233    }
6234
6235    /// How many rows one stripe holds, in the numbering [`Self::stripe_parts`] hands back.
6236    ///
6237    /// Off the directory, which is already in memory, rather than by the caller asking for each
6238    /// part in turn through the catalog. Nothing past the end holds any rows.
6239    #[must_use]
6240    pub fn stripe_rows(&self, stripe: usize) -> usize {
6241        self.table.stripes.get(stripe).map_or(0, |held| held.rows)
6242    }
6243
6244    /// Asks the page cache to keep `stripes` stripes of every column instead of the default.
6245    ///
6246    /// This only ever raises the number. A scan that gives each worker a whole stripe has one page
6247    /// per column per worker open at once, and a cache smaller than that is worse than no cache at
6248    /// all: every worker's page is evicted by the others before it has finished its stripe, so it
6249    /// reads a quarter of a megabyte for every part it takes out of it.
6250    pub fn keep_stripes(&self, stripes: usize) {
6251        self.cache.kept.fetch_max(stripes, Atomic::Relaxed);
6252    }
6253
6254    /// Rows in one part, or zero when the part number is past the table.
6255    #[must_use]
6256    pub fn part_rows(&self, at: usize) -> usize {
6257        self.places.get(at).map_or(0, |place| place.rows as usize)
6258    }
6259
6260    /// The committed table directory.
6261    #[must_use]
6262    pub fn table(&self) -> &Table {
6263        &self.table
6264    }
6265
6266    /// Exact leading frequencies when the stored synopsis proves a count-descending prefix.
6267    ///
6268    /// The returned list can be longer than `top`. Keeping the stored tail lets a later TopN apply
6269    /// additional ordering keys without losing a value tied with the requested boundary.
6270    ///
6271    /// # Errors
6272    ///
6273    /// If the column is outside the schema or a stored value does not fit its declared type.
6274    pub fn top_frequencies(&self, column: usize, top: usize) -> Result<Option<Vec<(Value, u64)>>> {
6275        let field = self
6276            .table
6277            .fields
6278            .get(column)
6279            .ok_or_else(|| invalid("frequency column index out of range"))?;
6280        let Some(summary) = self.frequency_summary(column)? else {
6281            return Ok(None);
6282        };
6283        if top == 0 || summary.entries.len() < top {
6284            return Ok(None);
6285        }
6286        let boundary = summary.entries[top - 1].count;
6287        if boundary <= summary.omitted_max {
6288            return Ok(None);
6289        }
6290        self.decode_frequencies(column, &field.ty, &summary.entries).map(Some)
6291    }
6292
6293    /// Exact leading counts for a numeric key paired with a stable-dictionary string key.
6294    ///
6295    /// Legacy pair summaries are parsed for file compatibility but never used as query output.
6296    ///
6297    /// # Errors
6298    ///
6299    /// If either column is outside the schema.
6300    pub fn top_pair_frequencies(
6301        &self,
6302        first: usize,
6303        second: usize,
6304        _top: usize,
6305    ) -> Result<Option<PairFrequencyCounts>> {
6306        if first >= self.table.fields.len() || second >= self.table.fields.len() {
6307            return Err(invalid("pair frequency column index out of range"));
6308        }
6309        Ok(None)
6310    }
6311
6312    /// Every value of one column with the number of rows holding it, when the synopsis is complete.
6313    ///
6314    /// The heavy hitter pass keeps a bounded set of candidates and decrements them all when it runs
6315    /// out of room, so what it usually ends with is the leading values and a bound on everything it
6316    /// dropped. `omitted_max` of zero says that never happened: no candidate was ever decremented and
6317    /// the entries did not overflow the stored budget, so the list is every distinct value of the
6318    /// column with an exact count, and a null counts as a value of its own rather than being skipped.
6319    ///
6320    /// That makes a whole class of question answerable without reading a row. How many rows hold a
6321    /// value, how many do not, and what a `GROUP BY` of that column with a count over it produces are
6322    /// all in here. It is only ever true of a column with few enough distinct values, which is the
6323    /// case worth having, because that is exactly the column a grouping or an equality filter would
6324    /// otherwise walk every row to answer.
6325    ///
6326    /// `None` when the column has no synopsis, or has one that dropped anything.
6327    ///
6328    /// # Errors
6329    ///
6330    /// If the column is outside the schema or a stored value does not fit its declared type.
6331    pub fn exact_frequencies(&self, column: usize) -> Result<Option<Vec<(Value, u64)>>> {
6332        let Some(prefix) = self.frequency_prefix(column)? else {
6333            return Ok(None);
6334        };
6335        Ok((prefix.omitted_max == 0).then_some(prefix.entries))
6336    }
6337
6338    /// Every value the synopsis lists with the number of rows holding it, and a bound on the rest.
6339    ///
6340    /// The counts are exact whether or not the list is complete. The heavy hitter pass keeps a
6341    /// bounded candidate set and then recounts only the candidates that survived it, so a value that
6342    /// made it into the list carries the number of rows that really hold it rather than whatever the
6343    /// pass had left over. What the pass loses is values, not counts.
6344    ///
6345    /// `omitted_max` is how many rows the most common value left out can hold, and zero says nothing
6346    /// was left out at all, which is what [`exact_frequencies`] asks for. Above zero the list is the
6347    /// leading values of the column and everything else is somewhere between no rows and that bound.
6348    ///
6349    /// That prefix is worth reading on its own. A column with a value in half its rows and a long
6350    /// tail behind it has no complete synopsis and never will, and it is the column where dividing
6351    /// the rows by the distinct count is furthest from the truth.
6352    ///
6353    /// `None` when the column has no synopsis.
6354    ///
6355    /// # Errors
6356    ///
6357    /// If the column is outside the schema or a stored value does not fit its declared type.
6358    ///
6359    /// [`exact_frequencies`]: Self::exact_frequencies
6360    pub fn frequency_prefix(&self, column: usize) -> Result<Option<FrequencyPrefix>> {
6361        let field = self
6362            .table
6363            .fields
6364            .get(column)
6365            .ok_or_else(|| invalid("frequency column index out of range"))?;
6366        let Some(summary) = self.frequency_summary(column)? else {
6367            return Ok(None);
6368        };
6369        let entries = self.decode_frequencies(column, &field.ty, &summary.entries)?;
6370        Ok(Some(FrequencyPrefix { entries, omitted_max: summary.omitted_max }))
6371    }
6372
6373    /// One column's synopsis, read back from the file when the directory left it there.
6374    fn frequency_summary(&self, column: usize) -> Result<Option<Cow<'_, FrequencySummary>>> {
6375        Ok(match self.table.frequencies.get(column) {
6376            None | Some(None) => None,
6377            Some(Some(Frequencies::Held(summary))) => Some(Cow::Borrowed(summary)),
6378            Some(Some(Frequencies::Stored { span, values })) => {
6379                let slot = self
6380                    .frequency_summaries
6381                    .get(column)
6382                    .ok_or_else(|| invalid("frequency column index out of range"))?;
6383                if let Some(summary) = slot.get() {
6384                    return Ok(Some(Cow::Borrowed(summary.as_ref())));
6385                }
6386                let field = self
6387                    .table
6388                    .fields
6389                    .get(column)
6390                    .ok_or_else(|| invalid("frequency column index out of range"))?;
6391                let mut bytes = vec![0; span.length as usize];
6392                read_at(&self.file, span.offset, &mut bytes)?;
6393                let summary =
6394                    decode_summary(&mut Cursor::new(&bytes), field, self.table.rows, *values)?;
6395                let summary = summary.ok_or_else(|| invalid("a stored synopsis is missing"))?;
6396                let _ = slot.set(Arc::new(summary));
6397                Some(Cow::Borrowed(slot.get().expect("the decoded summary was stored").as_ref()))
6398            }
6399        })
6400    }
6401
6402    /// Turns stored frequency entries into values of the column's own type.
6403    ///
6404    /// Remembered per column, because the planner asks once for every estimate that touches the
6405    /// column and the executor asks again, and the answer is a few hundred values. The codes of a
6406    /// string column are read through [`Vector::try_values_visited`], which does not keep the blocks
6407    /// it decodes, so what a query answered out of the synopsis holds is those values and not the
6408    /// hundred or so dictionary blocks they are scattered over.
6409    fn decode_frequencies(
6410        &self,
6411        column: usize,
6412        ty: &LogicalType,
6413        entries: &[FrequencyEntry],
6414    ) -> Result<Vec<(Value, u64)>> {
6415        if let Some(values) = self.frequency_values.get(column).and_then(OnceLock::get) {
6416            return Ok(values.as_ref().clone());
6417        }
6418        let values = self.decode_frequencies_once(column, ty, entries)?;
6419        if let Some(slot) = self.frequency_values.get(column) {
6420            let _ = slot.set(Arc::new(values.clone()));
6421        }
6422        Ok(values)
6423    }
6424
6425    fn decode_frequencies_once(
6426        &self,
6427        column: usize,
6428        ty: &LogicalType,
6429        entries: &[FrequencyEntry],
6430    ) -> Result<Vec<(Value, u64)>> {
6431        let stored_texts = self.table.frequency_texts.get(column).filter(|texts| !texts.is_empty());
6432        if stored_texts.is_some_and(|texts| texts.len() != entries.len()) {
6433            return Err(invalid("frequency text count differs from its synopsis"));
6434        }
6435        let dictionary =
6436            if coded_type(ty) && stored_texts.is_none() { self.dictionary(column)? } else { None };
6437        let mut codes = entries
6438            .iter()
6439            .filter_map(|entry| match entry.value {
6440                FrequencyValue::Code(code) => Some(code as usize),
6441                _ => None,
6442            })
6443            .collect::<Vec<_>>();
6444        codes.sort_unstable();
6445        codes.dedup();
6446        let texts = match &dictionary {
6447            Some(dictionary) if !codes.is_empty() => dictionary.try_values_visited(&codes)?,
6448            _ => Vec::new(),
6449        };
6450        let mut out = Vec::with_capacity(entries.len());
6451        for (entry_at, entry) in entries.iter().enumerate() {
6452            let value = match entry.value {
6453                FrequencyValue::Null => {
6454                    if stored_texts.and_then(|texts| texts[entry_at].as_ref()).is_some() {
6455                        return Err(invalid("a null frequency entry has text"));
6456                    }
6457                    Value::Null
6458                }
6459                FrequencyValue::Integer(value) => match *ty {
6460                    LogicalType::TinyInt => Value::TinyInt(
6461                        i8::try_from(value)
6462                            .map_err(|_| invalid("frequency TINYINT is out of range"))?,
6463                    ),
6464                    LogicalType::UTinyInt => Value::UTinyInt(
6465                        u8::try_from(value)
6466                            .map_err(|_| invalid("frequency UTINYINT is out of range"))?,
6467                    ),
6468                    LogicalType::USmallInt => Value::USmallInt(
6469                        u16::try_from(value)
6470                            .map_err(|_| invalid("frequency USMALLINT is out of range"))?,
6471                    ),
6472                    LogicalType::UInteger => Value::UInteger(
6473                        u32::try_from(value)
6474                            .map_err(|_| invalid("frequency UINTEGER is out of range"))?,
6475                    ),
6476                    LogicalType::UBigInt => Value::UBigInt(
6477                        u64::try_from(value)
6478                            .map_err(|_| invalid("frequency UBIGINT is out of range"))?,
6479                    ),
6480                    LogicalType::SmallInt => Value::SmallInt(
6481                        i16::try_from(value)
6482                            .map_err(|_| invalid("frequency SMALLINT is out of range"))?,
6483                    ),
6484                    LogicalType::Integer => Value::Integer(
6485                        i32::try_from(value)
6486                            .map_err(|_| invalid("frequency INTEGER is out of range"))?,
6487                    ),
6488                    LogicalType::BigInt => Value::BigInt(
6489                        i64::try_from(value)
6490                            .map_err(|_| invalid("frequency BIGINT is out of range"))?,
6491                    ),
6492                    LogicalType::Date => Value::Date(
6493                        i32::try_from(value)
6494                            .map_err(|_| invalid("frequency DATE is out of range"))?,
6495                    ),
6496                    LogicalType::Timestamp => Value::Timestamp(
6497                        i64::try_from(value)
6498                            .map_err(|_| invalid("frequency TIMESTAMP is out of range"))?,
6499                    ),
6500                    _ => return Err(invalid("integer frequency belongs to another type")),
6501                },
6502                FrequencyValue::Code(code) => {
6503                    if let Some(text) = stored_texts.and_then(|texts| texts[entry_at].as_ref()) {
6504                        if *ty == LogicalType::Blob {
6505                            Value::Blob(text.clone())
6506                        } else {
6507                            Value::Varchar(
6508                                String::from_utf8(text.clone())
6509                                    .map_err(|_| invalid("frequency text is not UTF-8"))?,
6510                            )
6511                        }
6512                    } else {
6513                        if dictionary.is_none() {
6514                            return Err(invalid("frequency code has no dictionary or stored text"));
6515                        }
6516                        let at = codes
6517                            .binary_search(&(code as usize))
6518                            .map_err(|_| invalid("frequency code was not among the codes read"))?;
6519                        texts[at].clone()
6520                    }
6521                }
6522            };
6523            out.push((value, entry.count));
6524        }
6525        Ok(out)
6526    }
6527
6528    /// Sparse rows belonging to the bounded numeric frequency candidate set.
6529    ///
6530    /// The list is omitted when collecting it would exceed the fixed storage budget. A composite
6531    /// aggregate may accept a result over these rows only when its requested boundary is strictly
6532    /// greater than `omitted_max`.
6533    ///
6534    /// # Errors
6535    ///
6536    /// If the column is outside the schema.
6537    pub fn frequency_occurrences(&self, column: usize) -> Result<Option<FrequencyOccurrences>> {
6538        let field = self
6539            .table
6540            .fields
6541            .get(column)
6542            .ok_or_else(|| invalid("frequency column index out of range"))?;
6543        let Some(summary) = self.frequency_summary(column)? else {
6544            return Ok(None);
6545        };
6546        if summary.ordinals.is_empty() {
6547            return Ok(None);
6548        }
6549        let (anchors, anchor_indices) = if summary.ordinal_entries.len() == summary.ordinals.len() {
6550            let entries = self.decode_frequencies(column, &field.ty, &summary.entries)?;
6551            (entries.into_iter().map(|(value, _)| value).collect(), summary.ordinal_entries.clone())
6552        } else {
6553            (Vec::new(), Vec::new())
6554        };
6555        Ok(Some(FrequencyOccurrences {
6556            omitted_max: summary.omitted_max,
6557            ordinals: summary.ordinals.clone(),
6558            anchors,
6559            anchor_indices,
6560        }))
6561    }
6562
6563    /// How many distinct values one column holds, counting a null as no value.
6564    ///
6565    /// A string column of this format is written against one dictionary that covers the whole table.
6566    /// A code is handed out the first time a value is seen and nothing ever removes one, so the
6567    /// number of codes is the number of distinct values exactly rather than an estimate. That makes
6568    /// `COUNT(DISTINCT column)` over a whole table a question the directory already knows the answer
6569    /// to, and the alternative is a hash table with a row per distinct value built from a pass over
6570    /// every row.
6571    ///
6572    /// A null in the column used to make this `None` and no longer does. A null row is written as
6573    /// the code for the empty string, so a nullable column's dictionary can hold an empty string
6574    /// that no row of it actually has, and the dictionary on its own does not say which case it is.
6575    /// The writer does know, because it counts the non-null rows that use each code on its way to
6576    /// the frequency summary, so it records how many codes any row holds and the directory carries
6577    /// that number. This reads it rather than the size of the dictionary, which also means the
6578    /// dictionary page is not opened to answer.
6579    ///
6580    /// An integer column has no dictionary, and its count comes from the set the writer keeps on its
6581    /// numeric frequency pass instead, which is exact up to a cap. `None` for a column past that cap
6582    /// and for every column that is neither, where a sketch would answer approximately and SQL asked
6583    /// for the exact number.
6584    ///
6585    /// # Errors
6586    ///
6587    /// If the column is outside the schema.
6588    pub fn distinct_values(&self, column: usize) -> Result<Option<u64>> {
6589        self.table
6590            .distincts
6591            .get(column)
6592            .copied()
6593            .ok_or_else(|| invalid("distinct column index out of range"))
6594    }
6595
6596    /// How many rows of one column are null, added up over the stripes.
6597    ///
6598    /// Every stripe records this exactly when it is written, because a null count is not a bound
6599    /// that is allowed to be wide the way a minimum and a maximum are: a filter that reads one too
6600    /// many is slow and a `COUNT` that reads one too many is wrong. Adding up a few hundred numbers
6601    /// already in memory is what makes `COUNT(column)` over a whole table free.
6602    ///
6603    /// # Errors
6604    ///
6605    /// If the column is outside the schema.
6606    pub fn null_count(&self, column: usize) -> Result<u64> {
6607        if column >= self.table.fields.len() {
6608            return Err(invalid("null count column index out of range"));
6609        }
6610        let mut nulls = 0_u64;
6611        for stripe in &self.table.stripes {
6612            let range = stripe
6613                .zone
6614                .column(column)
6615                .ok_or_else(|| invalid("stripe zone is narrower than the schema"))?;
6616            nulls = nulls
6617                .checked_add(range.nulls as u64)
6618                .ok_or_else(|| invalid("null count overflow"))?;
6619        }
6620        Ok(nulls)
6621    }
6622
6623    /// The smallest and the largest value of one string column, from the order beside its values.
6624    ///
6625    /// The dictionary holds exactly the values the column holds, so the first and the last of them
6626    /// in sorted order are the column's minimum and maximum. Two reads of a rank block settle what
6627    /// otherwise walks a million rows.
6628    ///
6629    /// `None` when the column is not a string, when the file was written before version 9 and so has
6630    /// no order, when the column has no values at all, or when it has a null in it, which is the
6631    /// placeholder again: the empty string a null is written as would sort ahead of every real
6632    /// value and be reported as the minimum.
6633    ///
6634    /// # Errors
6635    ///
6636    /// If the column is outside the schema, or a rank names a code the dictionary does not have.
6637    pub fn text_extremes(&self, column: usize) -> Result<Option<(Value, Value)>> {
6638        if self.null_count(column)? > 0 || self.demoted(column) {
6639            return Ok(None);
6640        }
6641        let Some(dictionary) = self.dictionary(column)? else { return Ok(None) };
6642        let Some(ranks) = dictionary.ranks() else { return Ok(None) };
6643        if ranks == 0 {
6644            return Ok(None);
6645        }
6646        let low = text_at_rank(&dictionary, 0)?;
6647        let high = text_at_rank(&dictionary, ranks - 1)?;
6648        Ok(Some((low, high)))
6649    }
6650
6651    /// The smallest and the largest value of one column, when every stripe wrote exact ends.
6652    ///
6653    /// A stripe's ends are allowed to be wider than the truth, because a bound that rules out a
6654    /// chunk that could not match is still correct when it rules out nothing. That is what makes
6655    /// them cheap to write for a bit packed or a dictionary column, and it is also what stops them
6656    /// answering a `MIN`. So each stripe says which of the two it wrote, and this answers only when
6657    /// all of them walked their rows.
6658    ///
6659    /// `None` for a column with no ends, for an empty table, and for a column any stripe of which
6660    /// guessed. Nulls need no special case, because the ends skip them the same way `MIN` does.
6661    ///
6662    /// One case is given up on that did not have to be. A stripe merges the ends of its sixty four
6663    /// parts, and a part with no ends at all erases the merged ones, because a part whose rows are
6664    /// not covered by the stripe's ends is a stripe that would skip rows it should keep. A part of
6665    /// nothing but nulls has no rows to cover and so did not need to erase anything, but the merge
6666    /// cannot tell that part from a part whose layout it could not read. So a column with a chunk
6667    /// of nothing but nulls in the middle of it goes and reads the rows. That is slow and right,
6668    /// and the fix is a row count per part rather than anything here.
6669    ///
6670    /// # Errors
6671    ///
6672    /// If the column is outside the schema.
6673    pub fn exact_extremes(&self, column: usize) -> Result<Option<(Bound, Bound)>> {
6674        if column >= self.table.fields.len() {
6675            return Err(invalid("extremes column index out of range"));
6676        }
6677        let mut low: Option<Bound> = None;
6678        let mut high: Option<Bound> = None;
6679        for stripe in &self.table.stripes {
6680            let range = stripe
6681                .zone
6682                .column(column)
6683                .ok_or_else(|| invalid("stripe zone is narrower than the schema"))?;
6684            if !range.exact {
6685                return Ok(None);
6686            }
6687            // A stripe of nothing but nulls has no ends and says nothing about the column's, which
6688            // is why this skips it rather than giving up on the whole column. A stripe that has
6689            // rows and still has no end is a layout whose values this cannot see, and skipping that
6690            // one would answer with an end taken from the other stripes, so it gives up instead.
6691            let (Some(small), Some(large)) = (range.low.as_ref(), range.high.as_ref()) else {
6692                if stripe.rows > range.nulls {
6693                    return Ok(None);
6694                }
6695                continue;
6696            };
6697            low = Some(low.map_or_else(|| small.clone(), |held| held.smaller(small.clone())));
6698            high = Some(high.map_or_else(|| large.clone(), |held| held.larger(large.clone())));
6699        }
6700        Ok(low.zip(high))
6701    }
6702
6703    /// The sum of one integer column and how many rows went into it, when every stripe wrote one.
6704    ///
6705    /// The count beside the sum is the non-null rows, because that is what a `SUM` adds up and what
6706    /// an `AVG` divides by, and a caller that had to work it out from the row count and the null
6707    /// count would be doing the same walk twice.
6708    ///
6709    /// `None` for anything that is not an integer column, for a file written by something that did
6710    /// not record it, and when adding the stripes together would overflow.
6711    ///
6712    /// # Errors
6713    ///
6714    /// If the column is outside the schema.
6715    pub fn exact_sum(&self, column: usize) -> Result<Option<(i128, u64)>> {
6716        if column >= self.table.fields.len() {
6717            return Err(invalid("sum column index out of range"));
6718        }
6719        let mut total = 0_i128;
6720        let mut rows = 0_u64;
6721        for stripe in &self.table.stripes {
6722            let range = stripe
6723                .zone
6724                .column(column)
6725                .ok_or_else(|| invalid("stripe zone is narrower than the schema"))?;
6726            let Some(part) = range.sum else { return Ok(None) };
6727            let Some(sum) = total.checked_add(part) else { return Ok(None) };
6728            total = sum;
6729            rows = rows.saturating_add(stripe.rows as u64 - range.nulls as u64);
6730        }
6731        Ok(Some((total, rows)))
6732    }
6733
6734    /// Legacy derived host groups are parsed for file compatibility but never used as query output.
6735    pub fn host_groups(
6736        &self,
6737        column: usize,
6738        _minimum_count: u64,
6739    ) -> Result<Option<Vec<host::HostEntry>>> {
6740        if column >= self.table.fields.len() {
6741            return Err(invalid("host group column index out of range"));
6742        }
6743        Ok(None)
6744    }
6745
6746    /// Whether the column's dictionary stopped taking values partway through the load, and so
6747    /// decodes the stripes written before that and says nothing about the column as a whole. See
6748    /// `DEMOTED`.
6749    #[must_use]
6750    pub fn demoted(&self, column: usize) -> bool {
6751        self.table.demoted.get(column).copied().unwrap_or(false)
6752    }
6753
6754    /// The global dictionary of a column, opened once however many workers ask for it at once.
6755    ///
6756    /// The unlocked look is first because it is the answer every time after the first and it costs a
6757    /// load. Everybody who misses it queues on [`Self::loading`] and looks again on the way in, so
6758    /// the one who arrived first does the reading and the rest take what it left. Waiting is the
6759    /// cheaper thing to do: the work behind the lock is a page read, a checksum and the decode of a
6760    /// dictionary that can hold half a million entries, and the alternative is every worker of the
6761    /// scan doing all of it and all but one dropping the result on the floor.
6762    fn dictionary(&self, column: usize) -> Result<Option<Arc<Vector>>> {
6763        let Some(page) = self.table.dictionaries[column] else { return Ok(None) };
6764        if let Some(dictionary) = self.dictionaries[column].get() {
6765            return Ok(Some(Arc::clone(dictionary)));
6766        }
6767        let _queued = self.loading[column].lock().map_err(|_| invalid("a poisoned dictionary"))?;
6768        if let Some(dictionary) = self.dictionaries[column].get() {
6769            return Ok(Some(Arc::clone(dictionary)));
6770        }
6771        self.opened.fetch_add(1, Atomic::Relaxed);
6772        let dictionary = Arc::new(open_global_dictionary(
6773            Arc::clone(&self.file),
6774            page,
6775            &self.table.fields[column].ty,
6776            TEXT_KEEP_BUDGET,
6777        )?);
6778        let _ = self.dictionaries[column].set(Arc::clone(&dictionary));
6779        Ok(Some(dictionary))
6780    }
6781
6782    /// Reads one section's extent table and checks it against the entry that names it.
6783    ///
6784    /// # Errors
6785    ///
6786    /// If the entry points outside the file, the table does not checksum, or it does not decode as
6787    /// a run of extents in element order.
6788    pub fn extents(&self, of: &Section) -> Result<Vec<section::Extent>> {
6789        if of.extent_bytes == 0 {
6790            return Ok(Vec::new());
6791        }
6792        let mut bytes = vec![0; of.extent_bytes as usize];
6793        read_at(&self.file, of.extent_page, &mut bytes)?;
6794        if checksum(&bytes) != of.hash {
6795            return Err(invalid("a section's extent table does not checksum"));
6796        }
6797        let extents = section::decode_extents(&bytes)?;
6798        if extents.len() != of.extents as usize {
6799            return Err(invalid("a section's extent table is not the length the entry says"));
6800        }
6801        Ok(extents)
6802    }
6803
6804    /// Reads and verifies one extent of a section.
6805    ///
6806    /// This is what section 3.2's second rule is for. A reduction that needs one extent of a two
6807    /// gigabyte forward link reads and checksums that extent and nothing else, which is the whole
6808    /// difference between a structure that works at SF100 and issue #745.
6809    ///
6810    /// # Errors
6811    ///
6812    /// If the extent points outside the file, or its bytes do not checksum.
6813    pub fn extent(&self, of: &section::Extent) -> Result<Vec<u8>> {
6814        let mut bytes = Vec::new();
6815        self.extent_into(of, &mut bytes)?;
6816        Ok(bytes)
6817    }
6818
6819    /// Read a verified extent into a caller-owned buffer so repeated extents can reuse its pages.
6820    fn extent_into(&self, of: &section::Extent, bytes: &mut Vec<u8>) -> Result<()> {
6821        let end = of
6822            .offset
6823            .checked_add(u64::from(of.length))
6824            .ok_or_else(|| invalid("an extent overflows the file"))?;
6825        if of.offset < HEADER || end > self.size {
6826            return Err(invalid("an extent is outside the file"));
6827        }
6828        bytes.resize(of.length as usize, 0);
6829        read_at(&self.file, of.offset, bytes)?;
6830        if checksum(bytes) != of.hash {
6831            return Err(invalid("an extent does not checksum"));
6832        }
6833        Ok(())
6834    }
6835
6836    /// Reads a whole section's payload, every extent of it, in order.
6837    ///
6838    /// For a structure that is resident anyway, which a key map is. Anything large enough that the
6839    /// split matters should be walking [`Reader::extents`] and taking the one it needs.
6840    ///
6841    /// # Errors
6842    ///
6843    /// If the extent table or any extent fails its check.
6844    pub fn payload(&self, of: &Section) -> Result<Vec<u8>> {
6845        let extents = self.extents(of)?;
6846        let mut bytes =
6847            Vec::with_capacity(sum(extents.iter().map(|one| u64::from(one.length))) as usize);
6848        for one in &extents {
6849            if one.first != bytes.len() as u64 {
6850                return Err(invalid("a section's extents do not join up"));
6851            }
6852            bytes.extend_from_slice(&self.extent(one)?);
6853        }
6854        // The same exception `write_section` makes: a budget record has no bytes, so its
6855        // `header_bytes` is a size rather than a header and there is nothing for it to run past.
6856        if !bytes.is_empty() && of.header_bytes as usize > bytes.len() {
6857            return Err(invalid("a section's header is longer than its payload"));
6858        }
6859        Ok(bytes)
6860    }
6861
6862    /// Reads only the named columns from one part.
6863    ///
6864    /// The whole stripe page each column lives in is read and kept, because a scan asks for the
6865    /// parts of a stripe one after another and this is what turns sixty four reads into one.
6866    ///
6867    /// # Errors
6868    ///
6869    /// If a part, column, page, or checksum is invalid.
6870    pub fn read(&self, part: usize, columns: &[usize]) -> Result<Chunk> {
6871        self.read_impl(part, columns, true, None)
6872    }
6873
6874    /// Reads named columns from one part without keeping the stripe page it came out of.
6875    ///
6876    /// This is for sparse row fetches after a selective TopN or filter, which reach a few parts of
6877    /// a stripe rather than all of them. A caller that will read most of a stripe should use
6878    /// [`Self::read`] instead, because this reads and discards the page index every time.
6879    ///
6880    /// # Errors
6881    ///
6882    /// If a part, column, page, or checksum is invalid.
6883    pub fn read_sparse(&self, part: usize, columns: &[usize]) -> Result<Chunk> {
6884        self.read_impl(part, columns, false, None)
6885    }
6886
6887    /// Counts one signed integer part from its encoded row values when it uses an all-valid
6888    /// cascade. Sparse and run-length cascades are folded without expanding their rows. Other
6889    /// page forms return `None` so the caller can use the ordinary reader.
6890    ///
6891    /// # Errors
6892    ///
6893    /// If a part, column, page checksum, or encoded integer is invalid.
6894    pub fn integer_tally(&self, part: usize, column: usize) -> Result<Option<Vec<(i64, u64)>>> {
6895        let place = *self.places.get(part).ok_or_else(|| invalid("part index out of range"))?;
6896        let field =
6897            self.table.fields.get(column).ok_or_else(|| invalid("column index out of range"))?;
6898        if !matches!(
6899            field.ty,
6900            LogicalType::TinyInt
6901                | LogicalType::SmallInt
6902                | LogicalType::Integer
6903                | LogicalType::BigInt
6904        ) {
6905            return Ok(None);
6906        }
6907        let stripe_index = place.stripe as usize;
6908        let stripe = self
6909            .table
6910            .stripes
6911            .get(stripe_index)
6912            .ok_or_else(|| invalid("stripe index out of range"))?;
6913        let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
6914        let held = self.held(stripe_index, stripe, column, true)?;
6915        let span = *held
6916            .index
6917            .get(place.part as usize)
6918            .ok_or_else(|| invalid("part index out of range"))?;
6919        let owned;
6920        let bytes = match &held.page {
6921            Some(page) => page.part(place.part as usize, span)?,
6922            None => {
6923                let offset = page
6924                    .offset
6925                    .checked_add(span.start as u64)
6926                    .ok_or_else(|| invalid("part range overflow"))?;
6927                let mut bytes = vec![0; span.length];
6928                read_at(&self.file, offset, &mut bytes)?;
6929                verify_part(&bytes, span)?;
6930                owned = bytes;
6931                &owned
6932            }
6933        };
6934        if bytes.first() != Some(&5) || bytes.get(1) != Some(&0) {
6935            return Ok(None);
6936        }
6937        let (rows, counts) = integer::tally(&bytes[2..])?;
6938        if rows != place.rows as usize {
6939            return Err(invalid("encoded integer part holds the wrong number of rows"));
6940        }
6941        for &(value, _) in &counts {
6942            let fits = match field.ty {
6943                LogicalType::TinyInt => i8::try_from(value).is_ok(),
6944                LogicalType::SmallInt => i16::try_from(value).is_ok(),
6945                LogicalType::Integer => i32::try_from(value).is_ok(),
6946                LogicalType::BigInt => true,
6947                _ => false,
6948            };
6949            if !fits {
6950                return Err(invalid("encoded integer value is outside its column type"));
6951            }
6952        }
6953        Ok(Some(counts))
6954    }
6955
6956    /// Reads named columns from one part, only at the rows `positions` names.
6957    ///
6958    /// For a scan that already knows which rows of the part it keeps, from the columns it read
6959    /// first. A compressed string page decompresses only those rows, and every other page is
6960    /// decoded whole and gathered, which is what reading it and narrowing it costs anyway. With
6961    /// `whole` the stripe's pages are kept the way [`Self::read`] keeps them, and without it they
6962    /// are not, the way [`Self::read_sparse`] does.
6963    ///
6964    /// # Errors
6965    ///
6966    /// If a part, column, page, or checksum is invalid, or the positions do not rise or run past
6967    /// the end of the part.
6968    pub fn read_rows(
6969        &self,
6970        part: usize,
6971        columns: &[usize],
6972        positions: &[u32],
6973        whole: bool,
6974    ) -> Result<Chunk> {
6975        self.read_impl(part, columns, whole, Some(positions))
6976    }
6977
6978    /// Whether an exact global-code membership index proves that the stripe holding a part cannot
6979    /// contain any of the sorted candidate codes.
6980    ///
6981    /// # Errors
6982    ///
6983    /// If the part, column, index page, checksum, or delta stream is invalid.
6984    pub fn skips_codes(&self, part: usize, column: usize, candidates: &[u32]) -> Result<bool> {
6985        // A demoted column's later stripes hold values the dictionary never coded, so no list of
6986        // codes can prove a stripe of it holds none of a value.
6987        if self.demoted(column) {
6988            return Ok(false);
6989        }
6990        if candidates.is_empty() {
6991            return Ok(true);
6992        }
6993        if candidates.windows(2).any(|pair| pair[0] >= pair[1]) {
6994            return Err(Error::internal("native code candidates are not sorted and unique"));
6995        }
6996        let stripe = self.stripe_of(part)?;
6997        let Some(page) = stripe.memberships.get(column) else {
6998            return Ok(false);
6999        };
7000        let mut bytes = vec![0; page.length as usize];
7001        read_at(&self.file, page.offset, &mut bytes)?;
7002        if checksum(&bytes) != page.hash {
7003            return Err(invalid("membership page checksum differs"));
7004        }
7005        let codes = decode_membership(&bytes)?;
7006        let mut left = 0;
7007        let mut right = 0;
7008        while left < codes.len() && right < candidates.len() {
7009            match codes[left].cmp(&candidates[right]) {
7010                Ordering::Less => left += 1,
7011                Ordering::Greater => right += 1,
7012                Ordering::Equal => return Ok(false),
7013            }
7014        }
7015        Ok(true)
7016    }
7017
7018    fn stripe_of(&self, part: usize) -> Result<&Stripe> {
7019        let place = self.places.get(part).ok_or_else(|| invalid("part index out of range"))?;
7020        self.table
7021            .stripes
7022            .get(place.stripe as usize)
7023            .ok_or_else(|| invalid("stripe index out of range"))
7024    }
7025
7026    /// The page index of one column of one stripe, and its page when the caller wants all of it.
7027    ///
7028    /// A scan hands parts out in order, so every worker on a column crosses into a new stripe within
7029    /// a few parts of the others and they all want the same page at the same moment. This used to
7030    /// let all of them read it, which cost the scan as many copies of every page as it had workers.
7031    /// On the full ClickBench file a `MIN(EventDate), MAX(EventDate)` moved 3.2 GB off the disk to
7032    /// look at 400 MB of column.
7033    ///
7034    /// A worker that finds the page it wants already being read neither waits for it nor reads it
7035    /// again. It comes back with the index alone, which sends [`Reader::read_impl`] down the path
7036    /// that reads the one part it came for, a few kilobytes against a quarter of a megabyte, and it
7037    /// picks the page up from the cache on its next part. Waiting would be the other way to avoid
7038    /// the duplicate read and it is worse: the pages that matter are the wide string ones, they take
7039    /// milliseconds to copy even warm, and every other worker would be stopped for all of it.
7040    ///
7041    /// The file is never read under the lock.
7042    fn held(&self, at: usize, stripe: &Stripe, column: usize, whole: bool) -> Result<CachedColumn> {
7043        let cache =
7044            self.cache.columns.get(column).ok_or_else(|| invalid("column index out of range"))?;
7045        let mut cached = cache.lock().map_err(|_| invalid("column page cache is poisoned"))?;
7046        let known = cached.index.get(at).and_then(Clone::clone);
7047        let page = cached.pages.get(at).and_then(Option::as_ref).map(|slot| {
7048            slot.used.store(true, Atomic::Relaxed);
7049            Arc::clone(&slot.page)
7050        });
7051        if let Some(index) = known.clone() {
7052            if !whole || page.is_some() {
7053                return Ok(CachedColumn { stripe: at, index, page });
7054            }
7055        }
7056        if cached.loading.contains(&at) {
7057            drop(cached);
7058            // The index is almost always already here, because somebody read this stripe to get
7059            // into the loading list in the first place, so this branch usually costs no read at
7060            // all and the one part read in `read_impl` is all the losing worker pays for.
7061            if let Some(index) = known {
7062                return Ok(CachedColumn { stripe: at, index, page: None });
7063            }
7064            let held = self.page_of(stripe, column, at, false, None)?;
7065            let mut cached = cache.lock().map_err(|_| invalid("column page cache is poisoned"))?;
7066            remember(&mut cached, &held);
7067            return Ok(held);
7068        }
7069        cached.loading.push(at);
7070        drop(cached);
7071
7072        let read = self.page_of(stripe, column, at, whole, known);
7073
7074        // The stripe leaves the loading list and its page enters the cache under one lock. Doing
7075        // them separately would leave a moment where another worker sees neither and reads the
7076        // page a second time, which is the whole thing this is here to stop.
7077        let mut cached = cache.lock().map_err(|_| invalid("column page cache is poisoned"))?;
7078        if let Some(position) = cached.loading.iter().position(|loading| *loading == at) {
7079            cached.loading.remove(position);
7080        }
7081        let held = read?;
7082        let taken = remember(&mut cached, &held);
7083        let first = taken.is_some()
7084            && cached.seen.get_mut(at).is_some_and(|seen| !std::mem::replace(seen, true));
7085        if first {
7086            let floor = self.cache.kept.load(Atomic::Relaxed).max(1);
7087            cached.passing.push_back(at);
7088            while cached.passing.len() > floor {
7089                let Some(old) = cached.passing.pop_front() else { break };
7090                if let Some(slot) = cached.pages.get_mut(old) {
7091                    *slot = None;
7092                }
7093            }
7094            return Ok(held);
7095        }
7096        drop(cached);
7097        if let Some((bytes, used)) = taken {
7098            self.cache.held[column].fetch_add(1, Atomic::Relaxed);
7099            self.pool.admit(Held {
7100                shelf: Arc::downgrade(&self.cache),
7101                column,
7102                stripe: at,
7103                bytes,
7104                used,
7105            });
7106        }
7107        Ok(held)
7108    }
7109
7110    /// Reads one stripe's index for a column, and its page when the caller wants all of it.
7111    ///
7112    /// `known` is the index when the reader has already read it, which after the first worker
7113    /// through a stripe it always has, because [`remember`] keeps every index for the life of the
7114    /// reader. Without that a scan reads the index again on every part that misses the page cache.
7115    fn page_of(
7116        &self,
7117        stripe: &Stripe,
7118        column: usize,
7119        at: usize,
7120        whole: bool,
7121        known: Option<Arc<Vec<PartSpan>>>,
7122    ) -> Result<CachedColumn> {
7123        let index = match known {
7124            Some(index) => index,
7125            None => {
7126                self.indexes.fetch_add(1, Atomic::Relaxed);
7127                Arc::new(read_index(&self.file, stripe, column)?)
7128            }
7129        };
7130        let page = if whole {
7131            self.pages.fetch_add(1, Atomic::Relaxed);
7132            let span = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
7133            let mut bytes = vec![0; span.length as usize];
7134            read_at(&self.file, span.offset, &mut bytes)?;
7135            let checked = index.iter().map(|_| AtomicBool::new(false)).collect();
7136            Some(Arc::new(HeldPage { bytes, checked }))
7137        } else {
7138            None
7139        };
7140        Ok(CachedColumn { stripe: at, index, page })
7141    }
7142
7143    fn read_impl(
7144        &self,
7145        at: usize,
7146        columns: &[usize],
7147        whole: bool,
7148        positions: Option<&[u32]>,
7149    ) -> Result<Chunk> {
7150        let place = *self.places.get(at).ok_or_else(|| invalid("part index out of range"))?;
7151        let index = place.stripe as usize;
7152        let stripe =
7153            self.table.stripes.get(index).ok_or_else(|| invalid("stripe index out of range"))?;
7154        let rows = place.rows as usize;
7155        let mut picked = Vec::with_capacity(columns.len());
7156        for &column in columns {
7157            let field = self
7158                .table
7159                .fields
7160                .get(column)
7161                .ok_or_else(|| invalid("column index out of range"))?;
7162            let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
7163            let held = self.held(index, stripe, column, whole)?;
7164            let span = *held
7165                .index
7166                .get(place.part as usize)
7167                .ok_or_else(|| invalid("part index out of range"))?;
7168            let owned;
7169            let bytes = match &held.page {
7170                Some(held) => held.part(place.part as usize, span),
7171                None => {
7172                    let offset = page
7173                        .offset
7174                        .checked_add(span.start as u64)
7175                        .ok_or_else(|| invalid("part range overflow"))?;
7176                    let mut bytes = vec![0; span.length];
7177                    read_at(&self.file, offset, &mut bytes)?;
7178                    owned = bytes;
7179                    verify_part(&owned, span).map(|()| owned.as_slice())
7180                }
7181            }
7182            .map_err(|error| {
7183                invalid(&format!(
7184                    "{}, column {column} part {} of the page at {}",
7185                    error.message(),
7186                    place.part,
7187                    page.offset,
7188                ))
7189            })?;
7190            let dictionary = self.dictionary(column)?;
7191            // Held as a page, because a column that came out of a file is handed out more than
7192            // once. A group by clones its key columns out of the chunk so the keys outlive it, a
7193            // projection of a bare column name does the same, and a cut of a flat run copies unless
7194            // the run is a page. One `Arc` per column per part buys all of those, and it moves the
7195            // run into the `Arc` without touching a value.
7196            let mut vector = match positions {
7197                None => decode(&field.ty, rows, bytes, dictionary)?,
7198                Some(positions) => decode_at(&field.ty, rows, bytes, dictionary, positions)?,
7199            };
7200            // A demoted column's codes are not the column's codes, only the codes of the stripes
7201            // written before the demotion, so they are not handed out as if they were. See
7202            // [`DEMOTED`].
7203            if self.demoted(column) && vector.stable_dictionary_parts().is_some() {
7204                vector = vector.flatten()?;
7205            }
7206            picked.push(vector.into_pages());
7207        }
7208        Chunk::with_rows(picked, positions.map_or(rows, <[u32]>::len))
7209    }
7210
7211    /// Whether persisted statistics prove that a part cannot match the predicates.
7212    ///
7213    /// Three of them, asked cheapest first.
7214    ///
7215    /// The stripe's bounds are in memory already, so they are free, and they are also the coarsest:
7216    /// every part of a stripe gets the same answer and a scan that skips one part that way skips all
7217    /// sixty four. Then the part's own bounds, which are a read of one page per column per stripe
7218    /// and are sixty four times finer. Then the sieves, which are per part and answer equality, the
7219    /// test bounds are worst at: a column of identifiers has every stripe and nearly every part
7220    /// covering the whole of its type, so bounds keep them all and the sieve keeps the ones that
7221    /// really hold the value.
7222    ///
7223    /// The middle one is what an ordered comparison on a column the rows are not sorted by needs. On
7224    /// ClickBench 24 the stripe bounds leave eight stripes of sixteen alive, which is half the file,
7225    /// and the part bounds leave thirty parts of nine hundred and seventy four.
7226    #[must_use]
7227    pub fn skips(&self, part: usize, probes: &[Probe]) -> bool {
7228        let Some(place) = self.places.get(part).copied() else { return false };
7229        let Some(stripe) = self.table.stripes.get(place.stripe as usize) else { return false };
7230        if stripe.zone.skips(probes) {
7231            return true;
7232        }
7233        probes.iter().any(|probe| self.outside(place, probe) || self.sifted(place, probe))
7234    }
7235
7236    /// Whether the bounds of one part rule out one probe.
7237    ///
7238    /// The part's own two ends, which are narrower than the stripe's and cost a page read the first
7239    /// time this is asked about a column. A column with no page here answers `false`, which is the
7240    /// answer a caller got before there were any.
7241    fn outside(&self, place: Place, probe: &Probe) -> bool {
7242        match self.stripe_part_ranges(place.stripe as usize, probe.column) {
7243            Some(ranges) => ranges
7244                .get(place.part as usize)
7245                .is_some_and(|range| range.excludes(probe.op, &probe.value)),
7246            None => false,
7247        }
7248    }
7249
7250    /// The per part ranges of one stripe of one column, read once and kept.
7251    ///
7252    /// `None` when the column has no page in that stripe and when the page is damaged, on the same
7253    /// reasoning as the sieves: this is an index over data that is still there, so a caller that
7254    /// cannot read one reads the rows and gets the right answer slowly.
7255    fn stripe_part_ranges(&self, stripe: usize, column: usize) -> Option<&[Range]> {
7256        let slot = self.part_ranges.get(column)?.get(stripe)?;
7257        if let Some(held) = slot.get() {
7258            return Some(held);
7259        }
7260        let page = self.table.stripes.get(stripe)?.part_ranges.get(column)?;
7261        let mut bytes = vec![0; page.length as usize];
7262        read_at(&self.file, page.offset, &mut bytes).ok()?;
7263        if checksum(&bytes) != page.hash {
7264            return None;
7265        }
7266        let ranges = Arc::new(decode_part_ranges(&bytes).ok()?);
7267        let _ = slot.set(ranges);
7268        slot.get().map(|held| held.as_slice())
7269    }
7270
7271    /// Whether persisted statistics prove that every row of a part matches the predicates.
7272    ///
7273    /// Only the bounds. The sieves say nothing here, because a sieve that holds a value is a sieve
7274    /// that may be holding somebody else's hash, so it can rule a part out and can never wave one
7275    /// through.
7276    ///
7277    /// The stripe first and the part after it, the same two steps and in the same order as
7278    /// [`Self::skips`]. The stripe's bounds are in memory already and its null count covers sixty
7279    /// four parts rather than one, so a stripe that answers is an answer for nothing, and the part's
7280    /// own bounds are only read for the probes it could not settle. Both directions are safe: a
7281    /// stretch where everything passes contains no narrower stretch where something fails, and a
7282    /// stripe with no nulls has no nulls in any of its parts.
7283    ///
7284    /// A string end a part recorded is cut down to its first few bytes, so a part's stretch can be
7285    /// wider than its rows really are as well. That is the same safe direction for the same reason,
7286    /// and it is why this asks the two ends rather than anything `exact` says.
7287    #[must_use]
7288    pub fn certain(&self, part: usize, probes: &[Probe]) -> bool {
7289        let Some(place) = self.places.get(part).copied() else { return false };
7290        let Some(stripe) = self.table.stripes.get(place.stripe as usize) else { return false };
7291        if stripe.zone.certain(probes) {
7292            return true;
7293        }
7294        probes
7295            .iter()
7296            .all(|probe| stripe.zone.certain(slice::from_ref(probe)) || self.inside(place, probe))
7297    }
7298
7299    /// Whether one part's own two ends prove that every row of it passes `probe`.
7300    ///
7301    /// The mirror of [`Self::outside`], reading the same page. `false` for a part whose stripe wrote
7302    /// no range page, which is a stripe of one part, because there the stripe's own bounds are the
7303    /// part's and the caller has already asked them.
7304    fn inside(&self, place: Place, probe: &Probe) -> bool {
7305        match self.stripe_part_ranges(place.stripe as usize, probe.column) {
7306            Some(ranges) => ranges
7307                .get(place.part as usize)
7308                .is_some_and(|range| range.certain(probe.op, &probe.value)),
7309            None => false,
7310        }
7311    }
7312
7313    /// Whether the bounds of one stripe prove that none of its parts can match the predicates.
7314    ///
7315    /// The cheap half of [`Self::skips`], asked about a whole stripe at once. The bounds live in the
7316    /// directory and are already in memory, so this answers without touching the file, and that is
7317    /// the reason it is worth having on its own: a caller that wants to know roughly where the work
7318    /// is before it starts any workers can ask this about sixteen stripes for nothing, where asking
7319    /// [`Self::skips`] about nine hundred parts would read and decode a sieve page per stripe first.
7320    ///
7321    /// It keeps stripes that [`Self::skips`] would rule out part by part, which is the right way for
7322    /// it to be wrong: the parts are still checked when they are read.
7323    #[must_use]
7324    pub fn stripe_skips(&self, stripe: usize, probes: &[Probe]) -> bool {
7325        self.table.stripes.get(stripe).is_some_and(|held| held.zone.skips(probes))
7326    }
7327
7328    /// Whether the sieve of one part rules out one probe.
7329    ///
7330    /// Only equality. An ordered comparison is what the bounds are for and a sieve says nothing
7331    /// about it, and a read that cannot answer keeps the part, which is the answer a caller with no
7332    /// sieve gets anyway.
7333    fn sifted(&self, place: Place, probe: &Probe) -> bool {
7334        if probe.op != Op::Equal {
7335            return false;
7336        }
7337        match self.stripe_sieves(place.stripe as usize, probe.column) {
7338            Some(sieves) => sieves
7339                .get(place.part as usize)
7340                .and_then(Option::as_ref)
7341                .is_some_and(|sieve| sieve.excludes(&probe.value)),
7342            None => false,
7343        }
7344    }
7345
7346    /// The sieves of one stripe of one column, read once and kept.
7347    ///
7348    /// `None` when the column has no sieves in that stripe, when the page is damaged, and when the
7349    /// bytes are not a page this version can read. A sieve is an index over data that is still there
7350    /// and a caller that cannot read one reads the rows, so this is the one place in the file where
7351    /// a bad checksum is a slow query rather than an error.
7352    fn stripe_sieves(&self, stripe: usize, column: usize) -> Option<&[Option<Sieve>]> {
7353        let slot = self.sieves.get(column)?.get(stripe)?;
7354        if let Some(held) = slot.get() {
7355            return Some(held);
7356        }
7357        let page = self.table.stripes.get(stripe)?.sieves.get(column)?;
7358        let mut bytes = vec![0; page.length as usize];
7359        read_at(&self.file, page.offset, &mut bytes).ok()?;
7360        if checksum(&bytes) != page.hash {
7361            return None;
7362        }
7363        let sieves = Arc::new(decode_sieves(&bytes).ok()?);
7364        let _ = slot.set(sieves);
7365        slot.get().map(|held| held.as_slice())
7366    }
7367}
7368
7369/// The value sitting at one position of a dictionary's sorted order.
7370fn text_at_rank(dictionary: &Vector, rank: usize) -> Result<Value> {
7371    let code = dictionary.code_at_rank(rank)? as usize;
7372    if dictionary.logical_type() == &LogicalType::Blob {
7373        let bytes = dictionary
7374            .try_bytes_at(code)?
7375            .ok_or_else(|| invalid("global dictionary order names a code it does not have"))?;
7376        return Ok(Value::Blob(bytes.to_vec()));
7377    }
7378    let text = dictionary
7379        .try_text_at(code)?
7380        .ok_or_else(|| invalid("global dictionary order names a code it does not have"))?;
7381    Ok(Value::Varchar(text.into()))
7382}
7383
7384/// Reads one span of a file at an offset, without moving a cursor anybody else can see.
7385///
7386/// Every reader of a table shares one [`File`] behind an [`Arc`], and a grouped aggregate reads its
7387/// pages from several threads at once, so this has to be positional. Seeking and then reading is
7388/// two calls with a gap in the middle, and in that gap another thread's seek lands and the read
7389/// comes back with somebody else's bytes.
7390///
7391/// The writer reads back through here too, out of the `rudb_io` file it writes through, which is
7392/// why this takes anything [`Positional`] rather than a [`File`].
7393fn read_at<F: Positional + ?Sized>(file: &F, offset: u64, bytes: &mut [u8]) -> Result<()> {
7394    file.fill_at(offset, bytes)
7395}
7396
7397/// Something a span of bytes can be read out of by offset.
7398///
7399/// There are two of these. The reader holds a `std::fs::File`, because it shares it between its
7400/// threads behind an [`Arc`] and every read it makes is on the hot path of a scan. The writer holds
7401/// an `rudb_io::File`, because everything it does to the file has to be something the simulated
7402/// filesystem can stop and crash. The few helpers both of them use, [`read_index`] and the choice
7403/// of committed slot, are written once over this rather than once for each.
7404trait Positional {
7405    /// Fills `bytes` from `offset`, or fails if the file ends first.
7406    ///
7407    /// Both kinds can come back short, so both loop. A read of zero bytes before the span is filled
7408    /// means the file stops earlier than the directory said it does.
7409    fn fill_at(&self, offset: u64, bytes: &mut [u8]) -> Result<()>;
7410}
7411
7412impl<T: Positional + ?Sized> Positional for &T {
7413    fn fill_at(&self, offset: u64, bytes: &mut [u8]) -> Result<()> {
7414        (**self).fill_at(offset, bytes)
7415    }
7416}
7417
7418impl<T: Positional + ?Sized> Positional for Arc<T> {
7419    fn fill_at(&self, offset: u64, bytes: &mut [u8]) -> Result<()> {
7420        (**self).fill_at(offset, bytes)
7421    }
7422}
7423
7424impl<T: Positional + ?Sized> Positional for Box<T> {
7425    fn fill_at(&self, offset: u64, bytes: &mut [u8]) -> Result<()> {
7426        (**self).fill_at(offset, bytes)
7427    }
7428}
7429
7430impl Positional for dyn rudb_io::File + '_ {
7431    fn fill_at(&self, mut offset: u64, mut bytes: &mut [u8]) -> Result<()> {
7432        while !bytes.is_empty() {
7433            let read = self.read_at(offset, bytes)?;
7434            if read == 0 {
7435                return Err(invalid("column page ends before its declared length"));
7436            }
7437            offset += read as u64;
7438            bytes = &mut bytes[read..];
7439        }
7440        Ok(())
7441    }
7442}
7443
7444impl Positional for File {
7445    #[cfg(unix)]
7446    fn fill_at(&self, mut offset: u64, mut bytes: &mut [u8]) -> Result<()> {
7447        use std::os::unix::fs::FileExt;
7448        while !bytes.is_empty() {
7449            let read = self.read_at(bytes, offset).map_err(io)?;
7450            if read == 0 {
7451                return Err(invalid("column page ends before its declared length"));
7452            }
7453            offset += read as u64;
7454            bytes = &mut bytes[read..];
7455        }
7456        Ok(())
7457    }
7458
7459    /// The same read, on the call Windows spells differently.
7460    ///
7461    /// `seek_read` is one `ReadFile` carrying the offset with it, so two of them cannot interleave
7462    /// the way a seek and a read can. It does leave the shared cursor somewhere afterwards, which is
7463    /// why nothing in this file may read that cursor.
7464    #[cfg(windows)]
7465    fn fill_at(&self, mut offset: u64, mut bytes: &mut [u8]) -> Result<()> {
7466        use std::os::windows::fs::FileExt;
7467        while !bytes.is_empty() {
7468            let read = self.seek_read(bytes, offset).map_err(io)?;
7469            if read == 0 {
7470                return Err(invalid("column page ends before its declared length"));
7471            }
7472            offset += read as u64;
7473            bytes = &mut bytes[read..];
7474        }
7475        Ok(())
7476    }
7477
7478    /// Somewhere that is neither, where the cursor is all there is.
7479    ///
7480    /// This one does race, and there is no way to write it so it does not. Nothing we build for
7481    /// runs here, so it exists to keep the crate compiling rather than to be correct under threads.
7482    #[cfg(not(any(unix, windows)))]
7483    fn fill_at(&self, offset: u64, bytes: &mut [u8]) -> Result<()> {
7484        use std::io::{Read, Seek, SeekFrom};
7485        let mut file = self.try_clone().map_err(io)?;
7486        file.seek(SeekFrom::Start(offset)).map_err(io)?;
7487        file.read_exact(bytes).map_err(io)
7488    }
7489}
7490
7491/// Overwrites one span of a file in place, which is how the tests damage a file on purpose.
7492///
7493/// The writer does not come through here. It writes through `rudb_io`, and this is a
7494/// `std::fs::File` opened by a test beside it.
7495#[cfg(test)]
7496fn write_at(file: &File, offset: u64, bytes: &[u8]) -> Result<()> {
7497    use std::io::{Seek, SeekFrom, Write};
7498    let mut file = file;
7499    file.seek(SeekFrom::Start(offset)).map_err(io)?;
7500    file.write_all(bytes).map_err(io)
7501}
7502
7503/// What a column type is called in the directory.
7504///
7505/// A tag is a number in a file somebody else wrote, so a tag that has been used is used forever and
7506/// the only thing that may happen to this list is that it grows. 1 to 13 are the tags the format
7507/// had when it could store thirteen types, and 14 to 27 are the rest, in the order they were added
7508/// rather than in an order that means anything.
7509fn type_tag(ty: &LogicalType) -> Result<u8> {
7510    match ty {
7511        LogicalType::SmallInt => Ok(1),
7512        LogicalType::Integer => Ok(2),
7513        LogicalType::BigInt => Ok(3),
7514        LogicalType::Varchar => Ok(4),
7515        LogicalType::Date => Ok(5),
7516        LogicalType::Timestamp => Ok(6),
7517        LogicalType::Boolean => Ok(7),
7518        LogicalType::TinyInt => Ok(8),
7519        LogicalType::UTinyInt => Ok(9),
7520        LogicalType::USmallInt => Ok(10),
7521        LogicalType::UInteger => Ok(11),
7522        LogicalType::UBigInt => Ok(12),
7523        LogicalType::Decimal { .. } => Ok(13),
7524        LogicalType::Float => Ok(14),
7525        LogicalType::Double => Ok(15),
7526        LogicalType::HugeInt => Ok(16),
7527        LogicalType::UHugeInt => Ok(17),
7528        LogicalType::Time => Ok(18),
7529        LogicalType::TimeTz => Ok(19),
7530        LogicalType::TimestampTz => Ok(20),
7531        LogicalType::Interval => Ok(21),
7532        LogicalType::Uuid => Ok(22),
7533        LogicalType::Blob => Ok(23),
7534        LogicalType::Bit => Ok(24),
7535        LogicalType::TimestampS => Ok(25),
7536        LogicalType::TimestampMs => Ok(26),
7537        LogicalType::TimestampNs => Ok(27),
7538        _ => Err(Error::not_implemented(format!("native storage for {ty}"))),
7539    }
7540}
7541
7542/// The tag of a column type, and the parameters of the ones that have any.
7543///
7544/// Only `DECIMAL` has parameters today. Width and scale go after the tag rather than into it
7545/// because they are what says how wide a value is on disk, and a reader that guessed would read the
7546/// wrong number of bytes per row rather than the wrong number of digits.
7547fn put_type(out: &mut Vec<u8>, ty: &LogicalType) -> Result<()> {
7548    out.push(type_tag(ty)?);
7549    if let LogicalType::Decimal { width, scale } = ty {
7550        out.push(*width);
7551        out.push(*scale);
7552    }
7553    Ok(())
7554}
7555
7556/// The other half of [`put_type`], reading the parameters the tag says are there.
7557fn read_type(cur: &mut Cursor<'_>) -> Result<LogicalType> {
7558    let tag = cur.u8()?;
7559    if tag == 13 {
7560        let width = cur.u8()?;
7561        let scale = cur.u8()?;
7562        return LogicalType::decimal(width, scale)
7563            .map_err(|_| invalid("decimal column width and scale are not a decimal"));
7564    }
7565    tag_type(tag)
7566}
7567
7568fn tag_type(tag: u8) -> Result<LogicalType> {
7569    match tag {
7570        1 => Ok(LogicalType::SmallInt),
7571        2 => Ok(LogicalType::Integer),
7572        3 => Ok(LogicalType::BigInt),
7573        4 => Ok(LogicalType::Varchar),
7574        5 => Ok(LogicalType::Date),
7575        6 => Ok(LogicalType::Timestamp),
7576        7 => Ok(LogicalType::Boolean),
7577        8 => Ok(LogicalType::TinyInt),
7578        9 => Ok(LogicalType::UTinyInt),
7579        10 => Ok(LogicalType::USmallInt),
7580        11 => Ok(LogicalType::UInteger),
7581        12 => Ok(LogicalType::UBigInt),
7582        14 => Ok(LogicalType::Float),
7583        15 => Ok(LogicalType::Double),
7584        16 => Ok(LogicalType::HugeInt),
7585        17 => Ok(LogicalType::UHugeInt),
7586        18 => Ok(LogicalType::Time),
7587        19 => Ok(LogicalType::TimeTz),
7588        20 => Ok(LogicalType::TimestampTz),
7589        21 => Ok(LogicalType::Interval),
7590        22 => Ok(LogicalType::Uuid),
7591        23 => Ok(LogicalType::Blob),
7592        24 => Ok(LogicalType::Bit),
7593        25 => Ok(LogicalType::TimestampS),
7594        26 => Ok(LogicalType::TimestampMs),
7595        27 => Ok(LogicalType::TimestampNs),
7596        _ => Err(invalid("column type tag is unknown")),
7597    }
7598}
7599
7600fn put_u16(out: &mut Vec<u8>, value: u16) {
7601    out.extend_from_slice(&value.to_le_bytes());
7602}
7603fn put_u32(out: &mut Vec<u8>, value: u32) {
7604    out.extend_from_slice(&value.to_le_bytes());
7605}
7606fn put_u64(out: &mut Vec<u8>, value: u64) {
7607    out.extend_from_slice(&value.to_le_bytes());
7608}
7609fn put_var_u64(out: &mut Vec<u8>, mut value: u64) {
7610    while value >= 0x80 {
7611        out.push((value as u8 & 0x7f) | 0x80);
7612        value >>= 7;
7613    }
7614    out.push(value as u8);
7615}
7616
7617fn frequency_order(left: FrequencyValue, right: FrequencyValue) -> Ordering {
7618    match (left, right) {
7619        (FrequencyValue::Null, FrequencyValue::Null) => Ordering::Equal,
7620        (FrequencyValue::Null, _) => Ordering::Less,
7621        (_, FrequencyValue::Null) => Ordering::Greater,
7622        (FrequencyValue::Integer(left), FrequencyValue::Integer(right)) => left.cmp(&right),
7623        (FrequencyValue::Code(left), FrequencyValue::Code(right)) => left.cmp(&right),
7624        (FrequencyValue::Integer(_), FrequencyValue::Code(_)) => Ordering::Less,
7625        (FrequencyValue::Code(_), FrequencyValue::Integer(_)) => Ordering::Greater,
7626    }
7627}
7628
7629/// Leaves the [`FREQUENCY_ENTRIES`] commonest entries in order and says what the next one counted.
7630///
7631/// There is one entry a distinct value, so on `URL` this is handed two and a quarter million of
7632/// them and keeps five hundred and twelve. Sorting all of them to throw almost all of them away is
7633/// the whole of what counting a dictionary column used to cost, 2.13 seconds of it on `URL` at eight
7634/// million rows against 11.93 for compressing the same column's values.
7635///
7636/// Partitioning answers both questions instead. It puts the five hundred and thirteenth entry where
7637/// it belongs and everything commoner in front of it, which is the entries to keep and the count to
7638/// report as the largest one omitted, and then only the part that survives is sorted. The order that
7639/// comes out is the order the sort gave, because the tie break makes the comparison total: two
7640/// entries never hold the same value.
7641fn keep_most_frequent(entries: &mut Vec<FrequencyEntry>) -> u64 {
7642    let order = |left: &FrequencyEntry, right: &FrequencyEntry| {
7643        right.count.cmp(&left.count).then_with(|| frequency_order(left.value, right.value))
7644    };
7645    let omitted_max = if entries.len() > FREQUENCY_ENTRIES {
7646        let (_, next, _) = entries.select_nth_unstable_by(FREQUENCY_ENTRIES, order);
7647        let omitted_max = next.count;
7648        entries.truncate(FREQUENCY_ENTRIES);
7649        omitted_max
7650    } else {
7651        0
7652    };
7653    entries.sort_unstable_by(order);
7654    omitted_max
7655}
7656
7657fn code_frequency(
7658    dictionary: &GlobalDictionary,
7659    flat: &[u8],
7660    bases: &[u64],
7661) -> Result<(FrequencySummary, Vec<Option<Vec<u8>>>)> {
7662    let mut entries = dictionary
7663        .counts
7664        .iter()
7665        .enumerate()
7666        .filter(|(_, count)| **count != 0)
7667        .map(|(code, &count)| FrequencyEntry { value: FrequencyValue::Code(code as u32), count })
7668        .collect::<Vec<_>>();
7669    if dictionary.nulls != 0 {
7670        entries.push(FrequencyEntry { value: FrequencyValue::Null, count: dictionary.nulls });
7671    }
7672    let omitted_max = keep_most_frequent(&mut entries);
7673    let mut spans = Vec::with_capacity(entries.len());
7674    let mut text_bytes = 0_usize;
7675    for entry in &entries {
7676        let span = match entry.value {
7677            FrequencyValue::Code(code) => {
7678                let span = GlobalDictionary::value_span(&dictionary.ends, bases, code as usize);
7679                let bytes = flat
7680                    .get(span.0..span.1)
7681                    .ok_or_else(|| invalid("a frequency code is outside its dictionary"))?;
7682                text_bytes = text_bytes.saturating_add(bytes.len());
7683                Some(span)
7684            }
7685            FrequencyValue::Null | FrequencyValue::Integer(_) => None,
7686        };
7687        spans.push(span);
7688    }
7689    let texts = if text_bytes > FREQUENCY_TEXT_BUDGET {
7690        Vec::new()
7691    } else {
7692        spans.into_iter().map(|span| span.map(|(from, to)| flat[from..to].to_vec())).collect()
7693    };
7694    Ok((
7695        FrequencySummary {
7696            entries,
7697            omitted_max,
7698            ordinals: Vec::new(),
7699            ordinal_entries: Vec::new(),
7700        },
7701        texts,
7702    ))
7703}
7704
7705fn encode_directory(table: &Table) -> Result<Vec<u8>> {
7706    let mut out = DIRECTORY.to_vec();
7707    let name = table.name.as_bytes();
7708    put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("table name too long"))?);
7709    out.extend_from_slice(name);
7710    put_u16(&mut out, u16::try_from(table.fields.len()).map_err(|_| invalid("too many columns"))?);
7711    for field in &table.fields {
7712        let name = field.name.as_bytes();
7713        put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("column name too long"))?);
7714        out.extend_from_slice(name);
7715        put_type(&mut out, &field.ty)?;
7716        out.push(u8::from(field.not_null));
7717    }
7718    for (field, dictionary) in table.fields.iter().zip(&table.dictionaries) {
7719        match dictionary {
7720            None => out.push(0),
7721            Some(page) => {
7722                out.push(dictionary_tag(&field.ty));
7723                put_u64(&mut out, page.offset);
7724                put_u32(&mut out, page.length);
7725                put_u64(&mut out, page.hash);
7726            }
7727        }
7728    }
7729    for distinct in &table.distincts {
7730        match distinct {
7731            None => out.push(0),
7732            Some(count) => {
7733                out.push(1);
7734                put_u64(&mut out, *count);
7735            }
7736        }
7737    }
7738    put_u64(&mut out, u64::try_from(table.rows).map_err(|_| invalid("row count overflow"))?);
7739    put_u32(&mut out, u32::try_from(table.stripes.len()).map_err(|_| invalid("too many stripes"))?);
7740    for stripe in &table.stripes {
7741        put_u32(
7742            &mut out,
7743            u32::try_from(stripe.parts.len()).map_err(|_| invalid("too many parts in a stripe"))?,
7744        );
7745        for &rows in &stripe.parts {
7746            put_u32(&mut out, rows);
7747        }
7748        put_u64(&mut out, stripe.index.offset);
7749        put_u32(&mut out, stripe.index.length);
7750        for page in &stripe.pages {
7751            put_u64(&mut out, page.offset);
7752            put_u32(&mut out, page.length);
7753        }
7754        // A membership index says which of a dictionary's codes a part holds, so a column the writer
7755        // decided against giving a dictionary has nothing for it to be about and writes none. Every
7756        // file written before that decision existed has a dictionary on every varchar column, so
7757        // this reads those files byte for byte the way it always did.
7758        for (column, ((field, dictionary), membership)) in
7759            table.fields.iter().zip(&table.dictionaries).zip(stripe.memberships.slots()).enumerate()
7760        {
7761            if !coded_type(&field.ty) || dictionary.is_none() {
7762                continue;
7763            }
7764            let page = match membership {
7765                Some(page) => page,
7766                None if table.demoted.get(column).copied().unwrap_or(false) => {
7767                    Page { offset: HEADER, length: 0, hash: 0 }
7768                }
7769                None => return Err(invalid("string page has no code membership index")),
7770            };
7771            put_u64(&mut out, page.offset);
7772            put_u32(&mut out, page.length);
7773            put_u64(&mut out, page.hash);
7774        }
7775        for sieve in stripe.sieves.slots() {
7776            match sieve {
7777                None => out.push(0),
7778                Some(page) => {
7779                    out.push(1);
7780                    put_u64(&mut out, page.offset);
7781                    put_u32(&mut out, page.length);
7782                    put_u64(&mut out, page.hash);
7783                }
7784            }
7785        }
7786        for held in stripe.part_ranges.slots() {
7787            match held {
7788                None => out.push(0),
7789                Some(page) => {
7790                    out.push(1);
7791                    put_u64(&mut out, page.offset);
7792                    put_u32(&mut out, page.length);
7793                    put_u64(&mut out, page.hash);
7794                }
7795            }
7796        }
7797        for range in stripe.zone.columns() {
7798            put_bound(&mut out, range.low.as_ref())?;
7799            put_bound(&mut out, range.high.as_ref())?;
7800            put_u32(
7801                &mut out,
7802                u32::try_from(range.nulls).map_err(|_| invalid("null count overflow"))?,
7803            );
7804            out.push(u8::from(range.exact));
7805            match range.sum {
7806                None => out.push(0),
7807                Some(total) => {
7808                    out.push(1);
7809                    out.extend_from_slice(&total.to_le_bytes());
7810                }
7811            }
7812        }
7813    }
7814    out.extend_from_slice(FREQUENCIES);
7815    put_u16(
7816        &mut out,
7817        u16::try_from(table.frequencies.len())
7818            .map_err(|_| invalid("too many frequency columns"))?,
7819    );
7820    for summary in &table.frequencies {
7821        let summary = match summary {
7822            None => {
7823                out.push(0);
7824                continue;
7825            }
7826            Some(Frequencies::Held(summary)) => summary,
7827            // Only a reader leaves a synopsis in the file, and nothing writes a reader's table back.
7828            Some(Frequencies::Stored { .. }) => {
7829                return Err(invalid("a synopsis left in the file cannot be written back"));
7830            }
7831        };
7832        out.push(1);
7833        put_u64(&mut out, summary.omitted_max);
7834        put_u32(
7835            &mut out,
7836            u32::try_from(summary.entries.len())
7837                .map_err(|_| invalid("too many frequency entries"))?,
7838        );
7839        for entry in &summary.entries {
7840            match entry.value {
7841                FrequencyValue::Null => out.push(0),
7842                FrequencyValue::Integer(value) => {
7843                    out.push(1);
7844                    out.extend_from_slice(&value.to_le_bytes());
7845                }
7846                FrequencyValue::Code(value) => {
7847                    out.push(2);
7848                    put_u32(&mut out, value);
7849                }
7850            }
7851            put_u64(&mut out, entry.count);
7852        }
7853        put_u32(
7854            &mut out,
7855            u32::try_from(summary.ordinals.len())
7856                .map_err(|_| invalid("too many frequency ordinals"))?,
7857        );
7858        let mut previous = 0_u64;
7859        for (at, &ordinal) in summary.ordinals.iter().enumerate() {
7860            let delta = if at == 0 {
7861                ordinal
7862            } else {
7863                ordinal
7864                    .checked_sub(previous)
7865                    .ok_or_else(|| invalid("frequency ordinals are not ordered"))?
7866            };
7867            if at != 0 && delta == 0 {
7868                return Err(invalid("frequency ordinals are not unique"));
7869            }
7870            put_var_u64(&mut out, delta);
7871            previous = ordinal;
7872        }
7873        if summary.ordinal_entries.len() != summary.ordinals.len() {
7874            return Err(invalid("frequency ordinal values have a different length"));
7875        }
7876        for &entry in &summary.ordinal_entries {
7877            if entry as usize >= summary.entries.len() {
7878                return Err(invalid("frequency ordinal value is outside its entries"));
7879            }
7880            put_u16(&mut out, entry);
7881        }
7882    }
7883    if !table.pair_frequencies.is_empty() {
7884        out.extend_from_slice(PAIR_FREQUENCIES);
7885        put_u16(
7886            &mut out,
7887            u16::try_from(table.pair_frequencies.len())
7888                .map_err(|_| invalid("too many pair frequency summaries"))?,
7889        );
7890        for summary in &table.pair_frequencies {
7891            put_u16(&mut out, summary.first);
7892            put_u16(&mut out, summary.second);
7893            put_u64(&mut out, summary.omitted_max);
7894            put_u16(
7895                &mut out,
7896                u16::try_from(summary.entries.len())
7897                    .map_err(|_| invalid("too many pair frequency entries"))?,
7898            );
7899            for entry in &summary.entries {
7900                put_u16(&mut out, entry.first_entry);
7901                match entry.second {
7902                    None => out.push(0),
7903                    Some(code) => {
7904                        out.push(1);
7905                        put_u32(&mut out, code);
7906                    }
7907                }
7908                put_u64(&mut out, entry.count);
7909            }
7910        }
7911    }
7912    let text_columns = table.frequency_texts.iter().filter(|texts| !texts.is_empty()).count();
7913    if text_columns != 0 {
7914        out.extend_from_slice(FREQUENCY_TEXTS);
7915        put_u16(
7916            &mut out,
7917            u16::try_from(text_columns)
7918                .map_err(|_| invalid("too many string frequency columns"))?,
7919        );
7920        for (column, texts) in table.frequency_texts.iter().enumerate() {
7921            if texts.is_empty() {
7922                continue;
7923            }
7924            put_u16(
7925                &mut out,
7926                u16::try_from(column).map_err(|_| invalid("frequency text column overflows"))?,
7927            );
7928            put_u16(
7929                &mut out,
7930                u16::try_from(texts.len())
7931                    .map_err(|_| invalid("too many frequency text entries"))?,
7932            );
7933            for text in texts {
7934                match text {
7935                    None => out.push(0),
7936                    Some(text) => {
7937                        out.push(1);
7938                        put_u32(
7939                            &mut out,
7940                            u32::try_from(text.len())
7941                                .map_err(|_| invalid("frequency text is too long"))?,
7942                        );
7943                        out.extend_from_slice(text);
7944                    }
7945                }
7946            }
7947        }
7948    }
7949    if let Some(summary) = &table.host_groups {
7950        out.extend_from_slice(HOST_GROUPS);
7951        put_u16(
7952            &mut out,
7953            u16::try_from(summary.column).map_err(|_| invalid("host column overflows"))?,
7954        );
7955        put_u64(&mut out, summary.omitted_max);
7956        put_u16(
7957            &mut out,
7958            u16::try_from(summary.entries.len()).map_err(|_| invalid("too many host groups"))?,
7959        );
7960        for entry in &summary.entries {
7961            put_u32(
7962                &mut out,
7963                u32::try_from(entry.host.len()).map_err(|_| invalid("host name is too long"))?,
7964            );
7965            out.extend_from_slice(entry.host.as_bytes());
7966            put_u64(&mut out, entry.count);
7967            out.extend_from_slice(&entry.bytes_sum.to_le_bytes());
7968            put_u32(
7969                &mut out,
7970                u32::try_from(entry.minimum.len())
7971                    .map_err(|_| invalid("host minimum is too long"))?,
7972            );
7973            out.extend_from_slice(entry.minimum.as_bytes());
7974        }
7975    }
7976    // Written only when there is a declaration, so that the common file is the same bytes it was
7977    // and the section is not a byte of zero on every table in the world that never asked for one.
7978    if let Some(clustering) = &table.clustering {
7979        out.extend_from_slice(CLUSTERING);
7980        out.push(clustering.width().tag());
7981        put_u16(
7982            &mut out,
7983            u16::try_from(clustering.columns().len())
7984                .map_err(|_| invalid("too many clustering columns"))?,
7985        );
7986        for &column in clustering.columns() {
7987            put_u16(
7988                &mut out,
7989                u16::try_from(column).map_err(|_| invalid("clustering column index overflow"))?,
7990            );
7991        }
7992    }
7993    let demoted = (0..table.fields.len())
7994        .filter(|&column| table.demoted.get(column).copied().unwrap_or(false))
7995        .collect::<Vec<_>>();
7996    if !demoted.is_empty() {
7997        out.extend_from_slice(DEMOTED);
7998        put_u16(
7999            &mut out,
8000            u16::try_from(demoted.len()).map_err(|_| invalid("too many demoted columns"))?,
8001        );
8002        for column in demoted {
8003            put_u16(
8004                &mut out,
8005                u16::try_from(column).map_err(|_| invalid("demoted column index overflow"))?,
8006            );
8007        }
8008    }
8009    // The section table, last, behind its own magic, for the same reason the frequency block is
8010    // behind its own: a reader that stops before it gets a table with no sections, and a table with
8011    // no sections is a correct table. The one difference from the blocks before it is that this one
8012    // is written even when it is empty, so that a file written by this build always says which
8013    // sections it has rather than leaving a reader to infer it from where the bytes ran out.
8014    out.extend_from_slice(SECTIONS);
8015    put_u64(&mut out, table.generation);
8016    put_u16(
8017        &mut out,
8018        u16::try_from(table.sections.len()).map_err(|_| invalid("too many sections"))?,
8019    );
8020    for held in &table.sections {
8021        held.encode(&mut out)?;
8022    }
8023    if table.dictionary_payloads.iter().any(|&bytes| bytes != 0) {
8024        out.extend_from_slice(DICTIONARY_PAYLOADS);
8025        put_u16(
8026            &mut out,
8027            u16::try_from(table.fields.len()).map_err(|_| invalid("too many columns"))?,
8028        );
8029        for at in 0..table.fields.len() {
8030            put_u64(&mut out, table.dictionary_payloads.get(at).copied().unwrap_or(0));
8031        }
8032    }
8033    Ok(out)
8034}
8035
8036/// The small level of the directory, naming every table in the file.
8037///
8038/// This is what a footer slot points at. Each entry carries its own checksum over its table
8039/// directory, so a table whose directory is torn is found when that table is first touched rather
8040/// than being trusted because the catalog around it checksummed.
8041///
8042/// The views go after the tables and are whole here, since a view is text and a column list and has
8043/// no pages for a second level to point at.
8044fn signed_integer(ty: &LogicalType) -> bool {
8045    matches!(
8046        ty,
8047        LogicalType::TinyInt | LogicalType::SmallInt | LogicalType::Integer | LogicalType::BigInt
8048    )
8049}
8050
8051fn integer_or_date(ty: &LogicalType) -> bool {
8052    matches!(
8053        ty,
8054        LogicalType::TinyInt
8055            | LogicalType::SmallInt
8056            | LogicalType::Integer
8057            | LogicalType::BigInt
8058            | LogicalType::UTinyInt
8059            | LogicalType::USmallInt
8060            | LogicalType::UInteger
8061            | LogicalType::UBigInt
8062            | LogicalType::Date
8063    )
8064}
8065
8066fn table_integer_extremes(table: &Table) -> Vec<StoredIntegerExtremes> {
8067    table
8068        .fields
8069        .iter()
8070        .enumerate()
8071        .map(|(column, field)| {
8072            if !integer_or_date(&field.ty) {
8073                return None;
8074            }
8075            let mut low: Option<i128> = None;
8076            let mut high: Option<i128> = None;
8077            for stripe in &table.stripes {
8078                let range = stripe.zone.column(column)?;
8079                if !range.exact {
8080                    return None;
8081                }
8082                match (range.low.as_ref(), range.high.as_ref()) {
8083                    (Some(Bound::Int(small)), Some(Bound::Int(large))) => {
8084                        low = Some(low.map_or(*small, |held| held.min(*small)));
8085                        high = Some(high.map_or(*large, |held| held.max(*large)));
8086                    }
8087                    (None, None) if stripe.rows == range.nulls => {}
8088                    _ => return None,
8089                }
8090            }
8091            Some(low.zip(high))
8092        })
8093        .collect()
8094}
8095
8096fn reader_integer_extremes(reader: &Reader) -> Result<Vec<StoredIntegerExtremes>> {
8097    reader
8098        .table
8099        .fields
8100        .iter()
8101        .enumerate()
8102        .map(|(column, field)| {
8103            if !integer_or_date(&field.ty) {
8104                return Ok(None);
8105            }
8106            match reader.exact_extremes(column)? {
8107                Some((Bound::Int(low), Bound::Int(high))) => Ok(Some(Some((low, high)))),
8108                None if reader.null_count(column)? == reader.table.rows as u64 => Ok(Some(None)),
8109                _ => Ok(None),
8110            }
8111        })
8112        .collect()
8113}
8114
8115fn table_complete_numeric_frequencies(table: &Table) -> Vec<StoredNumericFrequencies> {
8116    table
8117        .fields
8118        .iter()
8119        .enumerate()
8120        .map(|(column, field)| {
8121            if !integer_or_date(&field.ty) {
8122                return None;
8123            }
8124            let Some(Frequencies::Held(summary)) = table.frequencies.get(column)?.as_ref() else {
8125                return None;
8126            };
8127            if summary.omitted_max != 0 || summary.entries.len() > MAX_CATALOG_FREQUENCIES {
8128                return None;
8129            }
8130            let entries = summary
8131                .entries
8132                .iter()
8133                .map(|entry| {
8134                    let value = match entry.value {
8135                        FrequencyValue::Null => None,
8136                        FrequencyValue::Integer(value) => Some(value),
8137                        FrequencyValue::Code(_) => return None,
8138                    };
8139                    Some((value, entry.count))
8140                })
8141                .collect::<Option<Vec<_>>>()?;
8142            let rows = entries.iter().try_fold(0_u64, |sum, (_, count)| sum.checked_add(*count))?;
8143            (rows == table.rows as u64).then_some(entries)
8144        })
8145        .collect()
8146}
8147
8148/// The sixty four bits the close keys a numeric column's frequencies by, for a value the writer's
8149/// tally held.
8150///
8151/// The same bits [`Writer::visit_numeric`] hands over: a signed value sign extended to `i64`, and an
8152/// unsigned one as it is.
8153/// The value a column's sixty four bits stand for, read as signed or unsigned the way the column is.
8154fn integer_value(bits: u64, signed: bool) -> FrequencyValue {
8155    if signed {
8156        FrequencyValue::Integer(i128::from(bits as i64))
8157    } else {
8158        FrequencyValue::Integer(i128::from(bits))
8159    }
8160}
8161
8162fn frequency_bits(value: &Value) -> Option<u64> {
8163    Some(match value {
8164        Value::TinyInt(value) => i64::from(*value) as u64,
8165        Value::SmallInt(value) => i64::from(*value) as u64,
8166        Value::Integer(value) | Value::Date(value) => i64::from(*value) as u64,
8167        Value::BigInt(value) | Value::Timestamp(value) => *value as u64,
8168        Value::UTinyInt(value) => u64::from(*value),
8169        Value::USmallInt(value) => u64::from(*value),
8170        Value::UInteger(value) => u64::from(*value),
8171        Value::UBigInt(value) => *value,
8172        _ => return None,
8173    })
8174}
8175
8176fn numeric_frequency_value(value: &Value) -> Option<Option<i128>> {
8177    Some(match value {
8178        Value::Null => None,
8179        Value::TinyInt(value) => Some(i128::from(*value)),
8180        Value::SmallInt(value) => Some(i128::from(*value)),
8181        Value::Integer(value) | Value::Date(value) => Some(i128::from(*value)),
8182        Value::BigInt(value) => Some(i128::from(*value)),
8183        Value::UTinyInt(value) => Some(i128::from(*value)),
8184        Value::USmallInt(value) => Some(i128::from(*value)),
8185        Value::UInteger(value) => Some(i128::from(*value)),
8186        Value::UBigInt(value) => Some(i128::from(*value)),
8187        _ => return None,
8188    })
8189}
8190
8191fn reader_complete_numeric_frequencies(reader: &Reader) -> Result<Vec<StoredNumericFrequencies>> {
8192    reader
8193        .table
8194        .fields
8195        .iter()
8196        .enumerate()
8197        .map(|(column, field)| {
8198            if !integer_or_date(&field.ty) {
8199                return Ok(None);
8200            }
8201            let Some(summary) = reader.frequency_summary(column)? else { return Ok(None) };
8202            if summary.omitted_max != 0 || summary.entries.len() > MAX_CATALOG_FREQUENCIES {
8203                return Ok(None);
8204            }
8205            let entries = reader.decode_frequencies(column, &field.ty, &summary.entries)?;
8206            let Some(entries) = entries
8207                .iter()
8208                .map(|(value, count)| Some((numeric_frequency_value(value)?, *count)))
8209                .collect::<Option<Vec<_>>>()
8210            else {
8211                return Ok(None);
8212            };
8213            let rows = entries.iter().try_fold(0_u64, |sum, (_, count)| sum.checked_add(*count));
8214            Ok((rows == Some(reader.table.rows as u64)).then_some(entries))
8215        })
8216        .collect()
8217}
8218
8219fn table_exact_sum(table: &Table, column: usize) -> Option<(i128, u64)> {
8220    table.stripes.iter().try_fold((0_i128, 0_u64), |(sum, count), stripe| {
8221        let range = stripe.zone.column(column)?;
8222        let sum = sum.checked_add(range.sum?)?;
8223        let nonnull = (stripe.rows as u64).checked_sub(range.nulls as u64)?;
8224        Some((sum, count.checked_add(nonnull)?))
8225    })
8226}
8227
8228fn table_aggregate_sums(table: &Table) -> Vec<Option<(i128, u64)>> {
8229    table
8230        .fields
8231        .iter()
8232        .enumerate()
8233        .map(|(column, field)| {
8234            signed_integer(&field.ty).then(|| table_exact_sum(table, column)).flatten()
8235        })
8236        .collect()
8237}
8238
8239fn reader_aggregate_sums(reader: &Reader) -> Result<Vec<Option<(i128, u64)>>> {
8240    reader
8241        .table
8242        .fields
8243        .iter()
8244        .enumerate()
8245        .map(
8246            |(column, field)| {
8247                if signed_integer(&field.ty) { reader.exact_sum(column) } else { Ok(None) }
8248            },
8249        )
8250        .collect()
8251}
8252
8253fn encode_catalog(entries: &[Entry], views: &[ViewEntry]) -> Result<Vec<u8>> {
8254    let mut out = CATALOG.to_vec();
8255    put_u32(&mut out, u32::try_from(entries.len()).map_err(|_| invalid("too many tables"))?);
8256    for entry in entries {
8257        let name = entry.name.as_bytes();
8258        put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("table name too long"))?);
8259        out.extend_from_slice(name);
8260        put_u64(&mut out, u64::try_from(entry.rows).map_err(|_| invalid("row count overflow"))?);
8261        put_u16(
8262            &mut out,
8263            u16::try_from(entry.fields.len()).map_err(|_| invalid("too many columns"))?,
8264        );
8265        for field in &entry.fields {
8266            let name = field.name.as_bytes();
8267            put_u16(
8268                &mut out,
8269                u16::try_from(name.len()).map_err(|_| invalid("column name too long"))?,
8270            );
8271            out.extend_from_slice(name);
8272            put_type(&mut out, &field.ty)?;
8273            out.push(u8::from(field.not_null));
8274        }
8275        put_u64(&mut out, entry.directory.offset);
8276        put_u32(&mut out, entry.directory.length);
8277        put_u64(&mut out, entry.directory.hash);
8278    }
8279    put_u32(&mut out, u32::try_from(views.len()).map_err(|_| invalid("too many views"))?);
8280    for view in views {
8281        let name = view.name.as_bytes();
8282        put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("view name too long"))?);
8283        out.extend_from_slice(name);
8284        put_long_text(&mut out, &view.sql, "view body")?;
8285        put_long_text(&mut out, &view.statement, "view statement")?;
8286        put_u16(
8287            &mut out,
8288            u16::try_from(view.aliases.len()).map_err(|_| invalid("too many aliases"))?,
8289        );
8290        for alias in &view.aliases {
8291            let alias = alias.as_bytes();
8292            put_u16(
8293                &mut out,
8294                u16::try_from(alias.len()).map_err(|_| invalid("alias name too long"))?,
8295            );
8296            out.extend_from_slice(alias);
8297        }
8298        put_u16(
8299            &mut out,
8300            u16::try_from(view.columns.len()).map_err(|_| invalid("too many columns"))?,
8301        );
8302        for field in &view.columns {
8303            let name = field.name.as_bytes();
8304            put_u16(
8305                &mut out,
8306                u16::try_from(name.len()).map_err(|_| invalid("column name too long"))?,
8307            );
8308            out.extend_from_slice(name);
8309            put_type(&mut out, &field.ty)?;
8310            out.push(u8::from(field.not_null));
8311        }
8312    }
8313    out.extend_from_slice(NONZERO_COUNTS);
8314    for entry in entries {
8315        if entry.nonzero.len() != entry.fields.len() {
8316            return Err(invalid("nonzero count width differs from schema"));
8317        }
8318        for count in &entry.nonzero {
8319            match count {
8320                None => out.push(0),
8321                Some(count) => {
8322                    out.push(1);
8323                    put_u64(&mut out, *count);
8324                }
8325            }
8326        }
8327    }
8328    out.extend_from_slice(AGGREGATE_SUMS);
8329    for entry in entries {
8330        if entry.aggregates.len() != entry.fields.len() {
8331            return Err(invalid("aggregate sum width differs from schema"));
8332        }
8333        for summary in &entry.aggregates {
8334            match summary {
8335                None => out.push(0),
8336                Some((sum, count)) => {
8337                    out.push(1);
8338                    out.extend_from_slice(&sum.to_le_bytes());
8339                    put_u64(&mut out, *count);
8340                }
8341            }
8342        }
8343    }
8344    out.extend_from_slice(DISTINCT_COUNTS);
8345    for entry in entries {
8346        if entry.distincts.len() != entry.fields.len() {
8347            return Err(invalid("distinct count width differs from schema"));
8348        }
8349        for count in &entry.distincts {
8350            match count {
8351                None => out.push(0),
8352                Some(count) => {
8353                    if *count > entry.rows as u64 {
8354                        return Err(invalid("distinct count exceeds table rows"));
8355                    }
8356                    out.push(1);
8357                    put_u64(&mut out, *count);
8358                }
8359            }
8360        }
8361    }
8362    out.extend_from_slice(INTEGER_EXTREMES);
8363    for entry in entries {
8364        if entry.extremes.len() != entry.fields.len() {
8365            return Err(invalid("integer extremes width differs from schema"));
8366        }
8367        for (field, extremes) in entry.fields.iter().zip(&entry.extremes) {
8368            match extremes {
8369                None => out.push(0),
8370                Some(None) if integer_or_date(&field.ty) => out.push(1),
8371                Some(Some((low, high))) if integer_or_date(&field.ty) && low <= high => {
8372                    out.push(2);
8373                    out.extend_from_slice(&low.to_le_bytes());
8374                    out.extend_from_slice(&high.to_le_bytes());
8375                }
8376                _ => return Err(invalid("integer extremes type or range differs")),
8377            }
8378        }
8379    }
8380    out.extend_from_slice(COMPLETE_FREQUENCIES);
8381    for entry in entries {
8382        if entry.frequencies.len() != entry.fields.len() {
8383            return Err(invalid("numeric frequency width differs from schema"));
8384        }
8385        for (field, frequencies) in entry.fields.iter().zip(&entry.frequencies) {
8386            match frequencies {
8387                None => out.push(0),
8388                Some(entries)
8389                    if integer_or_date(&field.ty) && entries.len() <= MAX_CATALOG_FREQUENCIES =>
8390                {
8391                    let mut total = 0_u64;
8392                    for (at, (value, count)) in entries.iter().enumerate() {
8393                        if entries[..at].iter().any(|(held, _)| held == value) {
8394                            return Err(invalid("numeric frequency value repeats"));
8395                        }
8396                        total = total
8397                            .checked_add(*count)
8398                            .ok_or_else(|| invalid("numeric frequency count overflows"))?;
8399                    }
8400                    if total != entry.rows as u64 {
8401                        return Err(invalid("numeric frequencies do not cover table rows"));
8402                    }
8403                    out.push(1);
8404                    out.push(entries.len() as u8);
8405                    for (value, count) in entries {
8406                        match value {
8407                            None => out.push(0),
8408                            Some(value) => {
8409                                out.push(1);
8410                                out.extend_from_slice(&value.to_le_bytes());
8411                            }
8412                        }
8413                        put_u64(&mut out, *count);
8414                    }
8415                }
8416                _ => return Err(invalid("numeric frequency type or width differs")),
8417            }
8418        }
8419    }
8420    Ok(out)
8421}
8422
8423/// A length and that many bytes, for text that is allowed to be longer than a name.
8424fn put_long_text(out: &mut Vec<u8>, text: &str, what: &str) -> Result<()> {
8425    let bytes = text.as_bytes();
8426    put_u32(out, u32::try_from(bytes.len()).map_err(|_| invalid(&format!("{what} too long")))?);
8427    out.extend_from_slice(bytes);
8428    Ok(())
8429}
8430
8431/// Reads the catalog directory back, checking every span against the file before anything is
8432/// allocated for it.
8433fn decode_catalog(bytes: &[u8], size: u64) -> Result<(Vec<Entry>, Vec<ViewEntry>)> {
8434    let mut cur = Cursor::new(bytes);
8435    if cur.take(8)? != CATALOG {
8436        return Err(invalid("catalog magic differs"));
8437    }
8438    let count = cur.u32()? as usize;
8439    let mut entries: Vec<Entry> = Vec::with_capacity(count.min(1024));
8440    for _ in 0..count {
8441        let name = cur.text()?;
8442        let rows = usize::try_from(cur.u64()?).map_err(|_| invalid("row count does not fit"))?;
8443        let width = cur.u16()? as usize;
8444        let mut fields = Vec::with_capacity(width);
8445        for _ in 0..width {
8446            let name = cur.text()?;
8447            let ty = read_type(&mut cur)?;
8448            let not_null = match cur.u8()? {
8449                0 => false,
8450                1 => true,
8451                _ => return Err(invalid("nullability flag differs")),
8452            };
8453            fields.push(Field { name, ty, not_null });
8454        }
8455        let directory = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
8456        let end = directory
8457            .offset
8458            .checked_add(u64::from(directory.length))
8459            .ok_or_else(|| invalid("table directory offset overflow"))?;
8460        if directory.offset < HEADER
8461            || end > size
8462            || directory.length as usize > MAX_DIRECTORY
8463            || directory.length == 0
8464        {
8465            return Err(invalid("table directory range is outside the file"));
8466        }
8467        if entries.iter().any(|held| held.name == name) {
8468            return Err(invalid("two tables in the catalog have the same name"));
8469        }
8470        let nonzero = vec![None; fields.len()];
8471        let aggregates = vec![None; fields.len()];
8472        let distincts = vec![None; fields.len()];
8473        let extremes = vec![None; fields.len()];
8474        let frequencies = vec![None; fields.len()];
8475        entries.push(Entry {
8476            name,
8477            fields,
8478            rows,
8479            directory,
8480            nonzero,
8481            aggregates,
8482            distincts,
8483            extremes,
8484            frequencies,
8485        });
8486    }
8487    // A catalog that ends where the tables end is a catalog with no views in it, which is every
8488    // file written before format 25. That is why the count is allowed to be missing rather than
8489    // read as a zero that has to be there: an older file has nothing after the last table entry at
8490    // all, and [`READABLE`] says those files still open.
8491    let count = if cur.done() { 0 } else { cur.u32()? as usize };
8492    let mut views: Vec<ViewEntry> = Vec::with_capacity(count.min(1024));
8493    for _ in 0..count {
8494        let name = cur.text()?;
8495        let sql = cur.long_text()?;
8496        let statement = cur.long_text()?;
8497        let width = cur.u16()? as usize;
8498        let mut aliases = Vec::with_capacity(width);
8499        for _ in 0..width {
8500            aliases.push(cur.text()?);
8501        }
8502        let width = cur.u16()? as usize;
8503        let mut columns = Vec::with_capacity(width);
8504        for _ in 0..width {
8505            let name = cur.text()?;
8506            let ty = read_type(&mut cur)?;
8507            let not_null = match cur.u8()? {
8508                0 => false,
8509                1 => true,
8510                _ => return Err(invalid("nullability flag differs")),
8511            };
8512            columns.push(Field { name, ty, not_null });
8513        }
8514        // The same rule the tables above get, and for the same reason. Two entries under one name
8515        // is a catalog nothing can answer a lookup from, and finding that out here is better than
8516        // finding it out from whichever of the two a search happened to reach first.
8517        if views.iter().any(|held| held.name == name) {
8518            return Err(invalid("two views in the catalog have the same name"));
8519        }
8520        if entries.iter().any(|held| held.name == name) {
8521            return Err(invalid("a table and a view in the catalog have the same name"));
8522        }
8523        views.push(ViewEntry { name, sql, statement, aliases, columns });
8524    }
8525    if !cur.done() {
8526        if cur.take(8)? != NONZERO_COUNTS {
8527            return Err(invalid("catalog extension magic differs"));
8528        }
8529        for entry in &mut entries {
8530            for (field, count) in entry.fields.iter().zip(&mut entry.nonzero) {
8531                *count = match cur.u8()? {
8532                    0 => None,
8533                    1 if matches!(
8534                        field.ty,
8535                        LogicalType::TinyInt
8536                            | LogicalType::SmallInt
8537                            | LogicalType::Integer
8538                            | LogicalType::BigInt
8539                            | LogicalType::UTinyInt
8540                            | LogicalType::USmallInt
8541                            | LogicalType::UInteger
8542                            | LogicalType::UBigInt
8543                    ) =>
8544                    {
8545                        let value = cur.u64()?;
8546                        if value > entry.rows as u64 {
8547                            return Err(invalid("nonzero count exceeds rows"));
8548                        }
8549                        Some(value)
8550                    }
8551                    _ => return Err(invalid("nonzero count tag or column type differs")),
8552                };
8553            }
8554        }
8555    }
8556    if !cur.done() {
8557        if cur.take(8)? != AGGREGATE_SUMS {
8558            return Err(invalid("aggregate catalog extension magic differs"));
8559        }
8560        for entry in &mut entries {
8561            for (field, summary) in entry.fields.iter().zip(&mut entry.aggregates) {
8562                *summary = match cur.u8()? {
8563                    0 => None,
8564                    1 if signed_integer(&field.ty) => {
8565                        let sum = i128::from_le_bytes(
8566                            cur.take(16)?
8567                                .try_into()
8568                                .map_err(|_| invalid("aggregate sum is truncated"))?,
8569                        );
8570                        let count = cur.u64()?;
8571                        if count > entry.rows as u64 {
8572                            return Err(invalid("aggregate count exceeds table rows"));
8573                        }
8574                        Some((sum, count))
8575                    }
8576                    _ => return Err(invalid("aggregate sum tag or column type differs")),
8577                };
8578            }
8579        }
8580    }
8581    if !cur.done() {
8582        if cur.take(8)? != DISTINCT_COUNTS {
8583            return Err(invalid("distinct catalog extension magic differs"));
8584        }
8585        for entry in &mut entries {
8586            for count in &mut entry.distincts {
8587                *count = match cur.u8()? {
8588                    0 => None,
8589                    1 => {
8590                        let value = cur.u64()?;
8591                        if value > entry.rows as u64 {
8592                            return Err(invalid("distinct count exceeds table rows"));
8593                        }
8594                        Some(value)
8595                    }
8596                    _ => return Err(invalid("distinct count tag differs")),
8597                };
8598            }
8599        }
8600    }
8601    if !cur.done() {
8602        if cur.take(8)? != INTEGER_EXTREMES {
8603            return Err(invalid("integer extremes catalog extension magic differs"));
8604        }
8605        for entry in &mut entries {
8606            for (field, extremes) in entry.fields.iter().zip(&mut entry.extremes) {
8607                *extremes = match cur.u8()? {
8608                    0 => None,
8609                    1 if integer_or_date(&field.ty) => Some(None),
8610                    2 if integer_or_date(&field.ty) => {
8611                        let low = i128::from_le_bytes(
8612                            cur.take(16)?
8613                                .try_into()
8614                                .map_err(|_| invalid("minimum is truncated"))?,
8615                        );
8616                        let high = i128::from_le_bytes(
8617                            cur.take(16)?
8618                                .try_into()
8619                                .map_err(|_| invalid("maximum is truncated"))?,
8620                        );
8621                        if low > high {
8622                            return Err(invalid("integer extremes are reversed"));
8623                        }
8624                        Some(Some((low, high)))
8625                    }
8626                    _ => return Err(invalid("integer extremes tag or type differs")),
8627                };
8628            }
8629        }
8630    }
8631    if !cur.done() {
8632        if cur.take(8)? != COMPLETE_FREQUENCIES {
8633            return Err(invalid("numeric frequency catalog extension magic differs"));
8634        }
8635        for entry in &mut entries {
8636            for (field, frequencies) in entry.fields.iter().zip(&mut entry.frequencies) {
8637                *frequencies = match cur.u8()? {
8638                    0 => None,
8639                    1 if integer_or_date(&field.ty) => {
8640                        let len = cur.u8()? as usize;
8641                        if len > MAX_CATALOG_FREQUENCIES {
8642                            return Err(invalid("too many catalog numeric frequencies"));
8643                        }
8644                        let mut values = Vec::with_capacity(len);
8645                        let mut total = 0_u64;
8646                        for _ in 0..len {
8647                            let value = match cur.u8()? {
8648                                0 => None,
8649                                1 => Some(i128::from_le_bytes(cur.take(16)?.try_into().map_err(
8650                                    |_| invalid("numeric frequency value is truncated"),
8651                                )?)),
8652                                _ => return Err(invalid("numeric frequency value tag differs")),
8653                            };
8654                            if values.iter().any(|(held, _)| *held == value) {
8655                                return Err(invalid("numeric frequency value repeats"));
8656                            }
8657                            let count = cur.u64()?;
8658                            total = total
8659                                .checked_add(count)
8660                                .ok_or_else(|| invalid("numeric frequency count overflows"))?;
8661                            values.push((value, count));
8662                        }
8663                        if total != entry.rows as u64 {
8664                            return Err(invalid("numeric frequencies do not cover table rows"));
8665                        }
8666                        Some(values)
8667                    }
8668                    _ => return Err(invalid("numeric frequency tag or type differs")),
8669                };
8670            }
8671        }
8672    }
8673    if !cur.done() {
8674        return Err(invalid("catalog has trailing bytes"));
8675    }
8676    Ok((entries, views))
8677}
8678
8679/// Reads the fields of a directory or a catalog in order, off bytes in memory or out of the file.
8680///
8681/// A catalog is small and is read whole. A table directory is not: at ten million rows of `hits` it
8682/// is nearly a megabyte, and holding that buffer while the table it describes is built out of it
8683/// put both at the peak of every query. Out of the file, the cursor holds one window of
8684/// [`DIRECTORY_WINDOW`] bytes and moves it forward as the fields are read, so what a directory
8685/// costs at open is what it decodes into and not that plus its own bytes.
8686struct Cursor<'a> {
8687    bytes: &'a [u8],
8688    at: usize,
8689    window: Option<Window<'a>>,
8690}
8691
8692/// The part of a directory in the file that a [`Cursor`] has read in.
8693struct Window<'a> {
8694    file: &'a File,
8695    offset: u64,
8696    length: usize,
8697    /// Where `held` starts, counted from the start of the directory.
8698    start: usize,
8699    held: Vec<u8>,
8700    /// How much to read at once, which is [`DIRECTORY_WINDOW`] outside the tests.
8701    size: usize,
8702}
8703
8704/// How much of a directory a cursor reading one out of the file holds at once.
8705const DIRECTORY_WINDOW: usize = 64 << 10;
8706
8707impl<'a> Cursor<'a> {
8708    fn new(bytes: &'a [u8]) -> Self {
8709        Self { bytes, at: 0, window: None }
8710    }
8711
8712    /// A cursor over `length` bytes of `file` from `offset`, which it reads a window at a time.
8713    fn over(file: &'a File, offset: u64, length: usize) -> Self {
8714        let window =
8715            Window { file, offset, length, start: 0, held: Vec::new(), size: DIRECTORY_WINDOW };
8716        Self { bytes: &[], at: 0, window: Some(window) }
8717    }
8718
8719    /// How many bytes the cursor walks in all.
8720    fn len(&self) -> usize {
8721        self.window.as_ref().map_or(self.bytes.len(), |window| window.length)
8722    }
8723
8724    /// Makes sure the next `len` bytes are in memory.
8725    fn ensure(&mut self, len: usize) -> Result<()> {
8726        let end = self.at.checked_add(len).ok_or_else(|| invalid("directory offset overflow"))?;
8727        if end > self.len() {
8728            return Err(invalid("directory is truncated"));
8729        }
8730        let Some(window) = &mut self.window else { return Ok(()) };
8731        if self.at < window.start || end > window.start + window.held.len() {
8732            let want = len.max(window.size).min(window.length - self.at);
8733            window.start = self.at;
8734            window.held.resize(want, 0);
8735            read_at(window.file, window.offset + self.at as u64, &mut window.held)?;
8736        }
8737        Ok(())
8738    }
8739
8740    /// `len` bytes from `at`, which [`Self::ensure`] has already brought in.
8741    fn held(&self, at: usize, len: usize) -> &[u8] {
8742        match &self.window {
8743            Some(window) => &window.held[at - window.start..at - window.start + len],
8744            None => &self.bytes[at..at + len],
8745        }
8746    }
8747
8748    /// The next `len` bytes, without moving past them.
8749    #[inline]
8750    fn peek(&mut self, len: usize) -> Result<&[u8]> {
8751        if self.window.is_none() {
8752            let bytes = self.bytes;
8753            return Ok(&bytes[self.at..self.end(len)?]);
8754        }
8755        self.ensure(len)?;
8756        Ok(self.held(self.at, len))
8757    }
8758
8759    /// The next `len` bytes, moving past them.
8760    ///
8761    /// Every data page is decoded through this, a byte or a word at a time, so a cursor over bytes
8762    /// already in memory takes them here and never reaches [`Self::ensure`]. With the window check
8763    /// on every call, q06 on TPC-H spent a seventh of its instructions in it.
8764    #[inline]
8765    fn take(&mut self, len: usize) -> Result<&[u8]> {
8766        if self.window.is_none() {
8767            let bytes = self.bytes;
8768            let (at, end) = (self.at, self.end(len)?);
8769            self.at = end;
8770            return Ok(&bytes[at..end]);
8771        }
8772        self.take_windowed(len)
8773    }
8774
8775    /// Moves over a checked field without reading its payload from a windowed directory.
8776    fn skip(&mut self, len: usize) -> Result<()> {
8777        let end = self.at.checked_add(len).ok_or_else(|| invalid("directory offset overflow"))?;
8778        if end > self.len() {
8779            return Err(invalid("directory is truncated"));
8780        }
8781        self.at = end;
8782        Ok(())
8783    }
8784
8785    fn skip_bound(&mut self) -> Result<()> {
8786        match self.u8()? {
8787            0 => Ok(()),
8788            1 => self.skip(16),
8789            2 => self.skip(8),
8790            3 => {
8791                let length = self.u32()? as usize;
8792                self.skip(length)
8793            }
8794            4 => self.skip(17),
8795            _ => Err(invalid("a stored bound has an unknown tag")),
8796        }
8797    }
8798
8799    /// Where `len` bytes from here end, when they end inside the bytes.
8800    #[inline]
8801    fn end(&self, len: usize) -> Result<usize> {
8802        let end = self.at.checked_add(len).ok_or_else(|| invalid("directory offset overflow"))?;
8803        if end > self.bytes.len() {
8804            return Err(invalid("directory is truncated"));
8805        }
8806        Ok(end)
8807    }
8808
8809    /// [`Self::take`] out of the file, a window at a time.
8810    #[inline(never)]
8811    fn take_windowed(&mut self, len: usize) -> Result<&[u8]> {
8812        self.ensure(len)?;
8813        self.at += len;
8814        Ok(self.held(self.at - len, len))
8815    }
8816    #[inline]
8817    fn u8(&mut self) -> Result<u8> {
8818        Ok(self.take(1)?[0])
8819    }
8820    #[inline]
8821    fn u16(&mut self) -> Result<u16> {
8822        Ok(u16::from_le_bytes(self.take(2)?.try_into().expect("two bytes")))
8823    }
8824    #[inline]
8825    fn u32(&mut self) -> Result<u32> {
8826        Ok(u32::from_le_bytes(self.take(4)?.try_into().expect("four bytes")))
8827    }
8828    #[inline]
8829    fn u64(&mut self) -> Result<u64> {
8830        Ok(u64::from_le_bytes(self.take(8)?.try_into().expect("eight bytes")))
8831    }
8832    fn var_u64(&mut self) -> Result<u64> {
8833        let mut value = 0_u64;
8834        for shift in (0..=63).step_by(7) {
8835            let byte = self.u8()?;
8836            let part = u64::from(byte & 0x7f);
8837            if shift == 63 && part > 1 {
8838                return Err(invalid("frequency ordinal varint overflows"));
8839            }
8840            value |= part << shift;
8841            if byte & 0x80 == 0 {
8842                return Ok(value);
8843            }
8844        }
8845        Err(invalid("frequency ordinal varint is too long"))
8846    }
8847    /// A zone map's end, in the layout `rudb_common::bounds` defines.
8848    ///
8849    /// The bytes are the ones this directory has written since format 10 and the codec moved to
8850    /// rank zero rather than being copied, because a column summary now writes the same two ends
8851    /// and two encodings of one type is how the two quietly stop agreeing.
8852    ///
8853    /// A bound's length is in the bound, so out of the file the cursor offers the codec a few bytes
8854    /// and offers it twice as many whenever it runs out before the directory does.
8855    fn bound(&mut self) -> Result<Option<Bound>> {
8856        let rest = self.len().saturating_sub(self.at);
8857        let mut want = 32;
8858        loop {
8859            let offered = self.peek(want.min(rest))?;
8860            let mut used = 0;
8861            match bounds::get(offered, &mut used) {
8862                Ok(bound) => {
8863                    self.at += used;
8864                    return Ok(bound);
8865                }
8866                Err(_) if want < rest => want *= 2,
8867                Err(error) => return Err(error),
8868            }
8869        }
8870    }
8871    fn text(&mut self) -> Result<String> {
8872        let len = self.u16()? as usize;
8873        String::from_utf8(self.take(len)?.to_vec()).map_err(|_| invalid("name is not UTF-8"))
8874    }
8875    /// Whether everything has been read, which is how a section that an older file does not have at
8876    /// all is told from one that is there and empty.
8877    fn done(&self) -> bool {
8878        self.at >= self.len()
8879    }
8880    /// The same, for text that is a query rather than a name.
8881    ///
8882    /// A name fits in sixteen bits of length and a view body does not have to. Nobody writes a 64
8883    /// kilobyte identifier by accident and people do write generated queries that long, and a view
8884    /// that could not be written down because its body was too big would be a limit invented here
8885    /// rather than one anything else in the engine has.
8886    fn long_text(&mut self) -> Result<String> {
8887        let len = self.u32()? as usize;
8888        String::from_utf8(self.take(len)?.to_vec()).map_err(|_| invalid("text is not UTF-8"))
8889    }
8890}
8891
8892/// One column's frequency synopsis, or `None` for a column that has none, checked against the column.
8893fn decode_summary(
8894    cur: &mut Cursor<'_>,
8895    field: &Field,
8896    rows: usize,
8897    values: bool,
8898) -> Result<Option<FrequencySummary>> {
8899    Ok(match cur.u8()? {
8900        0 => None,
8901        1 => {
8902            let omitted_max = cur.u64()?;
8903            let count = cur.u32()? as usize;
8904            if count > FREQUENCY_ENTRIES {
8905                return Err(invalid("frequency entry count exceeds its bound"));
8906            }
8907            let mut entries = Vec::with_capacity(count);
8908            // row at a time: directory decoding validates each persisted bounded frequency entry.
8909            for _ in 0..count {
8910                let value = match cur.u8()? {
8911                    0 => FrequencyValue::Null,
8912                    1 => FrequencyValue::Integer(i128::from_le_bytes(
8913                        cur.take(16)?.try_into().expect("sixteen bytes"),
8914                    )),
8915                    2 => FrequencyValue::Code(cur.u32()?),
8916                    _ => return Err(invalid("frequency value tag differs")),
8917                };
8918                let valid = matches!(
8919                    (&field.ty, value),
8920                    (_, FrequencyValue::Null)
8921                        | (LogicalType::Varchar | LogicalType::Blob, FrequencyValue::Code(_))
8922                        | (
8923                            LogicalType::TinyInt
8924                                | LogicalType::SmallInt
8925                                | LogicalType::Integer
8926                                | LogicalType::BigInt
8927                                | LogicalType::UTinyInt
8928                                | LogicalType::USmallInt
8929                                | LogicalType::UInteger
8930                                | LogicalType::UBigInt
8931                                | LogicalType::Date
8932                                | LogicalType::Timestamp,
8933                            FrequencyValue::Integer(_),
8934                        )
8935                );
8936                if !valid {
8937                    return Err(invalid("frequency value does not match its column"));
8938                }
8939                let count = cur.u64()?;
8940                if count == 0 || count > rows as u64 {
8941                    return Err(invalid("frequency count is outside the table"));
8942                }
8943                entries.push(FrequencyEntry { value, count });
8944            }
8945            if entries.windows(2).any(|pair| pair[0].count < pair[1].count) {
8946                return Err(invalid("frequency entries are not descending"));
8947            }
8948            let ordinals = {
8949                let ordinal_count = cur.u32()? as usize;
8950                if ordinal_count > FREQUENCY_ORDINALS || ordinal_count > rows {
8951                    return Err(invalid("frequency ordinal count exceeds its bound"));
8952                }
8953                let mut ordinals = Vec::with_capacity(ordinal_count);
8954                let mut previous = 0_u64;
8955                for at in 0..ordinal_count {
8956                    let delta = cur.var_u64()?;
8957                    if at != 0 && delta == 0 {
8958                        return Err(invalid("frequency ordinals are not increasing"));
8959                    }
8960                    let ordinal = if at == 0 {
8961                        delta
8962                    } else {
8963                        previous
8964                            .checked_add(delta)
8965                            .ok_or_else(|| invalid("frequency ordinal overflows"))?
8966                    };
8967                    if ordinal >= rows as u64 {
8968                        return Err(invalid("frequency ordinal is outside the table"));
8969                    }
8970                    ordinals.push(ordinal);
8971                    previous = ordinal;
8972                }
8973                ordinals
8974            };
8975            let ordinal_entries = if values {
8976                let mut ordinal_entries = Vec::with_capacity(ordinals.len());
8977                for _ in 0..ordinals.len() {
8978                    let entry = cur.u16()?;
8979                    if entry as usize >= entries.len() {
8980                        return Err(invalid("frequency ordinal value is outside its entries"));
8981                    }
8982                    ordinal_entries.push(entry);
8983                }
8984                ordinal_entries
8985            } else {
8986                Vec::new()
8987            };
8988            Some(FrequencySummary { entries, omitted_max, ordinals, ordinal_entries })
8989        }
8990        _ => return Err(invalid("frequency summary tag differs")),
8991    })
8992}
8993
8994/// Skips a synopsis whose column the caller does not need. The directory checksum was checked
8995/// before this walk, and the fields still need their lengths and tags checked to find the next one.
8996fn skip_summary(cur: &mut Cursor<'_>, values: bool, rows: usize) -> Result<()> {
8997    match cur.u8()? {
8998        0 => Ok(()),
8999        1 => {
9000            cur.skip(8)?;
9001            let entries = cur.u32()? as usize;
9002            if entries > FREQUENCY_ENTRIES {
9003                return Err(invalid("frequency entry count exceeds its bound"));
9004            }
9005            for _ in 0..entries {
9006                match cur.u8()? {
9007                    0 => {}
9008                    1 => cur.skip(16)?,
9009                    2 => cur.skip(4)?,
9010                    _ => return Err(invalid("frequency value tag differs")),
9011                }
9012                cur.skip(8)?;
9013            }
9014            let ordinals = cur.u32()? as usize;
9015            if ordinals > FREQUENCY_ORDINALS || ordinals > rows {
9016                return Err(invalid("frequency ordinal count exceeds its bound"));
9017            }
9018            for _ in 0..ordinals {
9019                cur.var_u64()?;
9020            }
9021            if values {
9022                cur.skip(ordinals * 2)?;
9023            }
9024            Ok(())
9025        }
9026        _ => Err(invalid("frequency summary tag differs")),
9027    }
9028}
9029
9030/// Reads only the catalog, stripe null counts, and one frequency synopsis. This is the cold path
9031/// for a summary-backed count; constructing every page descriptor and zone map would make it cost
9032/// the size of the table directory even when no row is read.
9033fn quick_nonzero(
9034    mut cur: Cursor<'_>,
9035    name: &str,
9036    fields: &[Field],
9037    rows: usize,
9038    wanted: usize,
9039) -> Result<Option<u64>> {
9040    if cur.take(8)? != DIRECTORY || cur.text()? != name {
9041        return Err(invalid("table directory differs from the catalog"));
9042    }
9043    let width = cur.u16()? as usize;
9044    if width != fields.len() {
9045        return Err(invalid("table directory width differs from the catalog"));
9046    }
9047    for field in fields {
9048        let stored =
9049            Field { name: cur.text()?, ty: read_type(&mut cur)?, not_null: cur.u8()? != 0 };
9050        if &stored != field {
9051            return Err(invalid("table directory schema differs from the catalog"));
9052        }
9053    }
9054    let mut dictionaries = Vec::with_capacity(width);
9055    for field in fields {
9056        let held = match cur.u8()? {
9057            0 => false,
9058            tag if coded_type(&field.ty) && tag == dictionary_tag(&field.ty) => {
9059                cur.skip(20)?;
9060                true
9061            }
9062            _ => return Err(invalid("dictionary page tag differs")),
9063        };
9064        dictionaries.push(held);
9065    }
9066    for _ in 0..width {
9067        match cur.u8()? {
9068            0 => {}
9069            1 => cur.skip(8)?,
9070            _ => return Err(invalid("distinct count tag differs")),
9071        }
9072    }
9073    if cur.u64()? != rows as u64 {
9074        return Err(invalid("table row count differs from the catalog"));
9075    }
9076    let stripes = cur.u32()? as usize;
9077    let mut total = 0_usize;
9078    let mut nulls = 0_u64;
9079    for _ in 0..stripes {
9080        let parts = cur.u32()? as usize;
9081        if parts == 0 || parts > STRIPE_PARTS {
9082            return Err(invalid("stripe part count is outside its bound"));
9083        }
9084        let mut stripe_rows = 0_usize;
9085        for _ in 0..parts {
9086            stripe_rows = stripe_rows
9087                .checked_add(cur.u32()? as usize)
9088                .ok_or_else(|| invalid("stripe row count overflow"))?;
9089        }
9090        total =
9091            total.checked_add(stripe_rows).ok_or_else(|| invalid("stripe row count overflow"))?;
9092        cur.skip(12 + width * 12)?;
9093        for (field, held) in fields.iter().zip(&dictionaries) {
9094            if coded_type(&field.ty) && *held {
9095                cur.skip(20)?;
9096            }
9097        }
9098        for _ in 0..width * 2 {
9099            match cur.u8()? {
9100                0 => {}
9101                1 => cur.skip(20)?,
9102                _ => return Err(invalid("stripe page tag differs")),
9103            }
9104        }
9105        for column in 0..width {
9106            cur.skip_bound()?;
9107            cur.skip_bound()?;
9108            let count = cur.u32()? as u64;
9109            if count > stripe_rows as u64 {
9110                return Err(invalid("null count exceeds stripe rows"));
9111            }
9112            if column == wanted {
9113                nulls = nulls.checked_add(count).ok_or_else(|| invalid("null count overflow"))?;
9114            }
9115            cur.skip(1)?;
9116            match cur.u8()? {
9117                0 => {}
9118                1 => cur.skip(16)?,
9119                _ => return Err(invalid("a stripe sum has an unknown tag")),
9120            }
9121        }
9122    }
9123    if total != rows {
9124        return Err(invalid("table row count differs from stripes"));
9125    }
9126    if cur.done() {
9127        return Ok(None);
9128    }
9129    let magic = cur.take(8)?;
9130    let values = magic == FREQUENCIES;
9131    if !values && magic != FREQUENCIES_V2 {
9132        return Err(invalid("directory extension magic differs"));
9133    }
9134    if cur.u16()? as usize != width {
9135        return Err(invalid("frequency column count differs"));
9136    }
9137    for _ in 0..wanted {
9138        skip_summary(&mut cur, values, rows)?;
9139    }
9140    let Some(summary) = decode_summary(&mut cur, &fields[wanted], rows, values)? else {
9141        return Ok(None);
9142    };
9143    let zero = summary
9144        .entries
9145        .iter()
9146        .find(|entry| entry.value == FrequencyValue::Integer(0))
9147        .map(|entry| entry.count)
9148        .or_else(|| (summary.omitted_max == 0).then_some(0));
9149    Ok(zero.and_then(|zero| (rows as u64).checked_sub(nulls)?.checked_sub(zero)))
9150}
9151
9152/// Walks the row-oriented directory while retaining only one column's index and page spans.
9153/// The catalog supplies the schema and the caller checks the complete directory checksum first.
9154fn quick_integer_fold(
9155    file: &File,
9156    mut cur: Cursor<'_>,
9157    entry: &Entry,
9158    size: u64,
9159    wanted: usize,
9160    emit: &mut impl FnMut(i64, u64) -> Result<()>,
9161) -> Result<()> {
9162    let name = &entry.name;
9163    let fields = &entry.fields;
9164    let rows = entry.rows;
9165    if cur.take(8)? != DIRECTORY || cur.text()? != name.as_str() {
9166        return Err(invalid("table directory differs from the catalog"));
9167    }
9168    let width = cur.u16()? as usize;
9169    if width != fields.len() {
9170        return Err(invalid("table directory width differs from the catalog"));
9171    }
9172    for field in fields {
9173        let stored =
9174            Field { name: cur.text()?, ty: read_type(&mut cur)?, not_null: cur.u8()? != 0 };
9175        if &stored != field {
9176            return Err(invalid("table directory schema differs from the catalog"));
9177        }
9178    }
9179    let mut dictionaries = Vec::with_capacity(width);
9180    for field in fields {
9181        dictionaries.push(match cur.u8()? {
9182            0 => false,
9183            tag if coded_type(&field.ty) && tag == dictionary_tag(&field.ty) => {
9184                cur.skip(20)?;
9185                true
9186            }
9187            _ => return Err(invalid("dictionary page tag differs")),
9188        });
9189    }
9190    for _ in 0..width {
9191        match cur.u8()? {
9192            0 => {}
9193            1 => cur.skip(8)?,
9194            _ => return Err(invalid("distinct count tag differs")),
9195        }
9196    }
9197    if cur.u64()? != rows as u64 {
9198        return Err(invalid("table row count differs from the catalog"));
9199    }
9200    let stripes = cur.u32()? as usize;
9201    let mut total = 0_usize;
9202    let mut bytes = Vec::new();
9203    for _ in 0..stripes {
9204        let parts = cur.u32()? as usize;
9205        if parts == 0 || parts > STRIPE_PARTS {
9206            return Err(invalid("stripe part count is outside its bound"));
9207        }
9208        let mut part_rows = Vec::with_capacity(parts);
9209        for _ in 0..parts {
9210            let count = cur.u32()? as usize;
9211            if count == 0 {
9212                return Err(invalid("empty part"));
9213            }
9214            total = total.checked_add(count).ok_or_else(|| invalid("stripe row count overflow"))?;
9215            part_rows.push(count);
9216        }
9217        let index = Span { offset: cur.u64()?, length: cur.u32()? };
9218        let section = index_section(parts)?;
9219        let index_length =
9220            section.checked_mul(width).ok_or_else(|| invalid("index page length overflow"))?;
9221        if index.offset < HEADER
9222            || index.offset.checked_add(u64::from(index.length)).is_none_or(|end| end > size)
9223            || index.length as usize != index_length
9224        {
9225            return Err(invalid("index page range is outside the file"));
9226        }
9227        cur.skip(wanted * 12)?;
9228        let page = Span { offset: cur.u64()?, length: cur.u32()? };
9229        if page.offset < HEADER
9230            || page.offset.checked_add(u64::from(page.length)).is_none_or(|end| end > size)
9231            || page.length as usize > MAX_PAGE
9232        {
9233            return Err(invalid("column page range is outside the file"));
9234        }
9235        cur.skip((width - wanted - 1) * 12)?;
9236        for (field, held) in fields.iter().zip(&dictionaries) {
9237            if coded_type(&field.ty) && *held {
9238                cur.skip(20)?;
9239            }
9240        }
9241        for _ in 0..width * 2 {
9242            match cur.u8()? {
9243                0 => {}
9244                1 => cur.skip(20)?,
9245                _ => return Err(invalid("stripe page tag differs")),
9246            }
9247        }
9248        for _ in 0..width {
9249            cur.skip_bound()?;
9250            cur.skip_bound()?;
9251            cur.skip(5)?;
9252            match cur.u8()? {
9253                0 => {}
9254                1 => cur.skip(16)?,
9255                _ => return Err(invalid("a stripe sum has an unknown tag")),
9256            }
9257        }
9258        let spans = read_index_span(file, index, page, parts, wanted)?;
9259        for (span, expected_rows) in spans.into_iter().zip(part_rows) {
9260            bytes.resize(span.length, 0);
9261            let at = page
9262                .offset
9263                .checked_add(span.start as u64)
9264                .ok_or_else(|| invalid("part range overflow"))?;
9265            read_at(file, at, &mut bytes)?;
9266            if checksum(&bytes) != span.hash {
9267                return Err(invalid("integer part checksum differs"));
9268            }
9269            if bytes.first() == Some(&5) && bytes.get(1) == Some(&0) {
9270                let decoded_rows = integer::fold(&bytes[2..], |value, count| {
9271                    check_integer_tally_value(value, &fields[wanted].ty)?;
9272                    emit(value, count)
9273                })?;
9274                if decoded_rows != expected_rows {
9275                    return Err(invalid("encoded integer part holds the wrong number of rows"));
9276                }
9277            } else {
9278                let column = decode(&fields[wanted].ty, expected_rows, &bytes, None)?;
9279                if let Some(packed) = column.packed_parts() {
9280                    let validity = column.validity();
9281                    let all_valid = column.none_null();
9282                    let base = packed.base();
9283                    let mut codes = [0_u64; 64];
9284                    for from in (0..expected_rows).step_by(codes.len()) {
9285                        let count = (expected_rows - from).min(codes.len());
9286                        packed.unpack(from, &mut codes[..count]);
9287                        for (offset, &code) in codes[..count].iter().enumerate() {
9288                            if all_valid || validity.is_valid(from + offset) {
9289                                // Vector::packed checked that this entire range fits the type.
9290                                emit((base + i128::from(code)) as i64, 1)?;
9291                            }
9292                        }
9293                    }
9294                    continue;
9295                }
9296                let column = column.into_flat()?;
9297                let validity = column.validity();
9298                macro_rules! count_decoded {
9299                    ($values:expr) => {
9300                        for (row, &value) in $values.as_slice().iter().enumerate() {
9301                            if validity.is_valid(row) {
9302                                emit(i64::from(value), 1)?;
9303                            }
9304                        }
9305                    };
9306                }
9307                match column.data() {
9308                    Some(Data::Int8(values)) => count_decoded!(values),
9309                    Some(Data::Int16(values)) => count_decoded!(values),
9310                    Some(Data::Int32(values)) => count_decoded!(values),
9311                    Some(Data::Int64(values)) => count_decoded!(values),
9312                    _ => return Err(invalid("decoded integer part has the wrong type")),
9313                }
9314            }
9315        }
9316    }
9317    if total != rows {
9318        return Err(invalid("table row count differs from stripes"));
9319    }
9320    Ok(())
9321}
9322
9323fn check_integer_tally_value(value: i64, ty: &LogicalType) -> Result<()> {
9324    let fits = match ty {
9325        LogicalType::TinyInt => i8::try_from(value).is_ok(),
9326        LogicalType::SmallInt => i16::try_from(value).is_ok(),
9327        LogicalType::Integer => i32::try_from(value).is_ok(),
9328        LogicalType::BigInt => true,
9329        _ => false,
9330    };
9331    if fits { Ok(()) } else { Err(invalid("encoded integer value is outside its column type")) }
9332}
9333
9334fn decode_directory(bytes: &[u8], size: u64) -> Result<Table> {
9335    read_directory(Cursor::new(bytes), size, None)
9336}
9337
9338/// A directory out of `cur`, which is a whole one in memory or one being read out of the file.
9339///
9340/// `stored_at` is where the directory starts in the file when it is being read out of it, and then
9341/// every frequency synopsis is checked and left there, as [`Frequencies::Stored`].
9342fn read_directory(mut cur: Cursor<'_>, size: u64, stored_at: Option<u64>) -> Result<Table> {
9343    if cur.take(8)? != DIRECTORY {
9344        return Err(invalid("directory magic differs"));
9345    }
9346    let name = cur.text()?;
9347    let width = cur.u16()? as usize;
9348    let mut fields = Vec::with_capacity(width);
9349    for _ in 0..width {
9350        let name = cur.text()?;
9351        let ty = read_type(&mut cur)?;
9352        let not_null = match cur.u8()? {
9353            0 => false,
9354            1 => true,
9355            _ => return Err(invalid("nullability flag differs")),
9356        };
9357        fields.push(Field { name, ty, not_null });
9358    }
9359    let mut dictionaries = Vec::with_capacity(width);
9360    for field in &fields {
9361        dictionaries.push(match cur.u8()? {
9362            0 => None,
9363            tag if tag == dictionary_tag(&field.ty) => {
9364                let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
9365                let end = page
9366                    .offset
9367                    .checked_add(u64::from(page.length))
9368                    .ok_or_else(|| invalid("dictionary page offset overflow"))?;
9369                // A global dictionary covers a whole column, not one bounded stripe. Its lazy
9370                // payload is intentionally allowed to grow past `MAX_PAGE`; only ordinary column
9371                // pages are capped there. `Writer::finish` has already bounded this length by the
9372                // on-disk `u32`, and the range check below keeps it inside the file.
9373                if page.offset < HEADER || end > size {
9374                    return Err(invalid("dictionary page range is outside the file"));
9375                }
9376                Some(page)
9377            }
9378            _ => return Err(invalid("dictionary page tag differs")),
9379        });
9380    }
9381    let mut distincts = Vec::with_capacity(width);
9382    for _ in 0..width {
9383        distincts.push(match cur.u8()? {
9384            0 => None,
9385            1 => Some(cur.u64()?),
9386            _ => return Err(invalid("distinct count tag differs")),
9387        });
9388    }
9389    let rows = usize::try_from(cur.u64()?).map_err(|_| invalid("row count does not fit"))?;
9390    let count = cur.u32()? as usize;
9391    let mut stripes = Vec::with_capacity(count);
9392    let mut total = 0_usize;
9393    for _ in 0..count {
9394        let count = cur.u32()? as usize;
9395        if count == 0 || count > STRIPE_PARTS {
9396            return Err(invalid("stripe part count is outside its bound"));
9397        }
9398        let mut parts = Vec::with_capacity(count);
9399        let mut stripe_rows = 0_usize;
9400        for _ in 0..count {
9401            let rows = cur.u32()?;
9402            if rows == 0 {
9403                return Err(invalid("empty part"));
9404            }
9405            parts.push(rows);
9406            stripe_rows = stripe_rows
9407                .checked_add(rows as usize)
9408                .ok_or_else(|| invalid("stripe row count overflow"))?;
9409        }
9410        total =
9411            total.checked_add(stripe_rows).ok_or_else(|| invalid("stripe row count overflow"))?;
9412        let index = Span { offset: cur.u64()?, length: cur.u32()? };
9413        let section = index_section(count)?;
9414        let wanted = section
9415            .checked_mul(width)
9416            .and_then(|bytes| u32::try_from(bytes).ok())
9417            .ok_or_else(|| invalid("index page length overflow"))?;
9418        let end = index
9419            .offset
9420            .checked_add(u64::from(index.length))
9421            .ok_or_else(|| invalid("index page offset overflow"))?;
9422        if index.offset < HEADER || end > size || index.length != wanted {
9423            return Err(invalid("index page range is outside the file"));
9424        }
9425        let mut pages = Vec::with_capacity(width);
9426        for _ in 0..width {
9427            let offset = cur.u64()?;
9428            let length = cur.u32()?;
9429            let end = offset
9430                .checked_add(u64::from(length))
9431                .ok_or_else(|| invalid("page offset overflow"))?;
9432            if offset < HEADER || end > size || length as usize > MAX_PAGE {
9433                return Err(invalid("page range is outside the file"));
9434            }
9435            pages.push(Span { offset, length });
9436        }
9437        let mut memberships = vec![None; width];
9438        for (column, field) in fields.iter().enumerate() {
9439            if !coded_type(&field.ty) || dictionaries[column].is_none() {
9440                continue;
9441            }
9442            let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
9443            let end = page
9444                .offset
9445                .checked_add(u64::from(page.length))
9446                .ok_or_else(|| invalid("membership page offset overflow"))?;
9447            if page.offset < HEADER || end > size || page.length as usize > MAX_PAGE {
9448                return Err(invalid("membership page range is outside the file"));
9449            }
9450            // No bytes is a stripe written after the column's dictionary was demoted, see
9451            // [`DEMOTED`], which is checked once the block that says so has been read.
9452            if page.length != 0 {
9453                memberships[column] = Some(page);
9454            }
9455        }
9456        let mut sieves = vec![None; width];
9457        for sieve in sieves.iter_mut().take(width) {
9458            match cur.u8()? {
9459                0 => continue,
9460                1 => {}
9461                _ => return Err(invalid("a sieve page has an unknown tag")),
9462            }
9463            let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
9464            let end = page
9465                .offset
9466                .checked_add(u64::from(page.length))
9467                .ok_or_else(|| invalid("sieve page offset overflow"))?;
9468            if page.offset < HEADER || end > size || page.length as usize > MAX_PAGE {
9469                return Err(invalid("sieve page range is outside the file"));
9470            }
9471            *sieve = Some(page);
9472        }
9473        let mut part_ranges = vec![None; width];
9474        for held in part_ranges.iter_mut().take(width) {
9475            match cur.u8()? {
9476                0 => continue,
9477                1 => {}
9478                _ => return Err(invalid("a part range page has an unknown tag")),
9479            }
9480            let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
9481            let end = page
9482                .offset
9483                .checked_add(u64::from(page.length))
9484                .ok_or_else(|| invalid("part range page offset overflow"))?;
9485            if page.offset < HEADER || end > size || page.length as usize > MAX_PAGE {
9486                return Err(invalid("part range page range is outside the file"));
9487            }
9488            *held = Some(page);
9489        }
9490        let mut ranges = Vec::with_capacity(width);
9491        for column in 0..width {
9492            let low = cur.bound()?;
9493            let high = cur.bound()?;
9494            let nulls = cur.u32()? as usize;
9495            if nulls > stripe_rows {
9496                return Err(invalid("null count exceeds stripe rows"));
9497            }
9498            let exact = cur.u8()? != 0;
9499            let sum = match cur.u8()? {
9500                0 => None,
9501                1 => Some(i128::from_le_bytes(
9502                    cur.take(16)?.try_into().map_err(|_| invalid("a stripe sum is truncated"))?,
9503                )),
9504                _ => return Err(invalid("a stripe sum has an unknown tag")),
9505            };
9506            // Files written before the ends of a decimal or a timestamp column carried their power
9507            // of ten hold a bare integer here, and that integer is the one the column holds, which
9508            // is what the power is over. So the type puts it back on the way in and an old file
9509            // prunes as well as a new one. A file that already wrote the power keeps it, because
9510            // this leaves anything that is not an integer alone.
9511            let ty = &fields.get(column).ok_or_else(|| invalid("a stripe range has no column"))?.ty;
9512            let low = low.map(|bound| scaled_as(bound, ty));
9513            let high = high.map(|bound| scaled_as(bound, ty));
9514            ranges.push(Range { low, high, nulls, exact, sum });
9515        }
9516        stripes.push(Stripe {
9517            rows: stripe_rows,
9518            parts,
9519            index,
9520            pages,
9521            memberships: Pages::from_slots(memberships)?,
9522            sieves: Pages::from_slots(sieves)?,
9523            part_ranges: Pages::from_slots(part_ranges)?,
9524            zone: Zone::from_ranges(ranges),
9525        });
9526    }
9527    if total != rows {
9528        return Err(invalid("table row count differs from stripes"));
9529    }
9530    // How many entries each column's synopsis lists, which is all a pair summary is checked against,
9531    // kept apart because the synopses themselves may be left in the file.
9532    let mut entry_counts = vec![0; width];
9533    let frequencies = if cur.done() {
9534        vec![None; width]
9535    } else {
9536        let frequency_magic = cur.take(8)?;
9537        let frequency_values = frequency_magic == FREQUENCIES;
9538        if !frequency_values && frequency_magic != FREQUENCIES_V2 {
9539            return Err(invalid("directory extension magic differs"));
9540        }
9541        if cur.u16()? as usize != width {
9542            return Err(invalid("frequency column count differs"));
9543        }
9544        let mut frequencies = Vec::with_capacity(width);
9545        for (field, entry_count) in fields.iter().zip(&mut entry_counts) {
9546            let start = cur.at;
9547            let summary = decode_summary(&mut cur, field, rows, frequency_values)?;
9548            *entry_count = summary.as_ref().map_or(0, |summary| summary.entries.len());
9549            frequencies.push(match (summary, stored_at) {
9550                (None, _) => None,
9551                (Some(summary), None) => Some(Frequencies::Held(summary)),
9552                (Some(_), Some(offset)) => Some(Frequencies::Stored {
9553                    span: Span {
9554                        offset: offset + start as u64,
9555                        length: u32::try_from(cur.at - start)
9556                            .map_err(|_| invalid("a frequency synopsis is too long"))?,
9557                    },
9558                    values: frequency_values,
9559                }),
9560            });
9561        }
9562        frequencies
9563    };
9564    // There are two optional trailing blocks now rather than one, so the reader dispatches on the
9565    // magic it finds rather than on where the bytes ran out. That is what lets the two arrive
9566    // independently: a format 22 directory ends here and has neither, a directory written before
9567    // the section table has only the clustering declaration, and each one still opens without a
9568    // rewrite. It is the G1 exit criterion, which is that a reader that knows about sections opens
9569    // a file that predates them and answers every query, only without the graph path.
9570    //
9571    // A repeated block is refused rather than allowed to win, because two clustering declarations
9572    // in one directory is a torn directory and the only question is which of them is the lie.
9573    let mut clustering = None;
9574    let mut sections = Vec::new();
9575    let mut pair_frequencies = Vec::new();
9576    let mut seen_pair_frequencies = false;
9577    let mut frequency_texts = vec![Vec::new(); width];
9578    let mut seen_frequency_texts = false;
9579    let mut host_groups = None;
9580    let mut demoted = Vec::new();
9581    let mut seen_sections = false;
9582    let mut dictionary_payloads = Vec::new();
9583    let mut seen_payloads = false;
9584    // Zero until a section table says otherwise, which is what a format 22 table gets and what
9585    // makes every section stamp fail to match on one, because real generations start at one.
9586    let mut generation = 0;
9587    while !cur.done() {
9588        let mut tag = [0u8; 8];
9589        tag.copy_from_slice(cur.take(8)?);
9590        if &tag == PAIR_FREQUENCIES {
9591            if seen_pair_frequencies {
9592                return Err(invalid("directory names two pair frequency blocks"));
9593            }
9594            seen_pair_frequencies = true;
9595            let count = cur.u16()? as usize;
9596            if count > MAX_PAIR_FREQUENCIES {
9597                return Err(invalid("pair frequency count exceeds its bound"));
9598            }
9599            pair_frequencies = Vec::with_capacity(count);
9600            for _ in 0..count {
9601                let first = cur.u16()?;
9602                let second = cur.u16()?;
9603                let first_at = first as usize;
9604                let second_at = second as usize;
9605                if frequencies.get(first_at).and_then(Option::as_ref).is_none() {
9606                    return Err(invalid("pair frequency first column has no synopsis"));
9607                }
9608                let first_entries = entry_counts[first_at];
9609                if !matches!(fields.get(second_at), Some(field) if field.ty == LogicalType::Varchar)
9610                    || dictionaries.get(second_at).copied().flatten().is_none()
9611                {
9612                    return Err(invalid("pair frequency second column has no stable dictionary"));
9613                }
9614                if pair_frequencies
9615                    .iter()
9616                    .any(|held: &PairFrequencySummary| held.first == first && held.second == second)
9617                {
9618                    return Err(invalid("directory repeats a pair frequency summary"));
9619                }
9620                let omitted_max = cur.u64()?;
9621                if omitted_max > rows as u64 {
9622                    return Err(invalid("pair frequency omitted count exceeds the table"));
9623                }
9624                let entries_count = cur.u16()? as usize;
9625                if entries_count > FREQUENCY_ENTRIES {
9626                    return Err(invalid("pair frequency entry count exceeds its bound"));
9627                }
9628                let mut entries = Vec::with_capacity(entries_count);
9629                for _ in 0..entries_count {
9630                    let first_entry = cur.u16()?;
9631                    if first_entry as usize >= first_entries {
9632                        return Err(invalid("pair frequency anchor is outside its synopsis"));
9633                    }
9634                    let second = match cur.u8()? {
9635                        0 => None,
9636                        1 => Some(cur.u32()?),
9637                        _ => return Err(invalid("pair frequency string tag differs")),
9638                    };
9639                    let count = cur.u64()?;
9640                    if count == 0 || count > rows as u64 {
9641                        return Err(invalid("pair frequency count is outside the table"));
9642                    }
9643                    entries.push(PairFrequencyEntry { first_entry, second, count });
9644                }
9645                if entries.windows(2).any(|pair| pair[0].count < pair[1].count) {
9646                    return Err(invalid("pair frequency entries are not descending"));
9647                }
9648                pair_frequencies.push(PairFrequencySummary { first, second, entries, omitted_max });
9649            }
9650        } else if &tag == FREQUENCY_TEXTS {
9651            if seen_frequency_texts {
9652                return Err(invalid("directory names two frequency text blocks"));
9653            }
9654            seen_frequency_texts = true;
9655            let columns = cur.u16()? as usize;
9656            if columns > width {
9657                return Err(invalid("frequency text column count exceeds the schema"));
9658            }
9659            for _ in 0..columns {
9660                let column = cur.u16()? as usize;
9661                if !frequency_texts.get(column).is_some_and(Vec::is_empty) {
9662                    return Err(invalid("frequency text column is repeated or out of range"));
9663                }
9664                if !matches!(fields.get(column), Some(field) if coded_type(&field.ty))
9665                    || dictionaries.get(column).copied().flatten().is_none()
9666                    || frequencies.get(column).and_then(Option::as_ref).is_none()
9667                {
9668                    return Err(invalid("frequency texts belong to a non-string synopsis"));
9669                }
9670                let count = cur.u16()? as usize;
9671                if count == 0 || count != entry_counts[column] {
9672                    return Err(invalid("frequency text count differs from its synopsis"));
9673                }
9674                let mut texts = Vec::with_capacity(count);
9675                for _ in 0..count {
9676                    texts.push(match cur.u8()? {
9677                        0 => None,
9678                        1 => {
9679                            let length = cur.u32()? as usize;
9680                            let bytes = cur.take(length)?.to_vec();
9681                            if fields[column].ty == LogicalType::Varchar {
9682                                std::str::from_utf8(&bytes)
9683                                    .map_err(|_| invalid("frequency text is not UTF-8"))?;
9684                            }
9685                            Some(bytes)
9686                        }
9687                        _ => return Err(invalid("frequency text tag differs")),
9688                    });
9689                }
9690                frequency_texts[column] = texts;
9691            }
9692        } else if &tag == HOST_GROUPS {
9693            if host_groups.is_some() {
9694                return Err(invalid("directory names two host group blocks"));
9695            }
9696            let column = cur.u16()? as usize;
9697            if !matches!(fields.get(column), Some(field) if field.ty == LogicalType::Varchar)
9698                || dictionaries.get(column).copied().flatten().is_none()
9699            {
9700                return Err(invalid("host groups belong to a non-string dictionary"));
9701            }
9702            let omitted_max = cur.u64()?;
9703            if omitted_max > rows as u64 {
9704                return Err(invalid("host group bound exceeds the table"));
9705            }
9706            let count = cur.u16()? as usize;
9707            if count > host::CAPACITY {
9708                return Err(invalid("host group count exceeds its bound"));
9709            }
9710            let mut entries = Vec::with_capacity(count);
9711            let mut bytes = 0_usize;
9712            for _ in 0..count {
9713                let host_len = cur.u32()? as usize;
9714                bytes =
9715                    bytes.checked_add(host_len).ok_or_else(|| invalid("host bytes overflow"))?;
9716                if bytes > host::BYTE_BUDGET {
9717                    return Err(invalid("host groups exceed their byte budget"));
9718                }
9719                let host = std::str::from_utf8(cur.take(host_len)?)
9720                    .map_err(|_| invalid("host is not UTF-8"))?
9721                    .to_owned();
9722                let count = cur.u64()?;
9723                if count == 0 || count > rows as u64 {
9724                    return Err(invalid("host group count exceeds the table"));
9725                }
9726                let bytes_sum = i128::from_le_bytes(
9727                    cur.take(16)?
9728                        .try_into()
9729                        .map_err(|_| invalid("host length sum is truncated"))?,
9730                );
9731                if bytes_sum < 0 {
9732                    return Err(invalid("host length sum is negative"));
9733                }
9734                let minimum_len = cur.u32()? as usize;
9735                bytes =
9736                    bytes.checked_add(minimum_len).ok_or_else(|| invalid("host bytes overflow"))?;
9737                if bytes > host::BYTE_BUDGET {
9738                    return Err(invalid("host groups exceed their byte budget"));
9739                }
9740                let minimum = std::str::from_utf8(cur.take(minimum_len)?)
9741                    .map_err(|_| invalid("host minimum is not UTF-8"))?
9742                    .to_owned();
9743                entries.push(host::HostEntry { host, count, bytes_sum, minimum });
9744            }
9745            if entries.windows(2).any(|pair| pair[0].count < pair[1].count)
9746                || entries.iter().any(|entry| entry.host.is_empty() || entry.minimum.is_empty())
9747            {
9748                return Err(invalid("host groups are not in certified order"));
9749            }
9750            host_groups = Some(host::HostSummary { column, omitted_max, entries });
9751        } else if &tag == CLUSTERING {
9752            if clustering.is_some() {
9753                return Err(invalid("directory names two clustering declarations"));
9754            }
9755            let bucket = Width::from_tag(cur.u8()?)
9756                .ok_or_else(|| invalid("clustering width tag differs"))?;
9757            let count = cur.u16()? as usize;
9758            let mut columns = Vec::with_capacity(count.min(fields.len()));
9759            for _ in 0..count {
9760                columns.push(u32::from(cur.u16()?));
9761            }
9762            // Through the constructor and not built by hand, so that a file claiming a column the
9763            // table does not have is caught at open rather than at the first scan that trusted it.
9764            clustering = Some(Clustering::new(columns, bucket, &fields).map_err(|_| {
9765                invalid("stored clustering declaration does not match the table it is on")
9766            })?);
9767        } else if &tag == DEMOTED {
9768            if !demoted.is_empty() {
9769                return Err(invalid("directory names two demoted column blocks"));
9770            }
9771            let count = cur.u16()? as usize;
9772            if count == 0 || count > width {
9773                return Err(invalid("demoted column count is outside the schema"));
9774            }
9775            demoted = vec![false; width];
9776            for _ in 0..count {
9777                let column = cur.u16()? as usize;
9778                if dictionaries.get(column).copied().flatten().is_none() || demoted[column] {
9779                    return Err(invalid("a demoted column is repeated or has no dictionary"));
9780                }
9781                demoted[column] = true;
9782            }
9783        } else if &tag == SECTIONS {
9784            if seen_sections {
9785                return Err(invalid("directory names two section tables"));
9786            }
9787            seen_sections = true;
9788            generation = cur.u64()?;
9789            let count = cur.u16()? as usize;
9790            if count > MAX_SECTIONS {
9791                return Err(invalid("section count exceeds its bound"));
9792            }
9793            sections = Vec::with_capacity(count);
9794            // entry at a time: a malformed section entry is refused rather than turned into an
9795            // offset.
9796            for _ in 0..count {
9797                sections.push(Section::decode(cur.take(section::ENTRY_BYTES)?)?);
9798            }
9799            for held in &sections {
9800                let Some(end) = held.extent_page.checked_add(u64::from(held.extent_bytes)) else {
9801                    return Err(invalid("a section's extent table overflows the file"));
9802                };
9803                // The bound check is here and not in `section`, because only the caller knows how
9804                // big the file is. A section pointing past the end is a torn directory, and reading
9805                // the payload it names would be reading whatever else is at that offset.
9806                if held.extent_bytes != 0 && (held.extent_page < HEADER || end > size) {
9807                    return Err(invalid("a section's extent table is outside the file"));
9808                }
9809                if held.extents == 0 && held.extent_bytes != 0 {
9810                    return Err(invalid("a section with no extents names an extent table"));
9811                }
9812            }
9813        } else if &tag == DICTIONARY_PAYLOADS {
9814            if seen_payloads {
9815                return Err(invalid("directory names two dictionary payload blocks"));
9816            }
9817            seen_payloads = true;
9818            let count = cur.u16()? as usize;
9819            if count != fields.len() {
9820                return Err(invalid("dictionary payload block does not match the table's columns"));
9821            }
9822            dictionary_payloads = Vec::with_capacity(count);
9823            for _ in 0..count {
9824                let bytes = cur.u64()?;
9825                if bytes > size {
9826                    return Err(invalid("a dictionary payload is larger than the file"));
9827                }
9828                dictionary_payloads.push(bytes);
9829            }
9830        } else {
9831            return Err(invalid("directory extension magic differs"));
9832        }
9833    }
9834    if !cur.done() {
9835        return Err(invalid("directory has trailing bytes"));
9836    }
9837    for stripe in &stripes {
9838        for (column, field) in fields.iter().enumerate() {
9839            if coded_type(&field.ty)
9840                && dictionaries[column].is_some()
9841                && stripe.memberships.get(column).is_none()
9842                && !demoted.get(column).copied().unwrap_or(false)
9843            {
9844                return Err(invalid("string page has no code membership index"));
9845            }
9846        }
9847    }
9848    Ok(Table {
9849        name,
9850        fields,
9851        stripes,
9852        rows,
9853        dictionaries,
9854        dictionary_payloads,
9855        demoted,
9856        distincts,
9857        frequencies,
9858        pair_frequencies,
9859        frequency_texts,
9860        host_groups,
9861        clustering,
9862        generation,
9863        sections,
9864    })
9865}
9866
9867/// A zone map's end, in the layout `rudb_common::bounds` defines. See [`Cursor::bound`].
9868fn put_bound(out: &mut Vec<u8>, bound: Option<&Bound>) -> Result<()> {
9869    bounds::put(out, bound)
9870}
9871
9872/// Which cascades are worth trying on a run of dictionary codes.
9873///
9874/// The exhaustive chooser encodes every candidate at every level of a cascade three deep and keeps
9875/// the smallest, which on a part of 1024 codes is around a hundred full encodes to decide something
9876/// three candidates were always going to win. It is the right default for a crate that does not
9877/// know what it is looking at. Here we do know. Codes are counted from zero in the order the values
9878/// were first seen, so a part of them is one value, or a narrow band, or a few long runs, and those
9879/// are constant, frame of reference and run length. Nothing else has ever come first on this data.
9880///
9881/// A dictionary of dictionary codes is the one candidate that can never pay, because the codes are
9882/// already the dictionary, and it is also the most expensive one to try. Below the top level the
9883/// streams are an RLE's run values and run lengths, which are integers in their own right with no
9884/// runs left in them, so only the two flat candidates go down there.
9885///
9886/// This is size given up for time on purpose, and the ablation is this chooser against
9887/// [`chooser::EXHAUSTIVE`] on the same file.
9888#[derive(Debug)]
9889struct Codes;
9890
9891impl chooser::Chooser for Codes {
9892    fn name(&self) -> &'static str {
9893        "codes"
9894    }
9895
9896    fn narrow_strings(
9897        &self,
9898        _values: &[&[u8]],
9899        offered: &[string::Kind],
9900        _depth: u8,
9901    ) -> Vec<string::Kind> {
9902        // Never reached, because nothing here encodes strings through the cascade. The trait asks
9903        // for it and the honest answer to a question we have no opinion on is the whole list.
9904        offered.to_vec()
9905    }
9906
9907    fn narrow_integers(
9908        &self,
9909        _values: &[i64],
9910        offered: &[integer::Kind],
9911        depth: u8,
9912    ) -> Vec<integer::Kind> {
9913        // The contract is a non empty subset, and a chunk that offers none of the three is a chunk
9914        // this has no opinion about rather than one that cannot be written.
9915        narrowed_to(Codes::keep(depth), offered)
9916    }
9917
9918    fn considers_integer(&self, kind: integer::Kind, depth: u8) -> bool {
9919        Codes::keep(depth).contains(&kind)
9920    }
9921}
9922
9923impl Codes {
9924    fn keep(depth: u8) -> &'static [integer::Kind] {
9925        if depth == 0 {
9926            &[integer::Kind::Constant, integer::Kind::Packed, integer::Kind::Rle]
9927        } else {
9928            &[integer::Kind::Constant, integer::Kind::Packed]
9929        }
9930    }
9931}
9932
9933/// The kinds of `offered` that are in `keep`, or all of `offered` when none of them are.
9934///
9935/// `Packed` applies to every chunk and both choosers keep it, so the fallback is never taken on a
9936/// chunk the cascade offers. It is there because the contract is a non empty subset and a chooser
9937/// that returned nothing would be a chunk that cannot be written. It is also why saying no to a kind
9938/// in `considers_integer` is safe: a kind that is never offered could only have been kept through
9939/// this fallback, and the fallback is never reached.
9940fn narrowed_to(keep: &[integer::Kind], offered: &[integer::Kind]) -> Vec<integer::Kind> {
9941    let narrowed: Vec<integer::Kind> =
9942        offered.iter().copied().filter(|kind| keep.contains(kind)).collect();
9943    if narrowed.is_empty() { offered.to_vec() } else { narrowed }
9944}
9945
9946/// Which cascades are worth trying on a part of plain integers.
9947///
9948/// Wider than [`Codes`] because the values are not codes and carry whatever shape the column has.
9949/// A timestamp column climbs, so delta is the one that matters and is the reason this exists at
9950/// all: three timestamp columns in ClickBench were coming out at exactly eight bytes a row with
9951/// nothing asked of them. The same three columns are why the stride is here, since a timestamp
9952/// loaded from a source that recorded whole seconds is microseconds with twenty zero bits under
9953/// every value. A column that is one value with a handful of exceptions is sparse. What is still
9954/// left out is the dictionary, for the same reason as in [`Codes`]: it is the most
9955/// expensive candidate to try and this file already puts the columns that want one through a
9956/// dictionary of their own before they ever reach here.
9957#[derive(Debug)]
9958struct Fixed;
9959
9960impl chooser::Chooser for Fixed {
9961    fn name(&self) -> &'static str {
9962        "fixed"
9963    }
9964
9965    fn narrow_strings(
9966        &self,
9967        _values: &[&[u8]],
9968        offered: &[string::Kind],
9969        _depth: u8,
9970    ) -> Vec<string::Kind> {
9971        offered.to_vec()
9972    }
9973
9974    fn narrow_integers(
9975        &self,
9976        _values: &[i64],
9977        offered: &[integer::Kind],
9978        depth: u8,
9979    ) -> Vec<integer::Kind> {
9980        narrowed_to(Fixed::keep(depth), offered)
9981    }
9982
9983    fn considers_integer(&self, kind: integer::Kind, depth: u8) -> bool {
9984        Fixed::keep(depth).contains(&kind)
9985    }
9986}
9987
9988impl Fixed {
9989    fn keep(depth: u8) -> &'static [integer::Kind] {
9990        if depth == 0 {
9991            &[
9992                integer::Kind::Constant,
9993                integer::Kind::Packed,
9994                integer::Kind::Delta,
9995                integer::Kind::Rle,
9996                integer::Kind::Sparse,
9997                integer::Kind::Strided,
9998            ]
9999        } else {
10000            &[integer::Kind::Constant, integer::Kind::Packed, integer::Kind::Delta]
10001        }
10002    }
10003}
10004
10005/// Every value of an integer part as an `i64`, or `None` for a part this cannot widen without
10006/// losing one.
10007///
10008/// `UBIGINT` is the only integer type left out, because half its range does not fit and a page that
10009/// silently wrapped would be worse than a page that stays plain. Booleans and strings are not
10010/// integers and have their own ways of being small.
10011fn widened(data: &Data) -> Option<Vec<i64>> {
10012    match data {
10013        Data::Int8(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
10014        Data::UInt8(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
10015        Data::Int16(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
10016        Data::UInt16(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
10017        Data::Int32(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
10018        Data::UInt32(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
10019        Data::Int64(values) => Some(values.to_vec()),
10020        _ => None,
10021    }
10022}
10023
10024/// An integer type a cascaded page can be read back into, and how to tell whether a value fits.
10025///
10026/// This exists so that the check and the conversion can be two loops instead of one. `TryFrom` puts
10027/// them together, which is the right shape for one value and the wrong one for a page: a fallible
10028/// conversion a value at a time is a branch a value at a time, the branch decides whether the loop
10029/// keeps going, and a loop like that is one no compiler will widen.
10030trait Narrow: Copy {
10031    /// How wide this type is, and what to add to a value to put its range at the bottom of a `u64`.
10032    ///
10033    /// Half the width for a signed type, which is what moves its smallest value to zero, and nothing
10034    /// for an unsigned one, whose smallest value is already there.
10035    const BIASED: (u32, u64);
10036
10037    /// The value narrowed, which the caller has already shown fits.
10038    fn narrow(value: i64) -> Self;
10039}
10040
10041/// The bits of `value` a `T` cannot hold, and zero when the value fits.
10042///
10043/// The question is asked this way round because the answers or together. A page fits when every
10044/// residue in it is zero, so the loop is an or into an accumulator and the decision is one test
10045/// after it, where asking whether each value is between a floor and a ceiling gives an answer that
10046/// does not combine and turns into a running minimum and maximum.
10047///
10048/// Biasing and shifting is what the answer is made of, rather than anything that reads more like the
10049/// question, because those are the operations a machine has four of. A 64 bit integer minimum is
10050/// AVX-512. So is a 64 bit arithmetic shift right, which is how the sign extension this could be
10051/// written as would have to be done. An add and a logical shift right are AVX2 and are on every
10052/// machine this runs on, so this is the form that gets four values a cycle instead of one.
10053///
10054/// Adding the bias moves the type's range to `0..=2^bits`, wrapping, so everything in range shifts
10055/// away to nothing and everything outside it leaves something behind. A negative value under an
10056/// unsigned type is caught by the same shift, because a negative `i64` read as a `u64` is enormous.
10057#[allow(clippy::cast_sign_loss, reason = "a residue is a bit pattern and not a number")]
10058fn residue<T: Narrow>(value: i64) -> u64 {
10059    let (bits, bias) = T::BIASED;
10060    (value as u64).wrapping_add(bias) >> bits
10061}
10062
10063/// Says a primitive integer narrows with `as`, and where the bottom of its range is.
10064///
10065/// `as` is a truncation and is the right operation here only because [`fit`] has already found every
10066/// residue zero, and it is what makes the second loop a narrowing store with no branch in it.
10067macro_rules! narrows {
10068    ($($ty:ty => $bias:expr),* $(,)?) => {$(
10069        impl Narrow for $ty {
10070            const BIASED: (u32, u64) = (<$ty>::BITS, $bias);
10071
10072            #[allow(
10073                clippy::cast_possible_truncation,
10074                clippy::cast_sign_loss,
10075                reason = "the caller has checked the bits this truncates away"
10076            )]
10077            fn narrow(value: i64) -> Self {
10078                value as Self
10079            }
10080        }
10081    )*};
10082}
10083
10084narrows! {
10085    i8 => 1 << 7,
10086    u8 => 0,
10087    i16 => 1 << 15,
10088    u16 => 0,
10089    i32 => 1 << 31,
10090    u32 => 0,
10091}
10092
10093/// Narrows a page's values, refusing the page if any of them does not fit.
10094///
10095/// The check first and the conversion second, rather than a fallible conversion a value at a time.
10096/// Both loops here are ones a compiler widens: [`residue`] is three instructions a lane and a
10097/// narrowing store is one. The version before this was a `TryFrom` and a `collect` into a `Result`,
10098/// which is a compare, a branch and a short circuit a value at a time, and on ClickBench 39 it was
10099/// seven percent of the query. The version after that kept a running minimum and maximum, which is
10100/// the obvious way to ask and needs a 64 bit integer minimum that AVX2 does not have, so it stayed
10101/// a value at a time and was still ten percent of the same query.
10102///
10103/// An empty page has nothing to refuse, which falls out of the accumulator starting at zero rather
10104/// than needing a case of its own.
10105fn fit<T: Narrow>(values: &[i64]) -> Result<Vec<T>> {
10106    let mut spilled = 0u64;
10107    for value in values {
10108        spilled |= residue::<T>(*value);
10109    }
10110    if spilled != 0 {
10111        return Err(invalid("page value is not of its type"));
10112    }
10113    Ok(values.iter().map(|value| T::narrow(*value)).collect())
10114}
10115
10116/// The same values back in the width the column is declared at.
10117///
10118/// A value that does not fit is a page that disagrees with the directory about what the column is,
10119/// which is a damaged file rather than a caller error, so it is refused rather than truncated.
10120fn narrowed(ty: &LogicalType, values: Vec<i64>) -> Result<Data> {
10121    Ok(match ty {
10122        LogicalType::TinyInt => Data::Int8(fit::<i8>(&values)?.into()),
10123        LogicalType::UTinyInt => Data::UInt8(fit::<u8>(&values)?.into()),
10124        LogicalType::SmallInt => Data::Int16(fit::<i16>(&values)?.into()),
10125        LogicalType::USmallInt => Data::UInt16(fit::<u16>(&values)?.into()),
10126        LogicalType::Integer | LogicalType::Date => Data::Int32(fit::<i32>(&values)?.into()),
10127        LogicalType::UInteger => Data::UInt32(fit::<u32>(&values)?.into()),
10128        LogicalType::BigInt
10129        | LogicalType::Timestamp
10130        | LogicalType::Time
10131        | LogicalType::TimeTz
10132        | LogicalType::TimestampTz
10133        | LogicalType::TimestampS
10134        | LogicalType::TimestampMs
10135        | LogicalType::TimestampNs => Data::Int64(values.into()),
10136        // A decimal is an integer of unscaled units, so the cascade reads back into whichever
10137        // integer the declared width says the column is stored as.
10138        LogicalType::Decimal { .. } => match ty.physical() {
10139            PhysicalType::Int16 => Data::Int16(fit::<i16>(&values)?.into()),
10140            PhysicalType::Int32 => Data::Int32(fit::<i32>(&values)?.into()),
10141            PhysicalType::Int64 => Data::Int64(values.into()),
10142            _ => return Err(invalid("cascade codec belongs to a decimal that is not an integer")),
10143        },
10144        _ => return Err(invalid("cascade codec belongs to a page that is not integers")),
10145    })
10146}
10147
10148/// How many bytes a part of this type costs written out plainly, which is what the cascade has to
10149/// beat before it is worth the decode.
10150fn plain_width(ty: &LogicalType) -> Option<usize> {
10151    Some(match ty {
10152        LogicalType::TinyInt | LogicalType::UTinyInt => 1,
10153        LogicalType::SmallInt | LogicalType::USmallInt => 2,
10154        LogicalType::Integer | LogicalType::UInteger | LogicalType::Date => 4,
10155        LogicalType::BigInt
10156        | LogicalType::Timestamp
10157        | LogicalType::Time
10158        | LogicalType::TimeTz
10159        | LogicalType::TimestampTz
10160        | LogicalType::TimestampS
10161        | LogicalType::TimestampMs
10162        | LogicalType::TimestampNs => 8,
10163        LogicalType::Decimal { .. } => match ty.physical() {
10164            PhysicalType::Int16 => 2,
10165            PhysicalType::Int32 => 4,
10166            PhysicalType::Int64 => 8,
10167            // The widest decimals are stored as `i128`, which the cascade does not widen into, so
10168            // they take the plain path and there is nothing here to compare against.
10169            _ => return None,
10170        },
10171        _ => return None,
10172    })
10173}
10174
10175/// A part's plain integers through the cascade, or `None` when nothing it offers is worth it.
10176///
10177/// What it has to beat is whatever the page would otherwise have cost, which is the bit packed form
10178/// where there is one and the plain width where there is not. Both are cheaper to decode than a
10179/// cascade, so a tie goes to them.
10180fn cascaded(
10181    flat: &Vector,
10182    ty: &LogicalType,
10183    packed: Option<&Packed<'_>>,
10184    settling: &mut Settling,
10185) -> Result<Option<Vec<u8>>> {
10186    let (Some(width), Some(data)) = (plain_width(ty), flat.data()) else { return Ok(None) };
10187    let Some(values) = widened(data) else { return Ok(None) };
10188    let plain = values.len().saturating_mul(width);
10189    let best = match packed {
10190        // The tag, the base, the word count and the words, which is what the codec 2 branch writes.
10191        Some(packed) => plain.min(21 + size_of_val(packed.words())),
10192        None => plain,
10193    };
10194    let out = settling.encode(&values)?;
10195    Ok((out.len() < best).then_some(out))
10196}
10197
10198/// How often the parts of one column in one stripe search the cascade again, in parts.
10199///
10200/// A stripe is 64 parts, so this is four searches a stripe where there were 64. The search is
10201/// what the cascade costs: on ClickBench `hits` the integer cascade was about a tenth of the load's
10202/// CPU and nearly all of it under `encode_pages`, trying six trees on every part to keep the one
10203/// the part before had kept.
10204const SEARCH_EVERY: usize = 16;
10205
10206/// What the parts of one column in one stripe have settled on in the integer cascade.
10207///
10208/// One of these per column per stripe, used in part order, so what a part comes out as depends on
10209/// the stripe and not on which thread wrote it or on how many there were.
10210#[derive(Debug, Default)]
10211struct Settling {
10212    /// The shape of the last part that was searched, with what its top level offered, its length
10213    /// and its row count, which is the size a replay is held to.
10214    shape: Option<Shape>,
10215    /// Parts replayed since that search.
10216    since: usize,
10217}
10218
10219impl Settling {
10220    /// A part's integers through the cascade, replaying the settled shape where there is one.
10221    ///
10222    /// The replay is kept when it held and came out no more than a quarter bigger a row than the
10223    /// part the shape was searched on. Past that the column has changed under it and the part is
10224    /// searched. A replay that stopped fitting partway has already searched from where it stopped,
10225    /// so its shape is taken as the new one rather than searched a second time.
10226    fn encode(&mut self, values: &[i64]) -> Result<Vec<u8>> {
10227        if let Some(shape) = self.shape.as_ref().filter(|_| self.since < SEARCH_EVERY) {
10228            let replay = chooser::Replay::new(&shape.kinds, &Fixed).expecting(&shape.offered);
10229            let out = integer::encode_with(values, &replay)?;
10230            if !replay.held() {
10231                self.settle(&out, values.len(), replay.first_offered())?;
10232                return Ok(out);
10233            }
10234            let grown = (out.len() as u128) * (shape.rows as u128) * 4;
10235            if grown <= (shape.len as u128) * (values.len() as u128) * 5 {
10236                self.since += 1;
10237                return Ok(out);
10238            }
10239        }
10240        // A replay of nothing is the search, and says what the top level offered on the way.
10241        let search = chooser::Replay::new(&[], &Fixed);
10242        let out = integer::encode_with(values, &search)?;
10243        self.settle(&out, values.len(), search.first_offered())?;
10244        Ok(out)
10245    }
10246
10247    fn settle(&mut self, out: &[u8], rows: usize, offered: Vec<integer::Kind>) -> Result<()> {
10248        let kinds = integer::shape(out)?;
10249        self.shape = Some(Shape { kinds, offered, len: out.len().max(1), rows: rows.max(1) });
10250        self.since = 0;
10251        Ok(())
10252    }
10253}
10254
10255/// A searched part's cascade, what its top level was offered, and what it came to.
10256#[derive(Debug)]
10257struct Shape {
10258    kinds: Vec<integer::Kind>,
10259    offered: Vec<integer::Kind>,
10260    len: usize,
10261    rows: usize,
10262}
10263
10264/// A part's dictionary codes through the integer cascade, or `None` when the cascade did not pay.
10265///
10266/// Until now this stream was a `u32` a row with nothing asked of it, and on ClickBench that was
10267/// 400,185,326 bytes for every one of the 28 varchar columns, the same count for `URL` as for a
10268/// column holding the empty string in nearly every row. Codes are dense integers counted from zero
10269/// and a part holds 1024 of them, which is the shape frame of reference is best at, and a column
10270/// with one value everywhere comes back a constant costing nothing per row rather than four bytes.
10271///
10272/// The result is taken only when it is smaller than the plain form. A cascade is allowed to come
10273/// out larger on a part whose codes are genuinely wide, `URL` has about sixty million distinct
10274/// values, and there is no reason to pay for the decode when it does.
10275/// A varchar page as one FSST layer, or `None` when it did not pay.
10276///
10277/// Until now a varchar page that neither the global dictionary nor the per page dictionary claimed
10278/// was written out raw: four bytes of offset a row and then the bytes. That is the right answer for
10279/// a page of values with nothing in common and the wrong one for a page of English, and a column of
10280/// comments is the case this exists for.
10281///
10282/// One layer and not the full string cascade, which is what the payload blocks of a global
10283/// dictionary go through. The cascade is a search: it encodes the page under every candidate it has
10284/// and recurses into the integer cascade for the lengths of each one, and on TPC-H `orders` that
10285/// took the write from 6.9 s to 48.3 s. It reads back no faster than the dictionary it replaced
10286/// either, 1.807 G instructions against 1.810 G for `select o_comment from orders`, because
10287/// unpicking a nest of layers a value at a time costs what the dictionary's payload block decode
10288/// cost. Raw pages of the same column read in 0.686 G, which says the whole of the difference is
10289/// what the page has to be put back together from.
10290///
10291/// FSST alone keeps most of what the cascade found and gives all of that back. Decoding it is one
10292/// pass over the payload into one buffer, the values are laid end to end in it the way the raw form
10293/// already lays them out, and what the reader hands a chunk is views over that buffer.
10294///
10295/// The page dictionary gets first refusal because it is cheaper still, and it wins on a page whose
10296/// values repeat. What is left for this is the page whose values mostly do not, which is exactly the
10297/// page that was being written raw.
10298///
10299/// Taken only when it comes out smaller than the raw form, so a page of incompressible values pays
10300/// nothing at read time for having been offered.
10301fn text_compressed(flat: &Vector) -> Result<Option<Vec<u8>>> {
10302    let mut values: Vec<&[u8]> = Vec::with_capacity(flat.len());
10303    let mut payload = 0_usize;
10304    for row in 0..flat.len() {
10305        // bytes_at: the rows were checked for UTF-8 on the way in, and checking them again here
10306        // was most of what the loop cost.
10307        let text = flat.bytes_at(row).unwrap_or(b"");
10308        payload = payload.saturating_add(text.len());
10309        values.push(text);
10310    }
10311    // What codec 0 writes for a varchar page: an offset a row and one more, then the payload.
10312    let plain = (flat.len() + 1).saturating_mul(4).saturating_add(payload);
10313    let Some(out) = string::encode_only(string::Kind::Fsst, &values)? else {
10314        return Ok(None);
10315    };
10316    Ok((out.len() < plain).then_some(out))
10317}
10318
10319fn encoded_codes(codes: &[u32]) -> Result<Option<Vec<u8>>> {
10320    let wide: Vec<i64> = codes.iter().map(|code| i64::from(*code)).collect();
10321    let coded = integer::encode_with(&wide, &Codes)?;
10322    let plain = codes.len().saturating_mul(size_of::<u32>());
10323    Ok((coded.len() < plain).then_some(coded))
10324}
10325
10326/// The validity of a page, which is a flag and then, when some rows are null and some are not, a
10327/// bit a row with the valid ones set.
10328fn push_validity(out: &mut Vec<u8>, flat: &Vector) {
10329    let flag = match flat.validity() {
10330        Validity::AllValid => 0,
10331        Validity::AllInvalid => 1,
10332        Validity::Mask(_) => 2,
10333    };
10334    out.push(flag);
10335    if flag == 2 {
10336        for group in (0..flat.len()).step_by(8) {
10337            let mut bits = 0_u8;
10338            for bit in 0..8 {
10339                if group + bit < flat.len() && !flat.is_null_at(group + bit) {
10340                    bits |= 1 << bit;
10341                }
10342            }
10343            out.push(bits);
10344        }
10345    }
10346}
10347
10348/// One part of a column coded against its global dictionary as a page, from the codes and the
10349/// validity [`push_validity`] wrote for it.
10350///
10351/// The codes go through the integer cascade when that comes out smaller than four bytes a code,
10352/// which on a column that repeats itself it nearly always does, and are written as they are when it
10353/// does not.
10354fn coded_page(codes: &[u32], validity: &[u8]) -> Result<Vec<u8>> {
10355    let coded = encoded_codes(codes)?;
10356    let mut out = Vec::with_capacity(
10357        1 + validity.len() + coded.as_ref().map_or(size_of_val(codes), Vec::len),
10358    );
10359    out.push(if coded.is_some() { 4 } else { 3 });
10360    out.extend_from_slice(validity);
10361    match coded {
10362        Some(coded) => out.extend_from_slice(&coded),
10363        None => {
10364            for &code in codes {
10365                put_u32(&mut out, code);
10366            }
10367        }
10368    }
10369    Ok(out)
10370}
10371
10372/// One part of one column as a page, for every column that is not coded against a global
10373/// dictionary. Those are built by [`coded_page`] from codes [`prepare`] handed out.
10374fn encode(vector: &Vector, settling: &mut Settling) -> Result<Vec<u8>> {
10375    let ty = vector.logical_type();
10376    // flatten: the file writer needs a uniform scalar page and does it once per loaded chunk.
10377    let flat = vector.flatten()?;
10378    let mut out = Vec::new();
10379    let dictionary = if coded_type(ty) { string_dictionary(&flat)? } else { None };
10380    let compressed_text =
10381        if dictionary.is_none() && coded_type(ty) { text_compressed(&flat)? } else { None };
10382    let packed_vector = if dictionary.is_none() { Some(flat.bit_packed()?) } else { None };
10383    let packed = packed_vector.as_ref().and_then(Vector::packed_parts);
10384    // Only where nothing else has claimed the page, which is the plain integer case. A packed part
10385    // is still on the table because the cascade has to beat it too: the bit pack takes a part only
10386    // when it halves it, so a column that shrinks by a third was coming out whole.
10387    let cascade =
10388        if dictionary.is_none() { cascaded(&flat, ty, packed.as_ref(), settling)? } else { None };
10389    out.push(if cascade.is_some() {
10390        5
10391    } else if dictionary.is_some() {
10392        1
10393    } else if compressed_text.is_some() {
10394        6
10395    } else if packed.is_some() {
10396        2
10397    } else {
10398        0
10399    });
10400    push_validity(&mut out, &flat);
10401    if let Some(cascade) = cascade {
10402        out.extend_from_slice(&cascade);
10403        return Ok(out);
10404    }
10405    if let Some(dictionary) = dictionary {
10406        out.extend_from_slice(&dictionary);
10407        return Ok(out);
10408    }
10409    if let Some(compressed_text) = compressed_text {
10410        out.extend_from_slice(&compressed_text);
10411        return Ok(out);
10412    }
10413    if let Some(packed) = packed {
10414        if packed.offset() != 0 {
10415            return Err(invalid("writer received a sliced packed vector"));
10416        }
10417        out.push(u8::try_from(packed.width()).map_err(|_| invalid("packed width overflow"))?);
10418        out.extend_from_slice(&packed.base().to_le_bytes());
10419        put_u32(
10420            &mut out,
10421            u32::try_from(packed.words().len()).map_err(|_| invalid("too many packed words"))?,
10422        );
10423        for word in packed.words() {
10424            put_u64(&mut out, *word);
10425        }
10426        return Ok(out);
10427    }
10428    let data = flat.data().ok_or_else(|| invalid("scalar column did not flatten"))?;
10429    match (ty, data) {
10430        (LogicalType::TinyInt, Data::Int8(values)) => {
10431            for value in &**values {
10432                out.extend_from_slice(&value.to_le_bytes());
10433            }
10434        }
10435        (LogicalType::UTinyInt, Data::UInt8(values)) => {
10436            for value in &**values {
10437                out.extend_from_slice(&value.to_le_bytes());
10438            }
10439        }
10440        (LogicalType::SmallInt, Data::Int16(values)) => {
10441            for value in &**values {
10442                out.extend_from_slice(&value.to_le_bytes());
10443            }
10444        }
10445        (LogicalType::USmallInt, Data::UInt16(values)) => {
10446            for value in &**values {
10447                out.extend_from_slice(&value.to_le_bytes());
10448            }
10449        }
10450        (LogicalType::UInteger, Data::UInt32(values)) => {
10451            for value in &**values {
10452                out.extend_from_slice(&value.to_le_bytes());
10453            }
10454        }
10455        (LogicalType::UBigInt, Data::UInt64(values)) => {
10456            for value in &**values {
10457                out.extend_from_slice(&value.to_le_bytes());
10458            }
10459        }
10460        (LogicalType::Integer | LogicalType::Date, Data::Int32(values)) => {
10461            for value in &**values {
10462                out.extend_from_slice(&value.to_le_bytes());
10463            }
10464        }
10465        (
10466            LogicalType::BigInt
10467            | LogicalType::Timestamp
10468            | LogicalType::Time
10469            | LogicalType::TimeTz
10470            | LogicalType::TimestampTz
10471            | LogicalType::TimestampS
10472            | LogicalType::TimestampMs
10473            | LogicalType::TimestampNs,
10474            Data::Int64(values),
10475        ) => {
10476            for value in &**values {
10477                out.extend_from_slice(&value.to_le_bytes());
10478            }
10479        }
10480        // A hugeint and a uuid are both the 128 bit lane, and a uuid's bits are the ones the rest of
10481        // the engine already carries it in, so nothing about the value changes on the way down.
10482        (LogicalType::HugeInt | LogicalType::Uuid, Data::Int128(values)) => {
10483            for value in &**values {
10484                out.extend_from_slice(&value.to_le_bytes());
10485            }
10486        }
10487        (LogicalType::UHugeInt, Data::UInt128(values)) => {
10488            for value in &**values {
10489                out.extend_from_slice(&value.to_le_bytes());
10490            }
10491        }
10492        // Plainly, in the IEEE bytes. The integer encodings do not apply to a float and none of the
10493        // float codecs is worth having before somebody has measured a corpus of them.
10494        (LogicalType::Float, Data::Float32(values)) => {
10495            for value in &**values {
10496                out.extend_from_slice(&value.to_le_bytes());
10497            }
10498        }
10499        (LogicalType::Double, Data::Float64(values)) => {
10500            for value in &**values {
10501                out.extend_from_slice(&value.to_le_bytes());
10502            }
10503        }
10504        // Three counts and not one number. Months, days and microseconds stay apart on disk because
10505        // they are apart in the value: a month is not a fixed number of days and a day is not a
10506        // fixed number of microseconds, which is the whole reason the type has three fields.
10507        (LogicalType::Interval, Data::Interval(values)) => {
10508            for (months, days, micros) in &**values {
10509                out.extend_from_slice(&months.to_le_bytes());
10510                out.extend_from_slice(&days.to_le_bytes());
10511                out.extend_from_slice(&micros.to_le_bytes());
10512            }
10513        }
10514        (LogicalType::Boolean, Data::Bool(values)) => {
10515            for value in &**values {
10516                out.push(u8::from(*value));
10517            }
10518        }
10519        // The unscaled integer and nothing else. Scale is a property of the column and it is in the
10520        // directory already, so writing it a value at a time would be paying for it twice.
10521        (LogicalType::Decimal { .. }, Data::Int16(values)) => {
10522            for value in &**values {
10523                out.extend_from_slice(&value.to_le_bytes());
10524            }
10525        }
10526        (LogicalType::Decimal { .. }, Data::Int32(values)) => {
10527            for value in &**values {
10528                out.extend_from_slice(&value.to_le_bytes());
10529            }
10530        }
10531        (LogicalType::Decimal { .. }, Data::Int64(values)) => {
10532            for value in &**values {
10533                out.extend_from_slice(&value.to_le_bytes());
10534            }
10535        }
10536        (LogicalType::Decimal { .. }, Data::Int128(values)) => {
10537            for value in &**values {
10538                out.extend_from_slice(&value.to_le_bytes());
10539            }
10540        }
10541        // A blob and a bit string go down the way a varchar does, because the layout is the same
10542        // one: an offset a value and then the bytes. What is not the same is that nothing here may
10543        // read the payload as text, which is why this arm asks the column for bytes rather than for
10544        // a string, and why the codecs above that do read text are all asked of a varchar by name.
10545        (LogicalType::Varchar | LogicalType::Blob | LogicalType::Bit, Data::Varlen(values)) => {
10546            let mut bytes = Vec::new();
10547            put_u32(&mut out, 0);
10548            for row in 0..vector.len() {
10549                let value = values.bytes(row).ok_or_else(|| invalid("string view is invalid"))?;
10550                bytes.extend_from_slice(value);
10551                put_u32(
10552                    &mut out,
10553                    u32::try_from(bytes.len())
10554                        .map_err(|_| invalid("string payload exceeds 4GiB"))?,
10555                );
10556            }
10557            out.extend_from_slice(&bytes);
10558        }
10559        _ => return Err(Error::not_implemented(format!("native page for {ty}"))),
10560    }
10561    Ok(out)
10562}
10563
10564fn put_varint(out: &mut Vec<u8>, mut value: u32) {
10565    while value >= 0x80 {
10566        out.push((value as u8 & 0x7f) | 0x80);
10567        value >>= 7;
10568    }
10569    out.push(value as u8);
10570}
10571
10572/// The distinct codes of one part, which is what a stripe's membership index is merged from.
10573fn unique_codes(codes: &[u32]) -> Vec<u32> {
10574    let mut unique = codes.to_vec();
10575    unique.sort_unstable();
10576    unique.dedup();
10577    unique
10578}
10579
10580/// The union of the sorted distinct codes of every part in a stripe.
10581///
10582/// Pairwise up a tree rather than one long list concatenated and sorted. Both are the same order of
10583/// work on paper and the tree is the one that does not sort what is already in order: sixty four
10584/// sorted lists become one in six passes over the values.
10585fn merged_codes(lists: Vec<Vec<u32>>) -> Vec<u32> {
10586    let mut lists = lists;
10587    while lists.len() > 1 {
10588        let mut next = Vec::with_capacity(lists.len().div_ceil(2));
10589        for pair in lists.chunks(2) {
10590            match pair {
10591                [left, right] => next.push(merged_pair(left, right)),
10592                [only] => next.push(only.clone()),
10593                _ => {}
10594            }
10595        }
10596        lists = next;
10597    }
10598    lists.pop().unwrap_or_default()
10599}
10600
10601fn merged_pair(left: &[u32], right: &[u32]) -> Vec<u32> {
10602    let mut out = Vec::with_capacity(left.len().saturating_add(right.len()));
10603    let mut at = 0;
10604    let mut to = 0;
10605    while at < left.len() && to < right.len() {
10606        match left[at].cmp(&right[to]) {
10607            Ordering::Less => {
10608                out.push(left[at]);
10609                at += 1;
10610            }
10611            Ordering::Greater => {
10612                out.push(right[to]);
10613                to += 1;
10614            }
10615            Ordering::Equal => {
10616                out.push(left[at]);
10617                at += 1;
10618                to += 1;
10619            }
10620        }
10621    }
10622    out.extend_from_slice(&left[at..]);
10623    out.extend_from_slice(&right[to..]);
10624    out
10625}
10626
10627/// The widest bounds and the total null count of a stripe, from the bounds of its parts.
10628///
10629/// A bound that is missing from any part is missing from the stripe, because a missing bound means
10630/// nothing is known and a stripe that holds an unknown cannot claim one.
10631fn merged_range(ranges: impl Iterator<Item = Range>) -> Range {
10632    let mut merged = Range::default();
10633    let mut first = true;
10634    for range in ranges {
10635        merged.nulls = merged.nulls.saturating_add(range.nulls);
10636        // Both of these have to survive every part, so one part that could not say anything makes
10637        // the stripe unable to say it either. A sum is dropped on overflow rather than wrapped,
10638        // which leaves the stripe with exact ends and no total, which is a true thing to say.
10639        merged.sum = match (merged.sum.take(), range.sum) {
10640            (Some(held), Some(next)) if !first => held.checked_add(next),
10641            (_, next) if first => next,
10642            _ => None,
10643        };
10644        merged.exact = if first { range.exact } else { merged.exact && range.exact };
10645        if first {
10646            merged.low = range.low;
10647            merged.high = range.high;
10648            first = false;
10649            continue;
10650        }
10651        merged.low = match (merged.low.take(), range.low) {
10652            (Some(held), Some(next)) => Some(held.smaller(next)),
10653            _ => None,
10654        };
10655        merged.high = match (merged.high.take(), range.high) {
10656            (Some(held), Some(next)) => Some(held.larger(next)),
10657            _ => None,
10658        };
10659    }
10660    merged
10661}
10662
10663/// One stripe's sieves for one column: the part count, a length for each part, then their bytes.
10664///
10665/// One page for the whole stripe rather than one per part, because a part's sieve is a few hundred
10666/// bytes and sixty four of those are sixty four directory entries and sixty four reads for something
10667/// a scan walks straight through. A part with no sieve writes a length of zero and costs four bytes.
10668/// `bound` cut down to [`PART_BOUND_BYTES`], still a bound of the side it was.
10669///
10670/// A prefix of a string sorts at or before the string, so cutting one down leaves a low end that is
10671/// still a low end. A high end has to go the other way, so the cut prefix is stepped up at the last
10672/// byte that can carry it, and a prefix of nothing but `0xFF` has no such byte and gives up the
10673/// bound rather than claiming one that is too small. Anything that is not a string is already a
10674/// fixed width and is left alone.
10675fn shortened(bound: Option<Bound>, high: bool) -> Option<Bound> {
10676    match bound {
10677        Some(Bound::Bytes(mut value)) if value.len() > PART_BOUND_BYTES => {
10678            value.truncate(PART_BOUND_BYTES);
10679            if !high {
10680                return Some(Bound::Bytes(value));
10681            }
10682            while let Some(last) = value.pop() {
10683                if last < u8::MAX {
10684                    value.push(last + 1);
10685                    return Some(Bound::Bytes(value));
10686                }
10687            }
10688            None
10689        }
10690        other => other,
10691    }
10692}
10693
10694/// The ranges of one column's parts of one stripe, as a page.
10695///
10696/// The two ends and the null count, and not `exact` or the total. Those two answer a `MIN` or a
10697/// `SUM` out of the directory, and the directory already answers those per stripe, where the same
10698/// number costs sixty times less to keep. What a part range is for is skipping the part, and
10699/// skipping needs the ends. So a range read back from here says it is not exact, which is true of a
10700/// string end that was cut down anyway.
10701fn encode_part_ranges(ranges: &[Range]) -> Result<Vec<u8>> {
10702    let mut out = Vec::new();
10703    put_u32(
10704        &mut out,
10705        u32::try_from(ranges.len()).map_err(|_| invalid("too many parts in a stripe"))?,
10706    );
10707    for range in ranges {
10708        put_bound(&mut out, shortened(range.low.clone(), false).as_ref())?;
10709        put_bound(&mut out, shortened(range.high.clone(), true).as_ref())?;
10710        put_u32(&mut out, u32::try_from(range.nulls).map_err(|_| invalid("null count overflow"))?);
10711    }
10712    Ok(out)
10713}
10714
10715/// The ranges one encoded page holds, one entry per part of the stripe.
10716fn decode_part_ranges(bytes: &[u8]) -> Result<Vec<Range>> {
10717    let mut cur = Cursor::new(bytes);
10718    let parts = cur.u32()? as usize;
10719    let mut out = Vec::new();
10720    for _ in 0..parts {
10721        let low = cur.bound()?;
10722        let high = cur.bound()?;
10723        let nulls = cur.u32()? as usize;
10724        out.push(Range { low, high, nulls, exact: false, sum: None });
10725    }
10726    Ok(out)
10727}
10728
10729fn encode_sieves<'a>(sieves: impl Iterator<Item = &'a Option<Sieve>>) -> Result<Vec<u8>> {
10730    let held: Vec<&Option<Sieve>> = sieves.collect();
10731    let mut out = Vec::new();
10732    put_u32(
10733        &mut out,
10734        u32::try_from(held.len()).map_err(|_| invalid("too many parts in a stripe"))?,
10735    );
10736    for sieve in &held {
10737        let length = sieve.as_ref().map_or(0, Sieve::len);
10738        put_u32(&mut out, u32::try_from(length).map_err(|_| invalid("sieve length overflow"))?);
10739    }
10740    // flatten: a part with no sieve wrote a length of zero above and contributes no bytes here.
10741    for sieve in held.into_iter().flatten() {
10742        out.extend_from_slice(&sieve.to_bytes());
10743    }
10744    Ok(out)
10745}
10746
10747/// The sieves one encoded page holds, one entry per part of the stripe.
10748///
10749/// A part whose bytes are not a sieve this version understands comes back as `None`, which is a part
10750/// that gets read. That is how a file written by a later version of the sieve stays readable rather
10751/// than being a corrupt page.
10752fn decode_sieves(bytes: &[u8]) -> Result<Vec<Option<Sieve>>> {
10753    let parts = u32::from_le_bytes(
10754        bytes
10755            .get(..4)
10756            .ok_or_else(|| invalid("sieve page is truncated"))?
10757            .try_into()
10758            .map_err(|_| invalid("sieve page is truncated"))?,
10759    ) as usize;
10760    let mut lengths = Vec::with_capacity(parts);
10761    for part in 0..parts {
10762        let at = 4 + part * 4;
10763        let field = bytes.get(at..at + 4).ok_or_else(|| invalid("sieve page is truncated"))?;
10764        lengths.push(u32::from_le_bytes(
10765            field.try_into().map_err(|_| invalid("sieve page is truncated"))?,
10766        ) as usize);
10767    }
10768    let mut at = 4 + parts * 4;
10769    let mut out = Vec::with_capacity(parts);
10770    for length in lengths {
10771        if length == 0 {
10772            out.push(None);
10773            continue;
10774        }
10775        let end = at.checked_add(length).ok_or_else(|| invalid("sieve page is truncated"))?;
10776        let field = bytes.get(at..end).ok_or_else(|| invalid("sieve page is truncated"))?;
10777        out.push(Sieve::from_bytes(field));
10778        at = end;
10779    }
10780    if at != bytes.len() {
10781        return Err(invalid("sieve page has trailing bytes"));
10782    }
10783    Ok(out)
10784}
10785
10786/// One stripe's membership index: the code count and then the codes as ascending deltas.
10787///
10788/// The codes have to be sorted and distinct already, which is what [`unique_codes`] and
10789/// [`merged_codes`] hand over. Anything else decodes as different codes, so neither of those two is
10790/// a step a caller can skip.
10791fn encode_membership(unique: &[u32]) -> Vec<u8> {
10792    let mut out = Vec::with_capacity(unique.len().saturating_mul(2).saturating_add(5));
10793    put_varint(&mut out, u32::try_from(unique.len()).unwrap_or(u32::MAX));
10794    let mut previous = 0;
10795    for (at, &code) in unique.iter().enumerate() {
10796        put_varint(&mut out, if at == 0 { code } else { code - previous });
10797        previous = code;
10798    }
10799    out
10800}
10801
10802fn take_varint(bytes: &[u8], at: &mut usize) -> Result<u32> {
10803    let mut value = 0_u32;
10804    for shift in (0..35).step_by(7) {
10805        let byte = *bytes.get(*at).ok_or_else(|| invalid("membership varint is truncated"))?;
10806        *at += 1;
10807        let part = u32::from(byte & 0x7f);
10808        if shift == 28 && part > 0x0f {
10809            return Err(invalid("membership varint overflow"));
10810        }
10811        value = value
10812            .checked_add(
10813                part.checked_shl(shift).ok_or_else(|| invalid("membership varint overflow"))?,
10814            )
10815            .ok_or_else(|| invalid("membership varint overflow"))?;
10816        if byte & 0x80 == 0 {
10817            return Ok(value);
10818        }
10819    }
10820    Err(invalid("membership varint is too long"))
10821}
10822
10823fn decode_membership(bytes: &[u8]) -> Result<Vec<u32>> {
10824    let mut at = 0;
10825    let count = take_varint(bytes, &mut at)? as usize;
10826    let mut codes = Vec::with_capacity(count);
10827    let mut previous = 0_u32;
10828    for index in 0..count {
10829        let delta = take_varint(bytes, &mut at)?;
10830        let code = if index == 0 {
10831            delta
10832        } else {
10833            previous.checked_add(delta).ok_or_else(|| invalid("membership code overflow"))?
10834        };
10835        if index > 0 && code <= previous {
10836            return Err(invalid("membership codes are not increasing"));
10837        }
10838        codes.push(code);
10839        previous = code;
10840    }
10841    if at != bytes.len() {
10842        return Err(invalid("membership page has trailing bytes"));
10843    }
10844    Ok(codes)
10845}
10846
10847fn string_dictionary(vector: &Vector) -> Result<Option<Vec<u8>>> {
10848    let mut by_text = HashMap::new();
10849    let mut values = Vec::new();
10850    let mut codes = Vec::with_capacity(vector.len());
10851    let mut plain_bytes = 0_usize;
10852    for row in 0..vector.len() {
10853        let text = vector.bytes_at(row).unwrap_or(b"");
10854        plain_bytes = plain_bytes.saturating_add(text.len());
10855        let code = match by_text.get(text) {
10856            Some(&code) => code,
10857            None => {
10858                let code = u32::try_from(values.len())
10859                    .map_err(|_| invalid("too many dictionary values"))?;
10860                by_text.insert(text, code);
10861                values.push(text);
10862                code
10863            }
10864        };
10865        codes.push(code);
10866    }
10867    let dictionary_bytes = values.iter().map(|value| value.len()).sum::<usize>();
10868    let encoded = 8_usize
10869        .saturating_add((values.len() + 1).saturating_mul(4))
10870        .saturating_add(dictionary_bytes)
10871        .saturating_add(codes.len().saturating_mul(4));
10872    let plain = (vector.len() + 1).saturating_mul(4).saturating_add(plain_bytes);
10873    if encoded >= plain {
10874        return Ok(None);
10875    }
10876    let mut out = Vec::with_capacity(encoded);
10877    put_u32(
10878        &mut out,
10879        u32::try_from(values.len()).map_err(|_| invalid("too many dictionary values"))?,
10880    );
10881    put_u32(
10882        &mut out,
10883        u32::try_from(dictionary_bytes).map_err(|_| invalid("dictionary payload exceeds 4GiB"))?,
10884    );
10885    let mut offset = 0_u32;
10886    put_u32(&mut out, offset);
10887    for value in &values {
10888        offset = offset
10889            .checked_add(
10890                u32::try_from(value.len()).map_err(|_| invalid("dictionary value is too long"))?,
10891            )
10892            .ok_or_else(|| invalid("dictionary payload exceeds 4GiB"))?;
10893        put_u32(&mut out, offset);
10894    }
10895    for value in values {
10896        out.extend_from_slice(value);
10897    }
10898    for code in codes {
10899        put_u32(&mut out, code);
10900    }
10901    Ok(Some(out))
10902}
10903
10904/// The room one closing column takes under [`CLOSE_BYTES`], given back when dropped.
10905struct Room<'a, T> {
10906    state: &'a Mutex<(T, usize)>,
10907    finished: &'a Condvar,
10908    bytes: usize,
10909}
10910
10911impl<T> Drop for Room<'_, T> {
10912    fn drop(&mut self) {
10913        let mut held = self.state.lock().unwrap_or_else(PoisonError::into_inner);
10914        held.1 -= self.bytes;
10915        drop(held);
10916        self.finished.notify_all();
10917    }
10918}
10919
10920/// One column's work at the end of a load, as [`Writer::close_columns`] schedules it.
10921enum Closing<'a> {
10922    /// A numeric column's frequencies, and whether to count its distinct values exactly.
10923    Numeric {
10924        column: usize,
10925        counted: bool,
10926    },
10927    Dictionary {
10928        index: usize,
10929        dictionary: &'a GlobalDictionary,
10930    },
10931}
10932
10933/// What one [`Closing`] came back with, by column.
10934enum Closed {
10935    Numeric(usize, (Option<FrequencySummary>, Option<u64>)),
10936    Dictionary(usize, ClosedDictionary),
10937}
10938
10939/// What [`Writer::close_dictionary`] builds for one column and [`Writer::close`] writes.
10940struct ClosedDictionary {
10941    /// `None` for a demoted dictionary, which holds only some of the column. See [`DEMOTED`].
10942    distinct: Option<u64>,
10943    frequencies: Option<FrequencySummary>,
10944    texts: Vec<Option<Vec<u8>>>,
10945    hosts: Option<host::HostSummary>,
10946    encoded: EncodedDictionary,
10947    /// The bytes of the column's payload blocks, which are already in the file.
10948    payload: u64,
10949}
10950
10951struct EncodedDictionary {
10952    index: Vec<u8>,
10953    ranks: Vec<u8>,
10954    grams: Vec<u8>,
10955}
10956
10957/// Sorts codes into the byte order of the values they name, eight bytes of depth at a time.
10958///
10959/// # What the shape of the data does to a comparison sort
10960///
10961/// Distinct values against distinct prefixes, on the eight million row `hits`:
10962///
10963/// ```text
10964///   distinct   first 8   first 16   first 32   column
10965///  2,266,417        50      8,892    232,630   URL
10966///  2,346,025        49      8,534    204,060   Referer
10967///  1,357,764    81,362    348,340    861,579   Title
10968/// ```
10969///
10970/// Two and a quarter million URLs have fifty distinct first eight bytes between them, because they
10971/// all begin `http://` and then a host and there are not many hosts. So a sort that leads with
10972/// those eight bytes settles almost nothing on `URL` and `Referer`, whatever the comment on it used
10973/// to say, and almost every pair falls through to a comparison of whole values that agree for most
10974/// of their length. `Title` is free text and separates at eight bytes, which is why the design
10975/// looked right when it was written.
10976///
10977/// # What is done about it
10978///
10979/// Sort on eight bytes of the value at the current depth, held beside the code, and then take each
10980/// run that those eight bytes leave tied and sort it again on the next eight. A value is fetched
10981/// from the payload once per eight bytes of depth rather than once per comparison, and the sort
10982/// itself runs over an array of integers that is in cache rather than over pointers into a payload
10983/// that is hundreds of megabytes.
10984///
10985/// That is the whole trick, and it matters because the payload touch is the expensive part. The
10986/// bytes themselves are nearly free once the line is in cache, so reading eight at a time and
10987/// throwing away the ones that were not needed beats going back for each one.
10988///
10989/// # Why the length has to be carried
10990///
10991/// The eight bytes are padded with zero when the value has fewer than eight left, and a zero byte
10992/// can appear in a value, so equal keys do not mean equal bytes. What is true is that a value which
10993/// ran out inside the window is a prefix of any other value with the same key, and a prefix sorts
10994/// first, so how many of the eight bytes were real is the tie break and nothing further is needed.
10995/// A run is only worth another pass when all eight were real, because otherwise the run is one
10996/// value: a dictionary holds a value once.
10997fn sort_by_value<'a>(codes: &mut [u32], values: impl Fn(u32) -> &'a [u8]) {
10998    let mut work = vec![(0, codes.len(), 0)];
10999    let mut keyed: Vec<(u64, u8, u32)> = Vec::new();
11000    while let Some((from, to, depth)) = work.pop() {
11001        let part = &mut codes[from..to];
11002        keyed.clear();
11003        keyed.extend(part.iter().map(|&code| {
11004            let value = values(code);
11005            let rest = value.get(depth..).unwrap_or_default();
11006            (head(rest), rest.len().min(8) as u8, code)
11007        }));
11008        keyed.sort_unstable();
11009        for (slot, entry) in part.iter_mut().zip(keyed.iter()) {
11010            *slot = entry.2;
11011        }
11012        let mut start = 0;
11013        while start < keyed.len() {
11014            let (key, taken, _) = keyed[start];
11015            let mut end = start + 1;
11016            while end < keyed.len() && keyed[end].0 == key && keyed[end].1 == taken {
11017                end += 1;
11018            }
11019            if taken == 8 && end - start > 1 {
11020                work.push((from + start, from + end, depth + 8));
11021            }
11022            start = end;
11023        }
11024    }
11025}
11026
11027/// How few codes are worth sorting on more than one thread.
11028const PARALLEL_SORT_MIN: usize = 1 << 16;
11029
11030/// How many buckets a thread gets in [`sort_by_value_across`], so that a thread that drew a slow
11031/// bucket is not what the others wait for.
11032const BUCKETS_PER_WORKER: usize = 4;
11033
11034/// How many sampled codes stand for each bucket when the splitters are picked.
11035const SAMPLES_PER_BUCKET: usize = 32;
11036
11037/// [`sort_by_value`] over `workers` threads, with the same answer.
11038///
11039/// A sample sort. A sample of the codes is sorted and cut into as many equal runs as there are
11040/// buckets, and the values at the cuts are the splitters. Every code goes to the bucket its value
11041/// falls in by a binary search of the splitters, the buckets are laid end to end in splitter order,
11042/// and each bucket is then sorted on its own by whichever thread takes it. Every value in a bucket
11043/// sorts after every value in the bucket before, so the buckets sorted one by one are the codes
11044/// sorted.
11045///
11046/// The answer is the one [`sort_by_value`] gives down to the order of equal values, not only the
11047/// order of different ones. A global dictionary holds each value once, so there are none, but the
11048/// sort does not rely on it: equal values land in the same bucket in code order, which is the order
11049/// [`sort_by_value`] leaves them in, since the code is the last thing it sorts on.
11050///
11051/// On the 10m ClickBench sample the close sorts five columns of one to three and a half million
11052/// distinct values, one column at a time, and until this each sort ran on one thread while the
11053/// other thirty one waited for it.
11054fn sort_by_value_across<'a>(
11055    codes: &mut [u32],
11056    values: impl Fn(u32) -> &'a [u8] + Sync,
11057    workers: usize,
11058) {
11059    if workers <= 1 || codes.len() < PARALLEL_SORT_MIN {
11060        sort_by_value(codes, values);
11061        return;
11062    }
11063    let buckets = workers * BUCKETS_PER_WORKER;
11064    let wanted = buckets * SAMPLES_PER_BUCKET;
11065    let mut sample = (0..wanted).map(|at| codes[at * codes.len() / wanted]).collect::<Vec<_>>();
11066    sort_by_value(&mut sample, &values);
11067    let splitters =
11068        (1..buckets).map(|cut| values(sample[cut * sample.len() / buckets])).collect::<Vec<_>>();
11069    let values = &values;
11070    let splitters = &splitters;
11071    let per = codes.len().div_ceil(workers);
11072    // Which bucket each code goes to, a run of the codes per thread.
11073    let places = std::thread::scope(|scope| {
11074        codes
11075            .chunks(per)
11076            .map(|run| {
11077                scope.spawn(move || {
11078                    run.iter()
11079                        .map(|&code| {
11080                            let value = values(code);
11081                            splitters.partition_point(|splitter| *splitter <= value) as u32
11082                        })
11083                        .collect::<Vec<_>>()
11084                })
11085            })
11086            .collect::<Vec<_>>()
11087            .into_iter()
11088            .flat_map(|handle| {
11089                handle.join().unwrap_or_else(|panic| std::panic::resume_unwind(panic))
11090            })
11091            .collect::<Vec<_>>()
11092    });
11093    let mut starts = vec![0_usize; buckets + 1];
11094    for &place in &places {
11095        starts[place as usize + 1] += 1;
11096    }
11097    for bucket in 0..buckets {
11098        starts[bucket + 1] += starts[bucket];
11099    }
11100    let mut laid = vec![0_u32; codes.len()];
11101    let mut next = starts.clone();
11102    for (&code, &place) in codes.iter().zip(&places) {
11103        laid[next[place as usize]] = code;
11104        next[place as usize] += 1;
11105    }
11106    drop(places);
11107    let mut runs = Vec::with_capacity(buckets);
11108    let mut rest = laid.as_mut_slice();
11109    for bucket in 0..buckets {
11110        let (run, after) = rest.split_at_mut(starts[bucket + 1] - starts[bucket]);
11111        runs.push(run);
11112        rest = after;
11113    }
11114    // The largest buckets first, since they are taken from the back.
11115    runs.sort_by_key(|run| run.len());
11116    let queue = Mutex::new(runs);
11117    std::thread::scope(|scope| {
11118        for _ in 0..workers {
11119            scope.spawn(|| {
11120                loop {
11121                    let taken = queue.lock().unwrap_or_else(PoisonError::into_inner).pop();
11122                    let Some(run) = taken else { break };
11123                    sort_by_value(run, values);
11124                }
11125            });
11126        }
11127    });
11128    codes.copy_from_slice(&laid);
11129}
11130
11131/// The first eight bytes of a value as an integer that sorts the way the bytes sort.
11132fn head(bytes: &[u8]) -> u64 {
11133    let mut word = [0; 8];
11134    let take = bytes.len().min(8);
11135    word[..take].copy_from_slice(&bytes[..take]);
11136    u64::from_be_bytes(word)
11137}
11138
11139/// One column's dictionary page, which is its index and its sorted order.
11140///
11141/// The payload is not in it. Its blocks are in the file already, written as each was encoded, and
11142/// `places` says where, in block order. With `scattered` set the index records each block's start
11143/// and length, so a reader can find one wherever it went.
11144///
11145/// `scattered` false lays the blocks out the way a file written before format 26 has them, one
11146/// behind the next with only the ends recorded. Nothing in the writer asks for that any more. It is
11147/// kept because [`open_global_dictionary`] still reads those files and a reading path that nothing
11148/// can produce is a reading path nothing tests.
11149fn encode_global_dictionary(
11150    dictionary: &GlobalDictionary,
11151    order: &[(u64, u32)],
11152    places: &[Placed],
11153    scattered: bool,
11154) -> Result<EncodedDictionary> {
11155    let values = dictionary.values();
11156    if order.len() != values {
11157        return Err(invalid("global dictionary order does not cover its values"));
11158    }
11159    let blocks = values.div_ceil(TEXT_PAYLOAD_VALUES);
11160    if places.len() != blocks {
11161        return Err(invalid("global dictionary payload is not the blocks it says it is"));
11162    }
11163    if dictionary.grams.len() != blocks {
11164        return Err(invalid("global dictionary signatures do not cover its blocks"));
11165    }
11166    let (ranks, rank_ends) = encode_ranks(order, code_width(values))?;
11167    let rank_blocks = values.div_ceil(TEXT_RANK_BLOCK);
11168    let offset_bits = offset_width(&dictionary.ends);
11169    let payload_words = if scattered { 3 } else { 2 };
11170    let index_len = DICTIONARY_HEADER
11171        .checked_add(offset_bytes(values, offset_bits))
11172        .and_then(|len| len.checked_add(blocks.checked_mul(payload_words * 8)?))
11173        .and_then(|len| len.checked_add(rank_blocks.checked_mul(16)?))
11174        .and_then(|len| len.checked_add(8))
11175        .ok_or_else(|| invalid("global dictionary index length overflow"))?;
11176    let mut index = Vec::with_capacity(index_len);
11177    put_u32(
11178        &mut index,
11179        u32::try_from(values).map_err(|_| invalid("global dictionary has too many values"))?,
11180    );
11181    put_u32(&mut index, TEXT_PAYLOAD_VALUES as u32);
11182    put_u32(
11183        &mut index,
11184        u32::try_from(blocks).map_err(|_| invalid("global dictionary has too many blocks"))?,
11185    );
11186    let flag = (if scattered { DICTIONARY_SCATTERED } else { 0 })
11187        | DICTIONARY_GRAMS
11188        | DICTIONARY_WIDE_GRAMS;
11189    put_u32(&mut index, offset_bits as u32 | flag);
11190    encode_offsets(&dictionary.ends, offset_bits, &mut index)?;
11191    // Where each block is and how long it is, so a reader can find one. The stored blocks are
11192    // shorter than the decoded ones and by a different amount each, so their lengths are the one
11193    // thing the offsets above no longer say, and where they start is no longer arithmetic on the
11194    // block before once a block is written the moment it is encoded.
11195    let mut end = 0_u64;
11196    for place in places {
11197        if scattered {
11198            put_u64(&mut index, place.start);
11199            put_u64(&mut index, place.length);
11200        } else {
11201            end = end
11202                .checked_add(place.length)
11203                .ok_or_else(|| invalid("global dictionary payload overflow"))?;
11204            put_u64(&mut index, end);
11205        }
11206    }
11207    for place in places {
11208        put_u64(&mut index, place.hash);
11209    }
11210    // The same two lists for the sorted order. A rank block is packed at whatever width its own
11211    // heads need, so where one ends is no longer arithmetic on the block number.
11212    if rank_ends.len() != rank_blocks {
11213        return Err(invalid("global dictionary order is not the blocks it says it is"));
11214    }
11215    for end in &rank_ends {
11216        put_u64(&mut index, *end);
11217    }
11218    let mut at = 0_usize;
11219    for end in &rank_ends {
11220        let end = usize::try_from(*end).map_err(|_| invalid("global dictionary order overflow"))?;
11221        put_u64(&mut index, checksum(&ranks[at..end]));
11222        at = end;
11223    }
11224    let gram_len = blocks
11225        .checked_mul(TEXT_GRAM_BYTES)
11226        .ok_or_else(|| invalid("global dictionary signature count overflow"))?;
11227    let mut grams = Vec::with_capacity(gram_len);
11228    for block in &dictionary.grams {
11229        grams.extend_from_slice(block);
11230    }
11231    put_u64(&mut index, checksum(&grams));
11232    if index.len() != index_len {
11233        return Err(invalid("global dictionary index is not the length it was laid out for"));
11234    }
11235    Ok(EncodedDictionary { index, ranks, grams })
11236}
11237
11238/// How many blocks of the payload the shape is settled on.
11239///
11240/// Eight blocks is 8,192 values, which is the sample `chooser::Sampled` draws and is that size for
11241/// the same reason. They are spread across the dictionary rather than taken off the front, because
11242/// a dictionary is in the order values were first seen and the front of it is the first morsel of
11243/// the load.
11244const PAYLOAD_SAMPLE_BLOCKS: usize = 8;
11245
11246/// The shapes the payload encoder picks between.
11247///
11248/// Narrow on purpose. The exhaustive search encodes every candidate at every level and runs at two
11249/// to six megabytes a second on this data, which over the twelve gigabytes of dictionary `hits`
11250/// carries is about an hour of processor time, so it cannot be what a load does. Each of these
11251/// settles the outer level and the one below it, which is where almost all of that hour goes, and
11252/// leaves the levels under them to the exhaustive search where the chunks are small enough for it
11253/// to cost nothing.
11254///
11255/// Measured on the five ClickBench columns that have a dictionary worth the name, at 1,024 values a
11256/// block, against the exhaustive search over the same blocks:
11257///
11258/// | column | exhaustive | FRONT then LZ | LZ then FSST | LZ then PLAIN |
11259/// |---|---|---|---|---|
11260/// | 2 | 2.923 at 4.3 MB/s | 2.587 at 21.2 | 2.593 at 36.1 | 2.538 at 53.6 |
11261/// | 13 | 3.093 at 3.1 | 3.029 at 36.4 | 2.921 at 35.7 | 2.770 at 82.9 |
11262/// | 14 | 2.330 at 2.1 | 2.283 at 24.3 | 2.213 at 23.5 | 2.113 at 67.6 |
11263/// | 39 | 2.459 at 5.3 | 2.147 at 10.6 | 2.145 at 29.3 | 2.088 at 43.1 |
11264/// | 56 | 4.694 at 6.3 | 4.381 at 51.0 | 4.172 at 50.6 | 3.983 at 86.8 |
11265///
11266/// The best of the three per column is 98 percent of the exhaustive ratio for a tenth of the time.
11267/// `FSST` and `PLAIN` on their own are in the list as a floor rather than to win. `FSST` is the
11268/// right answer for text that does not share prefixes with its neighbours, and `PLAIN` is there so
11269/// that a column nothing compresses is found out in the sample and written at a gigabyte a second
11270/// rather than searched for an answer that does not exist.
11271fn payload_shapes() -> Vec<chooser::Settled> {
11272    let integers = vec![integer::Kind::Packed];
11273    [
11274        vec![string::Kind::Front, string::Kind::Lz],
11275        vec![string::Kind::Lz, string::Kind::Fsst],
11276        vec![string::Kind::Lz, string::Kind::Plain],
11277        vec![string::Kind::Fsst],
11278        vec![string::Kind::Plain],
11279    ]
11280    .into_iter()
11281    .map(|strings| chooser::Settled::new(strings, integers.clone()))
11282    .collect()
11283}
11284
11285/// Syncs the file, and counts the sync and how long it took as a publish wait when a load is being
11286/// profiled.
11287///
11288/// A wait rather than time, because the time is already in the publish span around it. What the
11289/// wait columns add is how much of publish was the device, which on the WSL2 disk of the gaming PC
11290/// is most of it: a sync there costs about two milliseconds (see `rudb_device_card`).
11291fn synced(file: &dyn rudb_io::File, profile: Option<&LoadProfile>) -> Result<()> {
11292    let started = profile.map(|_| std::time::Instant::now());
11293    file.sync()?;
11294    if let (Some(profile), Some(started)) = (profile, started) {
11295        profile.waited(
11296            Stage::Publish,
11297            u64::try_from(started.elapsed().as_nanos()).unwrap_or(u64::MAX),
11298        );
11299    }
11300    Ok(())
11301}
11302
11303/// One sealed dictionary block on its way to being encoded outside the writer's lock.
11304///
11305/// Handed out by the merge that sealed it and encoded with the pages of the same stripe. See
11306/// [`GlobalDictionary::hand_out`].
11307#[derive(Debug)]
11308pub(crate) struct Unencoded {
11309    column: usize,
11310    at: usize,
11311    ends: Vec<u32>,
11312    bytes: Vec<u8>,
11313    shape: chooser::Settled,
11314}
11315
11316impl Unencoded {
11317    /// The encoded block and its signature.
11318    pub(crate) fn encode(&self) -> Result<EncodedBlock> {
11319        let values = block_values(&self.ends, &self.bytes);
11320        Ok((string::encode_with(&values, &self.shape)?, block_grams(&values)))
11321    }
11322
11323    /// The column and the block number the encoded block goes back to.
11324    pub(crate) fn place(&self) -> (usize, usize) {
11325        (self.column, self.at)
11326    }
11327}
11328
11329/// One encoded dictionary block and the signature of the values in it.
11330///
11331/// Boxed because it is carried around in things that are otherwise small.
11332pub(crate) type EncodedBlock = (Vec<u8>, Box<[u8; TEXT_GRAM_BYTES]>);
11333
11334/// The conservative four-byte substring signature of one block's values.
11335fn block_grams(values: &[&[u8]]) -> Box<[u8; TEXT_GRAM_BYTES]> {
11336    let mut grams = Box::new([0_u8; TEXT_GRAM_BYTES]);
11337    for value in values {
11338        for gram in value.windows(4) {
11339            for bit in gram_bits(gram, TEXT_GRAM_BYTES) {
11340                grams[bit / 8] |= 1 << (bit % 8);
11341            }
11342        }
11343    }
11344    grams
11345}
11346
11347/// The values of one block, given where each of them ends relative to the block.
11348fn block_values<'a>(ends: &[u32], bytes: &'a [u8]) -> Vec<&'a [u8]> {
11349    let mut out = Vec::with_capacity(ends.len());
11350    let mut from = 0;
11351    for &to in ends {
11352        out.push(&bytes[from..to as usize]);
11353        from = to as usize;
11354    }
11355    out
11356}
11357
11358/// Encodes every block still raw at the end of a load: the part block each column ends on and,
11359/// for a column too small to have settled a shape, every block it has.
11360///
11361/// Across threads, the way [`encode_ready`] does it. This ran one column at a time on the thread
11362/// closing the table, and a column that never settled a shape encodes each block by trying every
11363/// candidate, so on a million rows of `hits` it was most of the load's CPU on one core.
11364fn finish_dictionaries(dictionaries: &mut [Option<GlobalDictionary>]) -> Result<()> {
11365    for dictionary in dictionaries.iter_mut().flatten() {
11366        if !dictionary.early.is_empty() {
11367            return Err(Error::internal("a dictionary block handed out never came back"));
11368        }
11369        dictionary.seal_rest();
11370        dictionary.settle_rest()?;
11371    }
11372    encode_waiting(dictionaries)?;
11373    // A block handed out and never given back leaves a gap nothing above would notice when it was
11374    // the last one, so the count is checked against the values as well.
11375    if dictionaries
11376        .iter()
11377        .flatten()
11378        .any(|dictionary| dictionary.encoded() != dictionary.values().div_ceil(TEXT_PAYLOAD_VALUES))
11379    {
11380        return Err(Error::internal("a dictionary block handed out never came back"));
11381    }
11382    Ok(())
11383}
11384
11385/// Encodes the waiting blocks of every dictionary across threads, and appends them to their columns
11386/// in order.
11387fn encode_waiting(dictionaries: &mut [Option<GlobalDictionary>]) -> Result<()> {
11388    let jobs = dictionaries
11389        .iter()
11390        .enumerate()
11391        .flat_map(|(column, held)| {
11392            (0..held.as_ref().map_or(0, |held| held.waiting.len())).map(move |at| (column, at))
11393        })
11394        .collect::<Vec<_>>();
11395    if jobs.is_empty() {
11396        return Ok(());
11397    }
11398    let one = |column: usize, at: usize| -> Result<(usize, usize, EncodedBlock)> {
11399        let held = dictionaries[column].as_ref().ok_or_else(|| Error::internal("no dictionary"))?;
11400        Ok((column, at, held.encode_waiting(at)?))
11401    };
11402    let workers = std::thread::available_parallelism()
11403        .map_or(1, usize::from)
11404        .min(MAX_FREQUENCY_WORKERS)
11405        .min(jobs.len());
11406    let made = if workers <= 1 {
11407        jobs.iter().map(|&(column, at)| one(column, at)).collect::<Result<Vec<_>>>()?
11408    } else {
11409        let next = AtomicUsize::new(0);
11410        let jobs = &jobs;
11411        let pieces = std::thread::scope(|scope| {
11412            (0..workers)
11413                .map(|_| {
11414                    scope.spawn(|| {
11415                        let mut mine = Vec::new();
11416                        loop {
11417                            let job = next.fetch_add(1, Atomic::Relaxed);
11418                            let Some(&(column, at)) = jobs.get(job) else { break };
11419                            mine.push(one(column, at)?);
11420                        }
11421                        Ok(mine)
11422                    })
11423                })
11424                .collect::<Vec<_>>()
11425                .into_iter()
11426                .map(|handle| {
11427                    handle
11428                        .join()
11429                        .map_err(|_| Error::internal("a dictionary encode worker panicked"))?
11430                })
11431                .collect::<Result<Vec<_>>>()
11432        })?;
11433        pieces.into_iter().flatten().collect()
11434    };
11435    let mut done: Vec<Vec<(usize, EncodedBlock)>> =
11436        (0..dictionaries.len()).map(|_| Vec::new()).collect();
11437    for (column, at, bytes) in made {
11438        done[column].push((at, bytes));
11439    }
11440    for (column, mut made) in done.into_iter().enumerate() {
11441        if made.is_empty() {
11442            continue;
11443        }
11444        let Some(held) = dictionaries[column].as_mut() else { continue };
11445        made.sort_by_key(|(at, _)| *at);
11446        let waiting = std::mem::take(&mut held.waiting);
11447        for ((at, _), (_, block)) in waiting.into_iter().zip(made) {
11448            if held.encoded() != at {
11449                return Err(Error::internal("a dictionary block was encoded out of order"));
11450            }
11451            held.push_block(block);
11452        }
11453    }
11454    Ok(())
11455}
11456
11457/// Which of [`payload_shapes`] comes out smallest over a sample of the blocks.
11458///
11459/// Every shape is encoded over the same sample and the smallest wins, which is the exhaustive
11460/// search moved up a level: over shapes of a column rather than over candidates of a chunk. The
11461/// sample is spread across the dictionary so that the first and last blocks are both in it, because
11462/// a dictionary written in first seen order has its common values at the front and its long tail at
11463/// the back, and those do not compress alike. Which blocks those are is
11464/// [`GlobalDictionary::seal`]'s to decide, because by the time this is called the rest of them have
11465/// been encoded and the raw bytes are gone.
11466fn settle_shape(sample: &[Vec<&[u8]>]) -> Result<chooser::Settled> {
11467    let mut best: Option<(chooser::Settled, usize)> = None;
11468    for shape in payload_shapes() {
11469        let mut size = 0;
11470        for block in sample {
11471            size += string::encode_with(block, &shape)?.len();
11472        }
11473        if best.as_ref().is_none_or(|(_, smallest)| size < *smallest) {
11474            best = Some((shape, size));
11475        }
11476    }
11477    best.map(|(shape, _)| shape)
11478        .ok_or_else(|| invalid("no shape applies to a global dictionary payload"))
11479}
11480
11481/// The sorted order laid out the way a reader reads it, in blocks of [`TEXT_RANK_BLOCK`] entries.
11482///
11483/// Each block holds its heads first and then its codes, rather than pairing them, because a search
11484/// asks for a head at every probe and for a code about once a search. Keeping the heads together
11485/// means a probe touches eight bytes of a block rather than twelve spread over it, and the last few
11486/// probes of a search, which are the ones that land in the same block, touch the same cache line.
11487fn encode_ranks(order: &[(u64, u32)], code_bits: usize) -> Result<(Vec<u8>, Vec<u64>)> {
11488    let mut out = Vec::with_capacity(order.len() * 4);
11489    let mut ends = Vec::with_capacity(order.len().div_ceil(TEXT_RANK_BLOCK));
11490    let mut heads = Vec::with_capacity(TEXT_RANK_BLOCK);
11491    let mut codes = Vec::with_capacity(TEXT_RANK_BLOCK);
11492    for block in order.chunks(TEXT_RANK_BLOCK) {
11493        // The order is sorted by value and a head is a prefix of a value, so the heads of a block
11494        // rise, the smallest is the first and the largest is the last.
11495        let base = block.first().map_or(0, |&(head, _)| head);
11496        let span = block.last().map_or(0, |&(head, _)| head.wrapping_sub(base));
11497        let width = (u64::BITS - span.leading_zeros()) as usize;
11498        heads.clear();
11499        codes.clear();
11500        for &(head, code) in block {
11501            heads.push(head.wrapping_sub(base));
11502            codes.push(u64::from(code));
11503        }
11504        put_u64(&mut out, base);
11505        out.push(width as u8);
11506        bitpack::pack_tail(&heads, width, &mut out)
11507            .map_err(|_| invalid("global dictionary heads do not pack"))?;
11508        bitpack::pack_tail(&codes, code_bits, &mut out)
11509            .map_err(|_| invalid("global dictionary codes do not pack"))?;
11510        ends.push(out.len() as u64);
11511    }
11512    Ok((out, ends))
11513}
11514
11515/// Opens a column's global dictionary, which reads its index and none of its payload.
11516///
11517/// `keep_budget` is how many decoded payload bytes this dictionary may hold on to, and every
11518/// caller bar the test of the ceiling passes [`TEXT_KEEP_BUDGET`]. It is a parameter rather than
11519/// the constant read where it is used because a test of a ceiling that cannot be moved has to build
11520/// a quarter of a gigabyte of dictionary to reach it.
11521fn open_global_dictionary(
11522    file: Arc<File>,
11523    page: Page,
11524    ty: &LogicalType,
11525    keep_budget: usize,
11526) -> Result<Vector> {
11527    if !coded_type(ty) {
11528        return Err(invalid("global dictionary belongs to a non-string column"));
11529    }
11530    let mut header = [0; DICTIONARY_HEADER];
11531    read_at(&file, page.offset, &mut header)?;
11532    let count = u32::from_le_bytes(header[0..4].try_into().expect("four bytes")) as usize;
11533    let per_block = u32::from_le_bytes(header[4..8].try_into().expect("four bytes")) as usize;
11534    let blocks = u32::from_le_bytes(header[8..12].try_into().expect("four bytes")) as usize;
11535    let width = u32::from_le_bytes(header[12..16].try_into().expect("four bytes"));
11536    let scattered = width & DICTIONARY_SCATTERED != 0;
11537    let has_grams = width & DICTIONARY_GRAMS != 0;
11538    let gram_width =
11539        if width & DICTIONARY_WIDE_GRAMS != 0 { TEXT_GRAM_BYTES } else { NARROW_GRAM_BYTES };
11540    let offset_bits = (width & !DICTIONARY_FLAGS) as usize;
11541    if per_block != TEXT_PAYLOAD_VALUES {
11542        return Err(invalid("global dictionary block width differs"));
11543    }
11544    if blocks != count.div_ceil(TEXT_PAYLOAD_VALUES) {
11545        return Err(invalid("global dictionary block count differs from its value count"));
11546    }
11547    if offset_bits > u32::BITS as usize {
11548        return Err(invalid("global dictionary packs offsets past a payload"));
11549    }
11550    let offset_len = offset_bytes(count, offset_bits);
11551    // The sorted order is kept out of the index on purpose. The index is read and checksummed in
11552    // full the moment the column is first touched, and the order is half again the size of the
11553    // offsets, so putting it there would make every query that reads a string column pay for a
11554    // search that most of them never make.
11555    let ranks = count;
11556    let rank_blocks = ranks.div_ceil(TEXT_RANK_BLOCK);
11557    // Three words a payload block, for where it starts, how long it is and what it hashes to, or
11558    // two of them on a file that has the blocks back to back and needs no start. Two a rank block
11559    // either way, since those are still one run.
11560    let payload_words = if scattered { 3 } else { 2 };
11561    let hash_len = blocks
11562        .checked_mul(payload_words * 8)
11563        .and_then(|len| len.checked_add(rank_blocks.checked_mul(16)?))
11564        .and_then(|len| len.checked_add(usize::from(has_grams) * 8))
11565        .ok_or_else(|| invalid("global dictionary block count overflow"))?;
11566    let gram_len = if has_grams {
11567        blocks
11568            .checked_mul(gram_width)
11569            .ok_or_else(|| invalid("global dictionary signature count overflow"))?
11570    } else {
11571        0
11572    };
11573    let index_len = DICTIONARY_HEADER
11574        .checked_add(offset_len)
11575        .and_then(|len| len.checked_add(hash_len))
11576        .ok_or_else(|| invalid("global dictionary header overflow"))?;
11577    if index_len > page.length as usize {
11578        return Err(invalid("global dictionary offset index exceeds its page"));
11579    }
11580    let mut index = vec![0; index_len];
11581    index[..DICTIONARY_HEADER].copy_from_slice(&header);
11582    read_at(&file, page.offset + DICTIONARY_HEADER as u64, &mut index[DICTIONARY_HEADER..])?;
11583    if checksum(&index) != page.hash {
11584        return Err(invalid("global dictionary index checksum differs"));
11585    }
11586    let word_end = index_len - usize::from(has_grams) * 8;
11587    let gram_hash = has_grams
11588        .then(|| u64::from_le_bytes(index[word_end..index_len].try_into().expect("eight bytes")));
11589    let mut words = index[DICTIONARY_HEADER + offset_len..word_end]
11590        .chunks_exact(8)
11591        .map(|part| u64::from_le_bytes(part.try_into().expect("eight bytes")))
11592        .collect::<Vec<_>>();
11593    let mut rest = words.split_off(blocks * payload_words);
11594    let rank_hashes = rest.split_off(rank_blocks);
11595    let rank_ends = rest;
11596    // A rank block packs its heads at whatever width its own values need, so its length is no longer
11597    // arithmetic on the block number and the reader has to be told where each one ends.
11598    if rank_ends.windows(2).any(|pair| pair[0] >= pair[1]) {
11599        return Err(invalid("global dictionary order blocks do not rise"));
11600    }
11601    let rank_len = usize::try_from(rank_ends.last().copied().unwrap_or_default())
11602        .map_err(|_| invalid("global dictionary rank overflow"))?;
11603    let body_len = index_len
11604        .checked_add(rank_len)
11605        .ok_or_else(|| invalid("global dictionary header overflow"))?;
11606    if body_len > page.length as usize {
11607        return Err(invalid("global dictionary order exceeds its page"));
11608    }
11609    let gram_end = body_len
11610        .checked_add(gram_len)
11611        .ok_or_else(|| invalid("global dictionary signature length overflow"))?;
11612    if gram_end > page.length as usize {
11613        return Err(invalid("global dictionary signatures exceed their page"));
11614    }
11615    let grams = gram_hash.map(|hash| NativeGrams {
11616        start: page.offset + body_len as u64,
11617        length: gram_len,
11618        width: gram_width,
11619        hash,
11620        verdicts: Mutex::new(Vec::new()),
11621    });
11622    // The offsets stay where they were read, behind the header, rather than being copied out. On a
11623    // dictionary of millions of values they are megabytes, and a copy is as many fresh pages to
11624    // fault in again on a query that may want a handful of strings.
11625    let mut offsets = index;
11626    offsets.truncate(DICTIONARY_HEADER + offset_len);
11627    let hashes = words.split_off(blocks * (payload_words - 1));
11628    let (starts, lengths) = if scattered {
11629        let mut starts = Vec::with_capacity(blocks);
11630        let mut lengths = Vec::with_capacity(blocks);
11631        for pair in words.chunks_exact(2) {
11632            starts.push(pair[0]);
11633            lengths.push(pair[1]);
11634        }
11635        (starts, lengths)
11636    } else {
11637        // A file written before the blocks said where they were has them behind one another at the
11638        // end of the page, so the base is where the sorted order stops and each end is the start of
11639        // the one after it. Turning them round here is what lets everything below take one shape.
11640        let base = page.offset + gram_end as u64;
11641        let mut starts = Vec::with_capacity(blocks);
11642        let mut lengths = Vec::with_capacity(blocks);
11643        let mut at = 0_u64;
11644        for &end in &words {
11645            let len = end
11646                .checked_sub(at)
11647                .ok_or_else(|| invalid("global dictionary block ends before it starts"))?;
11648            starts.push(base + at);
11649            lengths.push(len);
11650            at = end;
11651        }
11652        (starts, lengths)
11653    };
11654    // What the offsets bound is the decoded payload, and what the page length counts is the stored
11655    // one, so on a format 26 file the block lengths adding up to the rest of the page is the one
11656    // thing that ties the index to the page. From format 27 the blocks are written during the load
11657    // and the page is only the index and the order, so there the most that can be said is that
11658    // every block is somewhere in the file past its header.
11659    let stored_len = page.length as u64 - gram_end as u64;
11660    if scattered && stored_len == 0 {
11661        let size = file.metadata().map_err(io)?.len();
11662        let inside = starts.iter().zip(&lengths).all(|(&start, &len)| {
11663            start >= HEADER && start.checked_add(len).is_some_and(|end| end <= size)
11664        });
11665        if !inside {
11666            return Err(invalid("global dictionary block lies outside the file"));
11667        }
11668    } else if lengths.iter().try_fold(0_u64, |sum, len| sum.checked_add(*len)) != Some(stored_len) {
11669        return Err(invalid("global dictionary blocks do not bound the payload"));
11670    }
11671    Vector::external_text(
11672        ty.clone(),
11673        Arc::new(NativeText {
11674            file,
11675            values: count,
11676            offsets,
11677            offset_bits,
11678            value_ends: OnceLock::new(),
11679            value_lens: OnceLock::new(),
11680            ends_asked: AtomicUsize::new(0),
11681            ranks,
11682            rank_at: page.offset + index_len as u64,
11683            rank_ends,
11684            rank_hashes,
11685            rank_blocks: (0..rank_blocks).map(|_| OnceLock::new()).collect(),
11686            code_bits: code_width(count),
11687            code_ranks: OnceLock::new(),
11688            starts,
11689            lengths,
11690            hashes,
11691            grams,
11692            blocks: (0..blocks).map(|_| OnceLock::new()).collect(),
11693            char_lens: (0..blocks).map(|_| OnceLock::new()).collect(),
11694            keep_budget,
11695            payload_kept: AtomicUsize::new(0),
11696            swept: (0..blocks).map(|_| AtomicBool::new(false)).collect(),
11697            visit_dropped: AtomicUsize::new(0),
11698            searched: Mutex::new(HashMap::new()),
11699        }),
11700    )
11701}
11702
11703/// What a stored page is, without decoding a value out of it.
11704///
11705/// Two layers, and both of them belong in the answer. The codec byte at the front of every page is
11706/// the format's own choice, and it is what says whether the column came back as codes into a table
11707/// wide dictionary, as a bit packed page, as an encoding cascade or as the bytes themselves. Under
11708/// the cascade codecs there is a second choice the encoder made per chunk, and that is what
11709/// [`integer::describe`] and [`string::describe`] already write out as `DICT(PACKED, PACKED)`.
11710///
11711/// This mirrors the tags [`decode`] reads and has to be kept beside it. A page whose header this
11712/// cannot walk comes back as text rather than as an error, because a caller asking what a file
11713/// looks like is usually asking because something is wrong with it, and a report that stops at the
11714/// first bad page is a report that says nothing about the other nine hundred.
11715fn page_encoding(ty: &LogicalType, rows: usize, bytes: &[u8]) -> String {
11716    /// The page header is the codec, the validity tag and, for a page that stores a mask, the mask.
11717    fn cascade_at(rows: usize, bytes: &[u8]) -> Result<(u8, usize)> {
11718        let mut cur = Cursor::new(bytes);
11719        let codec = cur.u8()?;
11720        if cur.u8()? == 2 {
11721            cur.take(rows.div_ceil(8))?;
11722        }
11723        Ok((codec, cur.at))
11724    }
11725    let Ok((codec, at)) = cascade_at(rows, bytes) else {
11726        return "UNREADABLE".to_string();
11727    };
11728    let tail = &bytes[at..];
11729    let described = |described: Result<String>| described.unwrap_or_else(|_| "UNREADABLE".into());
11730    match codec {
11731        0 => match ty {
11732            LogicalType::Varchar | LogicalType::Blob => "PLAIN".to_string(),
11733            _ => "FIXED".to_string(),
11734        },
11735        1 => "DICT(PLAIN)".to_string(),
11736        2 => "FOR+BITPACK".to_string(),
11737        3 => "TABLE DICT".to_string(),
11738        4 => format!("TABLE DICT({})", described(integer::describe(tail))),
11739        5 => described(integer::describe(tail)),
11740        6 => described(string::describe(tail)),
11741        other => format!("CODEC {other}"),
11742    }
11743}
11744
11745/// Selected stable dictionary codes from one page.
11746///
11747/// Pair-frequency construction needs at most the bounded heavy-hitter rows. Reading those code
11748/// positions directly avoids materializing every code in each part that contains a candidate.
11749fn decode_selected_stable_codes(
11750    rows: usize,
11751    bytes: &[u8],
11752    positions: &[usize],
11753    out: &mut Vec<Option<u32>>,
11754) -> Result<bool> {
11755    if positions.windows(2).any(|pair| pair[0] >= pair[1])
11756        || positions.last().is_some_and(|&position| position >= rows)
11757    {
11758        return Err(invalid("selected code positions are not sorted and in range"));
11759    }
11760    let mut cur = Cursor::new(bytes);
11761    let codec = cur.u8()?;
11762    if codec != 3 && codec != 4 {
11763        return Ok(false);
11764    }
11765    let flag = cur.u8()?;
11766    let mask = match flag {
11767        0 | 1 => None,
11768        2 => {
11769            let at = cur.at;
11770            let len = rows.div_ceil(8);
11771            cur.take(len)?;
11772            Some((at, len))
11773        }
11774        _ => return Err(invalid("page validity tag differs")),
11775    };
11776    let valid = |row: usize| match flag {
11777        0 => true,
11778        1 => false,
11779        2 => mask.is_some_and(|(at, _)| bytes[at + row / 8] >> (row % 8) & 1 == 1),
11780        _ => unreachable!("the validity tag was checked"),
11781    };
11782    if codec == 4 {
11783        let wide = integer::decode_selected(&bytes[cur.at..], positions)?;
11784        for (&row, code) in positions.iter().zip(wide) {
11785            let code = u32::try_from(code).map_err(|_| invalid("code is not a code"))?;
11786            out.push(valid(row).then_some(code));
11787        }
11788        return Ok(true);
11789    }
11790    let codes_at = cur.at;
11791    let codes_len = rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?;
11792    cur.take(codes_len)?;
11793    if cur.at != bytes.len() {
11794        return Err(invalid("global code page has trailing bytes"));
11795    }
11796    let codes = &bytes[codes_at..codes_at + codes_len];
11797    for &row in positions {
11798        let at = row.checked_mul(4).ok_or_else(|| invalid("dictionary code offset overflow"))?;
11799        let code = u32::from_le_bytes(
11800            codes[at..at + 4].try_into().map_err(|_| invalid("dictionary code is truncated"))?,
11801        );
11802        out.push(valid(row).then_some(code));
11803    }
11804    Ok(true)
11805}
11806
11807/// [`decode`] of only the rows at `positions`, which rise.
11808///
11809/// A compressed text page decompresses only those rows, see [`string::decode_flat_at`], and checks
11810/// only those rows are text. Every other page is decoded whole and gathered, since its values are
11811/// fixed width or its strings are shared through a dictionary, and there picking comes after.
11812fn decode_at(
11813    ty: &LogicalType,
11814    rows: usize,
11815    bytes: &[u8],
11816    global: Option<Arc<Vector>>,
11817    positions: &[u32],
11818) -> Result<Vector> {
11819    if positions.last().is_some_and(|&last| last as usize >= rows) {
11820        return Err(invalid("a position is past the end of the part"));
11821    }
11822    if bytes.first() != Some(&6) {
11823        return decode(ty, rows, bytes, global)?.gather(positions);
11824    }
11825    if !coded_type(ty) {
11826        return Err(invalid("compressed text codec belongs to a non-string page"));
11827    }
11828    let mut cur = Cursor::new(bytes);
11829    cur.u8()?;
11830    let validity = match cur.u8()? {
11831        0 => Validity::AllValid,
11832        1 => Validity::AllInvalid,
11833        2 => {
11834            let mask = cur.take(rows.div_ceil(8))?;
11835            Validity::from_iter(positions.len(), |at| {
11836                let row = positions[at] as usize;
11837                mask[row / 8] >> (row % 8) & 1 == 1
11838            })
11839        }
11840        _ => return Err(invalid("page validity tag differs")),
11841    };
11842    let (payload, ends) = string::decode_flat_at(&bytes[cur.at..], positions)?.into_parts();
11843    let mut values = StringColumn::over(Buffer::from_vec(payload).into_page());
11844    push_values(&mut values, ty, &ends)?;
11845    Ok(Vector::flat(ty.clone(), Data::Varlen(values))?.with_validity(validity))
11846}
11847
11848/// The values of a string or blob page, laid end to end in the page's payload from its start, each
11849/// ending where `ends` says. A varchar is checked for text on the way in, once over the whole run,
11850/// and a blob or a bit string is not, since neither ever claimed to hold any.
11851fn push_values(values: &mut StringColumn, ty: &LogicalType, ends: &[usize]) -> Result<()> {
11852    if ty == &LogicalType::Varchar {
11853        return values.push_run_in_place(0, ends);
11854    }
11855    let mut start = 0;
11856    for &end in ends {
11857        let len = end
11858            .checked_sub(start)
11859            .ok_or_else(|| invalid("a string value ends before it starts"))?;
11860        values.push_bytes_in_place(start, len)?;
11861        start = end;
11862    }
11863    Ok(())
11864}
11865
11866fn decode(
11867    ty: &LogicalType,
11868    rows: usize,
11869    bytes: &[u8],
11870    global: Option<Arc<Vector>>,
11871) -> Result<Vector> {
11872    let mut cur = Cursor::new(bytes);
11873    let codec = cur.u8()?;
11874    let flag = cur.u8()?;
11875    let validity = match flag {
11876        0 => Validity::AllValid,
11877        1 => Validity::AllInvalid,
11878        2 => {
11879            let mask = cur.take(rows.div_ceil(8))?;
11880            Validity::from_iter(rows, |row| mask[row / 8] >> (row % 8) & 1 == 1)
11881        }
11882        _ => return Err(invalid("page validity tag differs")),
11883    };
11884    if codec == 1 {
11885        if !coded_type(ty) {
11886            return Err(invalid("dictionary codec belongs to a non-string page"));
11887        }
11888        let count = cur.u32()? as usize;
11889        let payload_len = cur.u32()? as usize;
11890        let offset_bytes = cur.take(
11891            (count + 1)
11892                .checked_mul(4)
11893                .ok_or_else(|| invalid("dictionary offset count overflow"))?,
11894        )?;
11895        let offsets = offset_bytes
11896            .chunks_exact(4)
11897            .map(|part| u32::from_le_bytes(part.try_into().expect("four bytes")))
11898            .collect::<Vec<_>>();
11899        let payload = cur.take(payload_len)?.to_vec();
11900        if offsets.first() != Some(&0)
11901            || offsets.last().copied().map(|last| last as usize) != Some(payload.len())
11902            || offsets.windows(2).any(|pair| pair[0] > pair[1])
11903        {
11904            return Err(invalid("dictionary offsets do not bound the payload"));
11905        }
11906        // A page, because every chunk cut out of this dictionary points at the same payload and a
11907        // page is what lets a cut be the views and nothing else.
11908        let mut strings = StringColumn::over(Buffer::from_vec(payload).into_page());
11909        let ends: Vec<usize> = offsets[1..].iter().map(|&end| end as usize).collect();
11910        push_values(&mut strings, ty, &ends)?;
11911        let mut codes = Vec::with_capacity(rows);
11912        for _ in 0..rows {
11913            codes.push(cur.u32()?);
11914        }
11915        if codes.iter().any(|code| *code as usize >= count) {
11916            return Err(invalid("dictionary code is out of range"));
11917        }
11918        if cur.at != bytes.len() {
11919            return Err(invalid("dictionary page has trailing bytes"));
11920        }
11921        let dictionary = Vector::flat(ty.clone(), Data::Varlen(strings))?;
11922        return Ok(Vector::dictionary(codes, dictionary)?.with_validity(validity));
11923    }
11924    if codec == 3 || codec == 4 {
11925        let dictionary = global.ok_or_else(|| invalid("global code page has no dictionary"))?;
11926        let codes = if codec == 4 {
11927            // The cascade holds the whole tail of the page and says how long it is itself, so the
11928            // check that nothing is left over is the one the decoder already makes.
11929            let wide = integer::decode(&bytes[cur.at..])?;
11930            if wide.len() != rows {
11931                return Err(invalid("encoded code page holds the wrong number of rows"));
11932            }
11933            // Checked once for the page rather than a fallible conversion per code. Every code a
11934            // file holds is inside a `u32` or the file is corrupt, so or the codes together and the
11935            // answer has a bit set above the low thirty two, or the sign bit, exactly when one of
11936            // them did. The or and the narrowing are two passes because each is then a vector
11937            // loop. As one loop with a `push` a code, the length check and the store kept it scalar,
11938            // and it was sixteen instructions a row on the two flag columns of q1.
11939            let seen = wide.iter().fold(0_i64, |seen, &code| seen | code);
11940            if seen < 0 || seen > i64::from(u32::MAX) {
11941                return Err(invalid("code is not a code"));
11942            }
11943            wide.iter().map(|&code| code as u32).collect()
11944        } else {
11945            let mut codes = Vec::with_capacity(rows);
11946            for _ in 0..rows {
11947                codes.push(cur.u32()?);
11948            }
11949            if cur.at != bytes.len() {
11950                return Err(invalid("global code page has trailing bytes"));
11951            }
11952            codes
11953        };
11954        let highest = codes.iter().copied().max();
11955        return Ok(Vector::stable_dictionary_validated(codes, dictionary, highest)?
11956            .with_validity(validity));
11957    }
11958    if codec == 6 {
11959        if !coded_type(ty) {
11960            return Err(invalid("compressed text codec belongs to a non-string page"));
11961        }
11962        // As codec 5, the layer holds the whole tail of the page and says how long it is itself.
11963        // It comes back as one buffer with the values laid end to end and where each one ends, which
11964        // is the raw form's layout, so what is left to do here is what codec 0 does.
11965        let (payload, ends) = string::decode_flat(&bytes[cur.at..])?.into_parts();
11966        if ends.len() != rows {
11967            return Err(invalid("compressed text page holds the wrong number of rows"));
11968        }
11969        // A page, because this is read once and handed out a chunk at a time, and a cut of a paged
11970        // payload moves views rather than bytes.
11971        let mut values = StringColumn::over(Buffer::from_vec(payload).into_page());
11972        push_values(&mut values, ty, &ends)?;
11973        return Ok(Vector::flat(ty.clone(), Data::Varlen(values))?.with_validity(validity));
11974    }
11975    if codec == 5 {
11976        // The cascade holds the whole tail of the page and says how long it is itself.
11977        let values = integer::decode(&bytes[cur.at..])?;
11978        if values.len() != rows {
11979            return Err(invalid("cascade page holds the wrong number of rows"));
11980        }
11981        let data = narrowed(ty, values)?;
11982        return Ok(Vector::flat(ty.clone(), data)?.with_validity(validity));
11983    }
11984    if codec == 2 {
11985        let width = u32::from(cur.u8()?);
11986        let base = i128::from_le_bytes(cur.take(16)?.try_into().expect("sixteen bytes"));
11987        let count = cur.u32()? as usize;
11988        let length = count.checked_mul(8).ok_or_else(|| invalid("packed page is too long"))?;
11989        let words: Vec<u64> = cur
11990            .take(length)?
11991            .chunks_exact(8)
11992            .map(|word| u64::from_le_bytes(word.try_into().expect("eight bytes")))
11993            .collect();
11994        if cur.at != bytes.len() {
11995            return Err(invalid("packed page has trailing bytes"));
11996        }
11997        return Ok(Vector::packed(ty.clone(), words, width, base, rows)?.with_validity(validity));
11998    }
11999    if codec != 0 {
12000        return Err(invalid("page codec is unknown"));
12001    }
12002    let data = match ty {
12003        LogicalType::TinyInt => {
12004            let values = cur.take(rows)?;
12005            Data::Int8(values.iter().map(|item| *item as i8).collect::<Vec<_>>().into())
12006        }
12007        LogicalType::UTinyInt => Data::UInt8(cur.take(rows)?.to_vec().into()),
12008        LogicalType::SmallInt => {
12009            let values =
12010                cur.take(rows.checked_mul(2).ok_or_else(|| invalid("page size overflow"))?)?;
12011            Data::Int16(
12012                values
12013                    .chunks_exact(2)
12014                    .map(|item| i16::from_le_bytes(item.try_into().expect("two bytes")))
12015                    .collect::<Vec<_>>()
12016                    .into(),
12017            )
12018        }
12019        LogicalType::USmallInt => {
12020            let values =
12021                cur.take(rows.checked_mul(2).ok_or_else(|| invalid("page size overflow"))?)?;
12022            Data::UInt16(
12023                values
12024                    .chunks_exact(2)
12025                    .map(|item| u16::from_le_bytes(item.try_into().expect("two bytes")))
12026                    .collect::<Vec<_>>()
12027                    .into(),
12028            )
12029        }
12030        LogicalType::UInteger => {
12031            let values =
12032                cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
12033            Data::UInt32(
12034                values
12035                    .chunks_exact(4)
12036                    .map(|item| u32::from_le_bytes(item.try_into().expect("four bytes")))
12037                    .collect::<Vec<_>>()
12038                    .into(),
12039            )
12040        }
12041        LogicalType::UBigInt => {
12042            let values =
12043                cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
12044            Data::UInt64(
12045                values
12046                    .chunks_exact(8)
12047                    .map(|item| u64::from_le_bytes(item.try_into().expect("eight bytes")))
12048                    .collect::<Vec<_>>()
12049                    .into(),
12050            )
12051        }
12052        LogicalType::Integer | LogicalType::Date => {
12053            let values =
12054                cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
12055            Data::Int32(
12056                values
12057                    .chunks_exact(4)
12058                    .map(|item| i32::from_le_bytes(item.try_into().expect("four bytes")))
12059                    .collect::<Vec<_>>()
12060                    .into(),
12061            )
12062        }
12063        LogicalType::BigInt
12064        | LogicalType::Timestamp
12065        | LogicalType::Time
12066        | LogicalType::TimeTz
12067        | LogicalType::TimestampTz
12068        | LogicalType::TimestampS
12069        | LogicalType::TimestampMs
12070        | LogicalType::TimestampNs => {
12071            let values =
12072                cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
12073            Data::Int64(
12074                values
12075                    .chunks_exact(8)
12076                    .map(|item| i64::from_le_bytes(item.try_into().expect("eight bytes")))
12077                    .collect::<Vec<_>>()
12078                    .into(),
12079            )
12080        }
12081        LogicalType::HugeInt | LogicalType::Uuid => {
12082            let values =
12083                cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
12084            Data::Int128(
12085                values
12086                    .chunks_exact(16)
12087                    .map(|item| i128::from_le_bytes(item.try_into().expect("sixteen bytes")))
12088                    .collect::<Vec<_>>()
12089                    .into(),
12090            )
12091        }
12092        LogicalType::UHugeInt => {
12093            let values =
12094                cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
12095            Data::UInt128(
12096                values
12097                    .chunks_exact(16)
12098                    .map(|item| u128::from_le_bytes(item.try_into().expect("sixteen bytes")))
12099                    .collect::<Vec<_>>()
12100                    .into(),
12101            )
12102        }
12103        LogicalType::Float => {
12104            let values =
12105                cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
12106            Data::Float32(
12107                values
12108                    .chunks_exact(4)
12109                    .map(|item| f32::from_le_bytes(item.try_into().expect("four bytes")))
12110                    .collect::<Vec<_>>()
12111                    .into(),
12112            )
12113        }
12114        LogicalType::Double => {
12115            let values =
12116                cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
12117            Data::Float64(
12118                values
12119                    .chunks_exact(8)
12120                    .map(|item| f64::from_le_bytes(item.try_into().expect("eight bytes")))
12121                    .collect::<Vec<_>>()
12122                    .into(),
12123            )
12124        }
12125        LogicalType::Interval => {
12126            let values =
12127                cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
12128            Data::Interval(
12129                values
12130                    .chunks_exact(16)
12131                    .map(|item| {
12132                        (
12133                            i32::from_le_bytes(item[..4].try_into().expect("four bytes")),
12134                            i32::from_le_bytes(item[4..8].try_into().expect("four bytes")),
12135                            i64::from_le_bytes(item[8..].try_into().expect("eight bytes")),
12136                        )
12137                    })
12138                    .collect::<Vec<_>>()
12139                    .into(),
12140            )
12141        }
12142        LogicalType::Boolean => {
12143            let values = cur.take(rows)?;
12144            if values.iter().any(|value| *value > 1) {
12145                return Err(invalid("boolean page has another value"));
12146            }
12147            Data::Bool(values.iter().map(|value| *value == 1).collect::<Vec<_>>().into())
12148        }
12149        // Whichever integer the declared width says, which is the mapping the rest of the engine
12150        // already uses for a decimal in memory.
12151        LogicalType::Decimal { .. } => match ty.physical() {
12152            PhysicalType::Int16 => {
12153                let values =
12154                    cur.take(rows.checked_mul(2).ok_or_else(|| invalid("page size overflow"))?)?;
12155                Data::Int16(
12156                    values
12157                        .chunks_exact(2)
12158                        .map(|item| i16::from_le_bytes(item.try_into().expect("two bytes")))
12159                        .collect::<Vec<_>>()
12160                        .into(),
12161                )
12162            }
12163            PhysicalType::Int32 => {
12164                let values =
12165                    cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
12166                Data::Int32(
12167                    values
12168                        .chunks_exact(4)
12169                        .map(|item| i32::from_le_bytes(item.try_into().expect("four bytes")))
12170                        .collect::<Vec<_>>()
12171                        .into(),
12172                )
12173            }
12174            PhysicalType::Int64 => {
12175                let values =
12176                    cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
12177                Data::Int64(
12178                    values
12179                        .chunks_exact(8)
12180                        .map(|item| i64::from_le_bytes(item.try_into().expect("eight bytes")))
12181                        .collect::<Vec<_>>()
12182                        .into(),
12183                )
12184            }
12185            _ => {
12186                let values =
12187                    cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
12188                Data::Int128(
12189                    values
12190                        .chunks_exact(16)
12191                        .map(|item| i128::from_le_bytes(item.try_into().expect("sixteen bytes")))
12192                        .collect::<Vec<_>>()
12193                        .into(),
12194                )
12195            }
12196        },
12197        LogicalType::Varchar | LogicalType::Blob | LogicalType::Bit => {
12198            let offset_bytes = cur
12199                .take((rows + 1).checked_mul(4).ok_or_else(|| invalid("offset count overflow"))?)?;
12200            let offsets = offset_bytes
12201                .chunks_exact(4)
12202                .map(|part| u32::from_le_bytes(part.try_into().expect("four bytes")))
12203                .collect::<Vec<_>>();
12204            let payload = cur.take(bytes.len() - cur.at)?.to_vec();
12205            if offsets.first() != Some(&0)
12206                || offsets.last().copied().map(|last| last as usize) != Some(payload.len())
12207                || offsets.windows(2).any(|pair| pair[0] > pair[1])
12208            {
12209                return Err(invalid("string offsets do not bound the payload"));
12210            }
12211            // A page for the reason the dictionary payload above is one: the page is read once and
12212            // handed out a chunk at a time, and a cut of a paged payload moves views rather than
12213            // bytes.
12214            //
12215            // A varchar is checked for text on the way in and a blob and a bit string are not,
12216            // because the second pair never claimed to hold any. Reading them through the checking
12217            // seam would refuse a column for holding exactly what it was told to hold.
12218            let mut values = StringColumn::over(Buffer::from_vec(payload).into_page());
12219            let ends: Vec<usize> = offsets[1..].iter().map(|&end| end as usize).collect();
12220            push_values(&mut values, ty, &ends)?;
12221            Data::Varlen(values)
12222        }
12223        _ => return Err(Error::not_implemented(format!("native page for {ty}"))),
12224    };
12225    if cur.at != bytes.len() {
12226        return Err(invalid("page has trailing bytes"));
12227    }
12228    Ok(Vector::flat(ty.clone(), data)?.with_validity(validity))
12229}
12230
12231#[cfg(test)]
12232mod tests {
12233    use std::fs::{self, OpenOptions};
12234    use std::io::{Seek, SeekFrom, Write};
12235    use std::path::PathBuf;
12236    use std::time::{SystemTime, UNIX_EPOCH};
12237
12238    use rudb_common::Stat;
12239    use rudb_common::Value;
12240    use rudb_common::bounds::{Frequencies, Op, Remainder, Zones};
12241    use rudb_common::stat::Provenance;
12242
12243    use super::*;
12244
12245    #[test]
12246    fn a_name_taken_in_pieces_is_the_name_of_the_pieces_joined() {
12247        let bytes: Vec<u8> =
12248            (0..300_u32).map(|at| (at.wrapping_mul(2_654_435_761) >> 13) as u8).collect();
12249        for length in [0, 1, 7, 31, 32, 33, 63, 64, 65, 100, 300] {
12250            let whole = content_name(&bytes[..length]);
12251            for step in [1, 3, 8, 31, 32, 33, 64, 301] {
12252                let mut namer = ContentNamer::default();
12253                bytes[..length].chunks(step).for_each(|piece| namer.update(piece));
12254                assert_eq!(namer.finish(), whole, "{length} bytes in pieces of {step}");
12255            }
12256        }
12257    }
12258
12259    /// The chooser as it was before it could rule kinds out up front: the same narrowing, with every
12260    /// kind tested for. What it writes is what the file used to hold.
12261    #[derive(Debug)]
12262    struct TestsEverything<'a>(&'a dyn chooser::Chooser);
12263
12264    impl chooser::Chooser for TestsEverything<'_> {
12265        fn name(&self) -> &'static str {
12266            "tests everything"
12267        }
12268
12269        fn narrow_strings(
12270            &self,
12271            values: &[&[u8]],
12272            offered: &[string::Kind],
12273            depth: u8,
12274        ) -> Vec<string::Kind> {
12275            self.0.narrow_strings(values, offered, depth)
12276        }
12277
12278        fn narrow_integers(
12279            &self,
12280            values: &[i64],
12281            offered: &[integer::Kind],
12282            depth: u8,
12283        ) -> Vec<integer::Kind> {
12284            self.0.narrow_integers(values, offered, depth)
12285        }
12286    }
12287
12288    #[test]
12289    fn ruling_kinds_out_before_testing_for_them_writes_the_same_bytes() {
12290        let columns: Vec<Vec<i64>> = vec![
12291            vec![],
12292            vec![5; 1000],
12293            (0..1000).collect(),
12294            (0..1000).map(|row| 1_600_000_000_000_000 + row * 1_000_000).collect(),
12295            (0..1000).map(|row| row / 50).collect(),
12296            (0..1000).map(|row| if row % 97 == 0 { row } else { 0 }).collect(),
12297            (0..1000).map(|row| (row * 7919) % 13).collect(),
12298            (0..1000).map(|row| (row * 2_654_435_761) % 1_000_003).collect(),
12299            (0..1000).map(|row| [3, 3, 3, 9, 9, 1][row as usize % 6]).collect(),
12300            (0..1000).map(|row| i64::MIN + row % 3).collect(),
12301        ];
12302        let choosers: [&dyn chooser::Chooser; 2] = [&Fixed, &Codes];
12303        for column in &columns {
12304            for chooser in choosers {
12305                let quick = integer::encode_with(column, chooser).unwrap();
12306                let full = integer::encode_with(column, &TestsEverything(chooser)).unwrap();
12307                assert_eq!(
12308                    quick,
12309                    full,
12310                    "{} on {:?}",
12311                    chooser.name(),
12312                    &column[..column.len().min(8)]
12313                );
12314            }
12315        }
12316    }
12317
12318    /// Parts of a column that all look alike come out of a settled shape byte for byte as they
12319    /// come out of a search, because the search would have kept the same tree on every one.
12320    #[test]
12321    fn parts_that_look_alike_replay_to_the_bytes_a_search_writes() {
12322        let mut settling = Settling::default();
12323        for part in 0..STRIPE_PARTS as i64 {
12324            let values: Vec<i64> = (0..2048)
12325                .map(|row| 1_600_000_000_000_000 + (part * 2048 + row) * 1_000_000 + row % 7)
12326                .collect();
12327            let searched = integer::encode_with(&values, &Fixed).unwrap();
12328            assert_eq!(settling.encode(&values).unwrap(), searched, "part {part}");
12329        }
12330    }
12331
12332    /// A column that changes shape partway through a stripe still reads back, and no part comes
12333    /// out much bigger than a search would have made it, because a replay that stops fitting or
12334    /// grows past a quarter a row is searched.
12335    #[test]
12336    fn a_column_that_changes_under_the_shape_is_searched_again() {
12337        let mut state = 0x9e37_79b9_7f4a_7c15_u64;
12338        let mut noise = move || {
12339            state ^= state << 13;
12340            state ^= state >> 7;
12341            state ^= state << 17;
12342            (state % 1_000_000) as i64
12343        };
12344        let mut settling = Settling::default();
12345        for part in 0..STRIPE_PARTS as i64 {
12346            let values: Vec<i64> = match part / 16 {
12347                0 => (0..2048).map(|row| (part * 2048 + row) / 300).collect(),
12348                1 => (0..2048).map(|_| noise()).collect(),
12349                2 => (0..2048).map(|row| if row % 97 == 0 { row } else { 42 }).collect(),
12350                _ => (0..2048).map(|row| 5 + (part * 2048 + row) * 1_000_000).collect(),
12351            };
12352            let settled = settling.encode(&values).unwrap();
12353            assert_eq!(integer::decode(&settled).unwrap(), values, "part {part}");
12354            let searched = integer::encode_with(&values, &Fixed).unwrap();
12355            assert!(
12356                settled.len() * 4 <= searched.len() * 5,
12357                "part {part}: {} settled against {} searched, {} against {}",
12358                settled.len(),
12359                searched.len(),
12360                integer::describe(&settled).unwrap(),
12361                integer::describe(&searched).unwrap(),
12362            );
12363        }
12364    }
12365
12366    #[test]
12367    fn checksum_matches_fixed_vectors() {
12368        assert_eq!(checksum(b""), 0xef46_db37_51d8_e999);
12369        assert_eq!(checksum(b"a"), 0xd24e_c4f1_a98c_6e5b);
12370        assert_eq!(checksum(b"abc"), 0x44bc_2cf5_ad77_0999);
12371    }
12372
12373    #[test]
12374    fn sorting_across_threads_matches_sorting_on_one() {
12375        let mut state = 0x9e37_79b9_7f4a_7c15_u64;
12376        let mut next = move || {
12377            state ^= state << 13;
12378            state ^= state >> 7;
12379            state ^= state << 17;
12380            state
12381        };
12382        let mut values = Vec::new();
12383        for at in 0..150_000_u64 {
12384            let value = match next() % 6 {
12385                0 => Vec::new(),
12386                1 => format!("https://example.com/{}", next() % 5_000).into_bytes(),
12387                2 => format!("https://example.com/path/{at}").into_bytes(),
12388                3 => b"same".to_vec(),
12389                4 => vec![0xff; (next() % 12) as usize],
12390                _ => (0..next() % 20).map(|_| (next() % 3) as u8).collect(),
12391            };
12392            values.push(value);
12393        }
12394        let value = |code: u32| values[code as usize].as_slice();
12395        for workers in [1, 2, 3, 8, 32] {
12396            let mut one = (0..values.len() as u32).rev().collect::<Vec<_>>();
12397            let mut across = one.clone();
12398            sort_by_value(&mut one, value);
12399            sort_by_value_across(&mut across, value, workers);
12400            assert_eq!(one, across, "{workers} workers");
12401        }
12402        let mut sorted = (0..values.len() as u32).collect::<Vec<_>>();
12403        sort_by_value_across(&mut sorted, value, 8);
12404        assert!(sorted.windows(2).all(|pair| value(pair[0]) <= value(pair[1])));
12405    }
12406
12407    fn path(label: &str) -> PathBuf {
12408        let stamp = SystemTime::now().duration_since(UNIX_EPOCH).expect("time advances").as_nanos();
12409        std::env::temp_dir().join(format!("rudb-native-{label}-{}-{stamp}.rdb", std::process::id()))
12410    }
12411
12412    /// Every value of a dictionary in code order, which the tests have no other way to ask for now
12413    /// that a dictionary does not keep the bytes of the values it has seen.
12414    ///
12415    /// Only valid once `finish_blocks` has run, because until then the last part block is still raw.
12416    fn dictionary_values(dictionary: &GlobalDictionary) -> Vec<Vec<u8>> {
12417        let (flat, bases) = dictionary.decoded(None).expect("the blocks decode");
12418        (0..dictionary.values())
12419            .map(|code| {
12420                let (from, to) = GlobalDictionary::value_span(&dictionary.ends, &bases, code);
12421                flat[from..to].to_vec()
12422            })
12423            .collect()
12424    }
12425
12426    /// The sections a test put in the table, which is every one the writer did not.
12427    ///
12428    /// A table now carries a summary and a sketch per column out of the write itself, and a test
12429    /// about the section table is not about those. Filtering by kind rather than by count, so a
12430    /// table that turns out to have no room for its summaries does not quietly change what these
12431    /// tests are asserting over.
12432    fn attached(table: &Table) -> Vec<&Section> {
12433        table.sections().iter().filter(|held| !held.among(section::STATISTICS_KINDS)).collect()
12434    }
12435
12436    /// A read names the offset it wants, so a cursor somebody else moved cannot reach it.
12437    #[test]
12438    fn a_read_at_an_offset_ignores_where_another_thread_left_the_cursor() {
12439        const SPANS: usize = 64;
12440        const SPAN: usize = 512;
12441        let path = path("positional");
12442        let content: Vec<u8> =
12443            (0..SPANS).flat_map(|span| std::iter::repeat_n(span as u8, SPAN)).collect();
12444        fs::write(&path, &content).expect("the file is written");
12445        let file = Arc::new(File::open(&path).expect("the file opens"));
12446        std::thread::scope(|scope| {
12447            for _ in 0..8 {
12448                let file = Arc::clone(&file);
12449                scope.spawn(move || {
12450                    for _ in 0..64 {
12451                        for span in 0..SPANS {
12452                            let mut bytes = [0_u8; SPAN];
12453                            read_at(&file, (span * SPAN) as u64, &mut bytes)
12454                                .expect("the span reads");
12455                            assert!(
12456                                bytes.iter().all(|byte| *byte == span as u8),
12457                                "span {span} came back as {}",
12458                                bytes[0],
12459                            );
12460                        }
12461                    }
12462                });
12463            }
12464        });
12465        let mut past = [0_u8; SPAN];
12466        let end = (SPANS * SPAN) as u64;
12467        let error = read_at(&file, end, &mut past).expect_err("a read past the end is refused");
12468        assert!(error.message().contains("ends before its declared length"), "{error}");
12469        drop(file);
12470        let _ = fs::remove_file(&path);
12471    }
12472
12473    /// The writer records where it put a page and puts it there.
12474    ///
12475    /// This used to move the file's cursor between the steps that record an offset, which is what
12476    /// reading the pages back to build the frequencies did on a platform with no `pread`, and the
12477    /// directory landed on top of a page. The writer's file is an `rudb_io` file now and has no
12478    /// cursor to move, so what is left is the check that every page is where the directory says.
12479    #[test]
12480    fn a_writer_puts_a_page_where_it_said_it_did_wherever_the_cursor_has_got_to() {
12481        let path = path("cursor");
12482        let mut writer = Writer::create(
12483            &path,
12484            "items",
12485            vec![
12486                Field::required("id", LogicalType::Integer),
12487                Field::new("text", LogicalType::Varchar),
12488            ],
12489        )
12490        .expect("new file");
12491        writer.append(&sample()).expect("first part");
12492        writer.append(&sample()).expect("second part");
12493        writer.finish().expect("commit");
12494        let reader = Reader::open(&path).expect("reopen from disk");
12495        assert_eq!(reader.table().rows(), 6);
12496        let ids = reader.read(0, &[0]).expect("the integer page reads back");
12497        assert_eq!(ids.value_at(0, 0), Value::Integer(4));
12498        assert_eq!(ids.value_at(2, 0), Value::Integer(-2));
12499        let text = reader.read(1, &[1]).expect("the text page reads back");
12500        assert_eq!(text.value_at(1, 0), Value::Null);
12501        assert_eq!(text.value_at(2, 0), Value::Varchar("long text after a slash".into()));
12502        // Nothing the directory points at may run past the end of the file, which is the shape the
12503        // failure took: a page recorded at an offset the directory had already been written over.
12504        let end = reader.table().stripes().iter().flat_map(|stripe| {
12505            stripe
12506                .pages
12507                .iter()
12508                .map(|page| page.offset + u64::from(page.length))
12509                .chain(std::iter::once(stripe.index.offset + u64::from(stripe.index.length)))
12510        });
12511        let last = end.fold(HEADER, u64::max);
12512        let directory = fs::metadata(&path).expect("the file is there").len();
12513        assert!(last <= directory, "a page runs to {last} in a file of {directory} bytes");
12514        fs::remove_file(path).expect("remove scratch file");
12515    }
12516
12517    /// How long a global dictionary index is, read out of the page's own header.
12518    ///
12519    /// The tests below damage a byte of the order or of the payload, so they need to know where each
12520    /// one starts, and working it out here rather than writing a number down means adding something
12521    /// to the index does not quietly turn one of them into a test that damages the index instead.
12522    fn dictionary_index_len(header: &[u8; DICTIONARY_HEADER]) -> u64 {
12523        let count = u64::from(u32::from_le_bytes(header[0..4].try_into().expect("four bytes")));
12524        let blocks = u64::from(u32::from_le_bytes(header[8..12].try_into().expect("four bytes")));
12525        let width = u32::from_le_bytes(header[12..16].try_into().expect("four bytes"));
12526        let bits = (width & !DICTIONARY_FLAGS) as usize;
12527        let payload_words = if width & DICTIONARY_SCATTERED == 0 { 2 } else { 3 };
12528        let rank_blocks = count.div_ceil(TEXT_RANK_BLOCK as u64);
12529        DICTIONARY_HEADER as u64
12530            + offset_bytes(count as usize, bits) as u64
12531            + blocks * payload_words * 8
12532            + rank_blocks * 16
12533            + if width & DICTIONARY_GRAMS == 0 { 0 } else { 8 }
12534    }
12535
12536    fn sample() -> Chunk {
12537        Chunk::new(vec![
12538            Vector::from_values(
12539                LogicalType::Integer,
12540                &[Value::Integer(4), Value::Integer(9), Value::Integer(-2)],
12541            )
12542            .expect("integers"),
12543            Vector::from_values(
12544                LogicalType::Varchar,
12545                &[
12546                    Value::Varchar("alpha".into()),
12547                    Value::Null,
12548                    Value::Varchar("long text after a slash".into()),
12549                ],
12550            )
12551            .expect("strings"),
12552        ])
12553        .expect("matching rows")
12554    }
12555
12556    fn sample_ids() -> Chunk {
12557        Chunk::new(vec![
12558            Vector::flat(LogicalType::Integer, Data::Int32(vec![7, 8, 9].into()))
12559                .expect("integers"),
12560        ])
12561        .expect("one column")
12562    }
12563
12564    #[test]
12565    fn the_planner_gets_the_null_count_off_the_same_directory_the_bounds_are_in() {
12566        // Six rows, two of them null. `IS NULL` used to get the same fifth any unreadable
12567        // condition gets, and the number was in the stripe entry next to the bounds all along.
12568        let path = path("nulls_for_the_planner");
12569        let mut writer =
12570            Writer::create(&path, "items", vec![Field::new("a", LogicalType::Integer)])
12571                .expect("new file");
12572        let rows = Chunk::new(vec![
12573            Vector::from_values(
12574                LogicalType::Integer,
12575                &[
12576                    Value::Integer(4),
12577                    Value::Null,
12578                    Value::Integer(9),
12579                    Value::Null,
12580                    Value::Integer(1),
12581                    Value::Integer(2),
12582                ],
12583            )
12584            .expect("integers"),
12585        ])
12586        .expect("one column");
12587        writer.append(&rows).expect("the only part");
12588        writer.finish().expect("commit");
12589        let reader = Reader::open(&path).expect("reopen from disk");
12590        let stripes = Stripes::new(reader);
12591        let column = stripes.column("a").expect("the file has that column");
12592        assert_eq!(stripes.nulls(column), Stat::exact(2, Provenance::NullCount));
12593        // A column the file does not have. Zero here would be a fact about a column that is not
12594        // there, which the planner would then divide by.
12595        assert_eq!(stripes.nulls(column + 1), Stat::Unknown);
12596        fs::remove_file(&path).expect("clean up");
12597    }
12598
12599    #[test]
12600    fn the_planner_gets_a_leading_count_without_a_complete_numeric_synopsis() {
12601        // Six rows hold three values. The two leading counts help equality planning, while the
12602        // omitted value keeps the directory from being a complete grouped-count result.
12603        let path = path("frequencies_for_the_planner");
12604        let mut writer =
12605            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
12606                .expect("new file");
12607        let rows = Chunk::new(vec![
12608            Vector::from_values(
12609                LogicalType::Integer,
12610                &[
12611                    Value::Integer(4),
12612                    Value::Integer(4),
12613                    Value::Integer(4),
12614                    Value::Integer(9),
12615                    Value::Integer(9),
12616                    Value::Integer(1),
12617                ],
12618            )
12619            .expect("integers"),
12620        ])
12621        .expect("one column");
12622        writer.append(&rows).expect("the only part");
12623        writer.finish().expect("commit");
12624        let reader = Reader::open(&path).expect("reopen from disk");
12625        let common = Common::new(reader);
12626        assert_eq!(common.rows(), 6);
12627        let column = common.column("id").expect("the file has that column");
12628        assert_eq!(common.column("nothing"), None);
12629        assert_eq!(
12630            common.rows_with(column, &Bound::Int(4)),
12631            Stat::exact(3, Provenance::FrequencySynopsis)
12632        );
12633        // An absent value cannot be distinguished from the omitted one by the synopsis.
12634        assert_eq!(common.rows_with(column, &Bound::Int(7)), Stat::Unknown);
12635        // A constant of another domain against an integer column. Nothing in the list compares
12636        // with it, so the zero above would be an artefact of the mismatch rather than a fact.
12637        assert_eq!(common.rows_with(column, &Bound::Bytes(b"four".to_vec())), Stat::Unknown);
12638        assert!(common.remainder(column).is_some());
12639        fs::remove_file(&path).expect("clean up");
12640    }
12641
12642    #[test]
12643    fn string_frequency_estimates_do_not_open_the_global_dictionary() {
12644        let path = path("string_frequencies_for_the_planner");
12645        let mut writer =
12646            Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
12647                .expect("new file");
12648        let rows = Chunk::new(vec![
12649            Vector::from_values(
12650                LogicalType::Varchar,
12651                &[
12652                    Value::Varchar(String::new()),
12653                    Value::Varchar("alpha".into()),
12654                    Value::Varchar(String::new()),
12655                    Value::Varchar("beta".into()),
12656                    Value::Varchar(String::new()),
12657                ],
12658            )
12659            .expect("strings"),
12660        ])
12661        .expect("one column");
12662        writer.append(&rows).expect("the only part");
12663        writer.finish().expect("commit");
12664
12665        let reader = Reader::open(&path).expect("reopen from disk");
12666        assert_eq!(reader.reads().dictionaries, 0, "open reads only the directory");
12667        let common = Common::new(reader.clone());
12668        let column = common.column("text").expect("the file has that column");
12669        assert_eq!(
12670            common.rows_with(column, &Bound::Bytes(Vec::new())),
12671            Stat::exact(3, Provenance::FrequencySynopsis)
12672        );
12673        assert_eq!(
12674            common.rows_with(column, &Bound::Bytes(b"missing".to_vec())),
12675            Stat::exact(0, Provenance::FrequencySynopsis)
12676        );
12677        assert_eq!(
12678            reader.reads().dictionaries,
12679            0,
12680            "the bounded spellings answer without opening the dictionary index"
12681        );
12682        fs::remove_file(&path).expect("clean up");
12683    }
12684
12685    #[test]
12686    fn host_groups_certify_omitted_hosts_and_keep_exact_aggregates() {
12687        let path = path("certified_host_groups");
12688        let mut writer =
12689            Writer::create(&path, "hits", vec![Field::required("Referer", LogicalType::Varchar)])
12690                .expect("new file");
12691        let mut values = vec![Value::Varchar("http://www.example.com/a".into()); 150];
12692        values.extend(vec![Value::Varchar("https://example.com/b".into()); 70]);
12693        values.extend((0..550).map(|at| Value::Varchar(format!("https://site{at}.test/x"))));
12694        values.push(Value::Varchar(String::new()));
12695        for part in values.chunks(512) {
12696            writer
12697                .append(
12698                    &Chunk::new(vec![
12699                        Vector::from_values(LogicalType::Varchar, part).expect("strings"),
12700                    ])
12701                    .expect("one column"),
12702                )
12703                .expect("part written");
12704        }
12705        writer.finish().expect("commit");
12706        let reader = Reader::open(&path).expect("reopen");
12707        assert!(reader.table.host_groups.is_none(), "no query-specific host result is stored");
12708        fs::remove_file(&path).expect("clean up");
12709    }
12710
12711    /// A table directory with nothing in it but a name and one column, for the section tests.
12712    ///
12713    /// The section table is orthogonal to everything else in a directory, so the tests that pin it
12714    /// say so by starting from the emptiest table that encodes.
12715    fn bare_table(sections: Vec<Section>) -> Table {
12716        Table {
12717            name: "linked".to_owned(),
12718            fields: vec![Field::required("id", LogicalType::Integer)],
12719            stripes: Vec::new(),
12720            rows: 0,
12721            dictionaries: vec![None],
12722            dictionary_payloads: Vec::new(),
12723            demoted: Vec::new(),
12724            distincts: vec![None],
12725            frequencies: vec![None],
12726            pair_frequencies: Vec::new(),
12727            frequency_texts: Vec::new(),
12728            host_groups: None,
12729            clustering: None,
12730            generation: 1,
12731            sections,
12732        }
12733    }
12734
12735    fn a_key_map_section() -> Section {
12736        Section {
12737            kind: *section::KEY_MAP,
12738            id: 1,
12739            generation: 3,
12740            extents: 1,
12741            extent_page: HEADER,
12742            extent_bytes: section::EXTENT_BYTES as u32,
12743            hash: 0x1234_5678_9abc_def0,
12744            flags: 0,
12745            header_bytes: 24,
12746        }
12747    }
12748
12749    #[test]
12750    fn a_section_table_round_trips_through_a_directory() {
12751        let mut later = a_key_map_section();
12752        later.kind = *b"RUDBZZ9\0";
12753        later.id = 2;
12754        let table = bare_table(vec![a_key_map_section(), later]);
12755        let directory = encode_directory(&table).expect("directory");
12756        let decoded = decode_directory(&directory, 1 << 20).expect("reopen");
12757        assert_eq!(decoded.sections(), &[a_key_map_section(), later]);
12758        // The second is a kind this build has no name for, and it survived the round trip anyway.
12759        // That is what keeps an old build from silently discarding a newer build's work when it
12760        // rewrites a directory.
12761        assert!(decoded.sections()[0].known());
12762        assert!(!decoded.sections()[1].known());
12763    }
12764
12765    #[test]
12766    fn a_directory_written_before_the_section_table_reads_as_a_table_with_none() {
12767        // The G1 exit criterion, at the directory level. A format 22 directory is exactly this
12768        // build's directory with the trailing section block cut off, so cutting it off is the
12769        // honest way to make one: no fixture to go stale, and no separate encoder to drift.
12770        let directory = encode_directory(&bare_table(Vec::new())).expect("directory");
12771        let block = SECTIONS.len() + size_of::<u64>() + size_of::<u16>();
12772        let older = &directory[..directory.len() - block];
12773        let decoded = decode_directory(older, 1 << 20).expect("a directory from before sections");
12774        assert!(decoded.sections().is_empty());
12775        assert_eq!(decoded.generation(), 0, "a format 22 table recorded no generation");
12776        assert_eq!(decoded.name(), "linked");
12777        assert_eq!(decoded.fields().len(), 1, "everything before the block still decodes");
12778    }
12779
12780    #[test]
12781    fn a_file_stamped_with_the_previous_format_still_opens_and_reads() {
12782        // The same criterion end to end, which is the one the milestone actually asks for: a build
12783        // that knows about sections opens a file written by a build that did not, with no rewrite
12784        // and no repair, and answers from it. The version field is patched rather than a file
12785        // committed by an old binary because the bytes either side of it are identical: format 22
12786        // and format 23 differ only in a trailing directory block, and a reader that stops before
12787        // that block gets a table with no sections.
12788        let path = path("format_twenty_two");
12789        let mut writer =
12790            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
12791                .expect("new file");
12792        let rows = Chunk::new(vec![
12793            Vector::from_values(
12794                LogicalType::Integer,
12795                &[Value::Integer(1), Value::Integer(2), Value::Integer(3)],
12796            )
12797            .expect("integers"),
12798        ])
12799        .expect("one column");
12800        writer.append(&rows).expect("the only part");
12801        writer.finish().expect("commit");
12802
12803        let file = OpenOptions::new().write(true).open(&path).expect("reopen to patch");
12804        write_at(&file, 8, &22_u32.to_le_bytes()).expect("stamp the older format");
12805        drop(file);
12806
12807        let reader = Reader::open(&path).expect("a format 22 file opens unchanged");
12808        assert_eq!(reader.table().rows(), 3);
12809        // The rows and not the section table, because the section block is found by the magic at
12810        // the end of the directory rather than by the number in the header, so stamping the header
12811        // back does not take away the summaries this writer put there. What the test is about is
12812        // that the version check accepts 22, and the rows coming back is what says it did.
12813        assert_eq!(reader.read(0, &[0]).expect("the part still reads").len(), 3);
12814
12815        // And a format this build has never written is still refused, so the accept set is a list
12816        // and not an absence of a check.
12817        let file = OpenOptions::new().write(true).open(&path).expect("reopen to patch");
12818        write_at(&file, 8, &21_u32.to_le_bytes()).expect("stamp an unreadable format");
12819        drop(file);
12820        let error = Reader::open(&path).expect_err("format 21 is not readable");
12821        assert!(error.to_string().contains("format 21"), "{error}");
12822
12823        fs::remove_file(&path).expect("clean up");
12824    }
12825
12826    #[test]
12827    fn a_section_whose_extent_table_is_outside_the_file_is_refused() {
12828        // The bound the format has to check and `section` cannot, because only the reader knows how
12829        // big the file is. Reading the payload a section like this names would be reading whatever
12830        // else happens to be at that offset, which is the one way a graph section could turn into a
12831        // wrong answer rather than a slow one.
12832        let mut past = a_key_map_section();
12833        past.extent_page = 1 << 30;
12834        let directory = encode_directory(&bare_table(vec![past])).expect("directory");
12835        let error = decode_directory(&directory, 1 << 20).expect_err("refused");
12836        assert!(error.to_string().contains("outside the file"), "{error}");
12837
12838        let mut inside_the_header = a_key_map_section();
12839        inside_the_header.extent_page = 8;
12840        let directory = encode_directory(&bare_table(vec![inside_the_header])).expect("directory");
12841        assert!(
12842            decode_directory(&directory, 1 << 20).is_err(),
12843            "a section may not overlap a header"
12844        );
12845    }
12846
12847    #[test]
12848    fn a_section_recorded_as_not_built_is_legal_and_names_no_bytes() {
12849        // Section 3.7: a relationship that does not fit the budget is recorded with its size so
12850        // that `rudb_links()` can report what a larger budget would buy. That record is a section
12851        // entry with no extents, so it has to survive a round trip while naming nothing.
12852        let not_built = Section {
12853            kind: *section::FORWARD_LINK,
12854            id: 9,
12855            generation: 3,
12856            extents: 0,
12857            extent_page: 0,
12858            extent_bytes: 0,
12859            hash: 0,
12860            flags: 0,
12861            header_bytes: 0,
12862        };
12863        let directory = encode_directory(&bare_table(vec![not_built])).expect("directory");
12864        let decoded = decode_directory(&directory, 1 << 20).expect("reopen");
12865        assert_eq!(decoded.sections(), &[not_built]);
12866
12867        // But a section with no extents that still names an extent table is incoherent, and an
12868        // incoherent entry is a torn directory rather than a relationship that was skipped.
12869        let mut incoherent = not_built;
12870        incoherent.extent_bytes = 28;
12871        incoherent.extent_page = HEADER;
12872        let directory = encode_directory(&bare_table(vec![incoherent])).expect("directory");
12873        assert!(decode_directory(&directory, 1 << 20).is_err());
12874    }
12875
12876    #[test]
12877    fn a_directory_naming_more_sections_than_the_bound_is_refused() {
12878        let directory = encode_directory(&bare_table(Vec::new())).expect("directory");
12879        let mut torn = directory.clone();
12880        let count_at = torn.len() - size_of::<u16>();
12881        torn[count_at..].copy_from_slice(&u16::MAX.to_le_bytes());
12882        // Not an allocation of sixty five thousand entries off a torn count: either the bound
12883        // refuses it or the bytes run out, and both are errors rather than a read past the end.
12884        assert!(decode_directory(&torn, 1 << 20).is_err());
12885    }
12886
12887    /// A committed one column file of `rows` integers, for the attach tests.
12888    fn linked_file(label: &str, rows: i32) -> PathBuf {
12889        let path = path(label);
12890        let mut writer =
12891            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
12892                .expect("new file");
12893        let values = (0..rows).map(Value::Integer).collect::<Vec<_>>();
12894        let chunk =
12895            Chunk::new(vec![Vector::from_values(LogicalType::Integer, &values).expect("integers")])
12896                .expect("one column");
12897        writer.append(&chunk).expect("the only part");
12898        writer.finish().expect("commit");
12899        path
12900    }
12901
12902    fn a_key_map_payload() -> Vec<u8> {
12903        // Shaped like one without being one: this crate never reads a payload, so what matters here
12904        // is that every byte comes back and that the header the entry measures is at the front.
12905        (0..512_u32).flat_map(u32::to_le_bytes).collect()
12906    }
12907
12908    #[test]
12909    fn a_section_attached_to_a_committed_file_reads_back_byte_for_byte() {
12910        let path = linked_file("attach", 64);
12911        let payload = a_key_map_payload();
12912        let table = attach(
12913            &path,
12914            "items",
12915            &[section::Attachment {
12916                kind: *section::KEY_MAP,
12917                id: 0,
12918                flags: 2,
12919                header_bytes: 40,
12920                bytes: &payload,
12921            }],
12922        )
12923        .expect("attach a key map");
12924        assert_eq!(attached(&table).len(), 1);
12925
12926        let reader = Reader::open(&path).expect("reopen after the attach");
12927        let held = attached(reader.table());
12928        assert_eq!(held.len(), 1);
12929        assert_eq!(held[0].kind, *section::KEY_MAP);
12930        assert_eq!(held[0].flags, 2, "the form a reader must not have to guess");
12931        assert_eq!(held[0].header_bytes, 40);
12932        // The generation is the one the pages were written at, not the one the attach committed at.
12933        // Attaching a section moved no row, so a section written by it is current, and a second
12934        // table added to this file later would not make it stale.
12935        assert_eq!(held[0].generation, 1);
12936        assert!(held[0].usable(reader.table().generation()));
12937        assert_eq!(reader.payload(held[0]).expect("read the payload"), payload);
12938        assert_eq!(reader.extents(held[0]).expect("extent table").len(), 1);
12939
12940        fs::remove_file(&path).expect("clean up");
12941    }
12942
12943    #[test]
12944    fn attaching_a_section_answers_every_row_exactly_as_before() {
12945        // Section 3.1 end to end, and the reason the whole layer is safe to build incrementally. A
12946        // file with a section in it and the same file without one have to agree row for row, so the
12947        // comparison is made against the answers taken before the attach rather than against a
12948        // constant somebody typed.
12949        let path = linked_file("attach_changes_nothing", 300);
12950        let before = Reader::open(&path).expect("open before");
12951        let rows = before.table().rows();
12952        let first = before.read(0, &[0]).expect("read before");
12953        let values = (0..rows).map(|at| first.value_at(at, 0)).collect::<Vec<_>>();
12954        let layout = before.layout().columns_total();
12955        drop(before);
12956
12957        let payload = a_key_map_payload();
12958        attach(
12959            &path,
12960            "items",
12961            &[section::Attachment {
12962                kind: *section::KEY_MAP,
12963                id: 0,
12964                flags: 0,
12965                header_bytes: 0,
12966                bytes: &payload,
12967            }],
12968        )
12969        .expect("attach");
12970
12971        let after = Reader::open(&path).expect("open after");
12972        assert_eq!(after.table().rows(), rows);
12973        let read = after.read(0, &[0]).expect("read after");
12974        for (at, value) in values.iter().enumerate() {
12975            assert_eq!(&read.value_at(at, 0), value, "row {at} moved");
12976        }
12977        assert_eq!(
12978            after.layout().columns_total(),
12979            layout,
12980            "an attach appends and does not rewrite a column page"
12981        );
12982
12983        fs::remove_file(&path).expect("clean up");
12984    }
12985
12986    #[test]
12987    fn a_rebuilt_section_replaces_the_one_it_supersedes() {
12988        // Rebuilding a key map has to be a write and not a question. If an attach added rather than
12989        // replaced, a table rebuilt a few times would name several maps for one column and a reader
12990        // would have to pick, which is a decision with no right answer in it.
12991        let path = linked_file("attach_twice", 32);
12992        let one = a_key_map_payload();
12993        let two = vec![7_u8; 1024];
12994        let entry = |bytes| section::Attachment {
12995            kind: *section::KEY_MAP,
12996            id: 4,
12997            flags: 1,
12998            header_bytes: 0,
12999            bytes,
13000        };
13001        attach(&path, "items", &[entry(&one)]).expect("first build");
13002        attach(&path, "items", &[entry(&two)]).expect("rebuild");
13003
13004        let reader = Reader::open(&path).expect("reopen");
13005        let held = attached(reader.table());
13006        assert_eq!(held.len(), 1, "one map per column and not one per build");
13007        assert_eq!(reader.payload(held[0]).expect("payload"), two);
13008
13009        fs::remove_file(&path).expect("clean up");
13010    }
13011
13012    #[test]
13013    fn an_attach_carries_through_a_kind_it_does_not_know() {
13014        // The first of section 3.2's three rules, at the point where it is easiest to break: a build
13015        // that rewrites a directory has to carry an entry it has no name for, or opening a file with
13016        // an older binary and attaching one section quietly deletes the work of a newer one.
13017        let path = linked_file("attach_unknown", 16);
13018        let payload = vec![3_u8; 96];
13019        attach(
13020            &path,
13021            "items",
13022            &[section::Attachment {
13023                kind: *b"RUDBZZ9\0",
13024                id: 1,
13025                flags: 0,
13026                header_bytes: 0,
13027                bytes: &payload,
13028            }],
13029        )
13030        .expect("a kind this build does not know still writes");
13031        let key_map = a_key_map_payload();
13032        attach(
13033            &path,
13034            "items",
13035            &[section::Attachment {
13036                kind: *section::KEY_MAP,
13037                id: 0,
13038                flags: 0,
13039                header_bytes: 0,
13040                bytes: &key_map,
13041            }],
13042        )
13043        .expect("attach beside it");
13044
13045        let reader = Reader::open(&path).expect("reopen");
13046        let held = attached(reader.table());
13047        assert_eq!(held.len(), 2, "the unfamiliar entry survived a directory rewrite");
13048        let unknown = held.iter().find(|one| !one.known()).expect("the unfamiliar one");
13049        assert_eq!(reader.payload(unknown).expect("its bytes are still there"), payload);
13050
13051        fs::remove_file(&path).expect("clean up");
13052    }
13053
13054    #[test]
13055    fn a_payload_of_nothing_is_a_relationship_recorded_as_not_built() {
13056        let path = linked_file("attach_not_built", 8);
13057        attach(
13058            &path,
13059            "items",
13060            &[section::Attachment {
13061                kind: *section::FORWARD_LINK,
13062                id: 2,
13063                flags: 0,
13064                header_bytes: 0,
13065                bytes: &[],
13066            }],
13067        )
13068        .expect("record a link that did not fit the budget");
13069
13070        let reader = Reader::open(&path).expect("reopen");
13071        let held = attached(reader.table());
13072        assert_eq!(held.len(), 1);
13073        assert_eq!(held[0].extents, 0);
13074        assert_eq!(held[0].extent_page, 0, "an entry that names no bytes points at none");
13075        assert!(reader.extents(held[0]).expect("no extent table").is_empty());
13076        assert!(reader.payload(held[0]).expect("no payload").is_empty());
13077
13078        fs::remove_file(&path).expect("clean up");
13079    }
13080
13081    #[test]
13082    fn a_payload_past_one_extent_is_split_and_joined_back() {
13083        // Issue #745's rule, exercised rather than argued. One byte past the bound is the smallest
13084        // payload that has to be two extents, and it is the case a split written for the common
13085        // size gets wrong.
13086        let path = linked_file("attach_two_extents", 8);
13087        let payload = vec![0x5a_u8; section::MAX_EXTENT as usize + 1];
13088        attach(
13089            &path,
13090            "items",
13091            &[section::Attachment {
13092                kind: *section::KEY_MAP,
13093                id: 0,
13094                flags: 0,
13095                header_bytes: 0,
13096                bytes: &payload,
13097            }],
13098        )
13099        .expect("attach a payload past the bound");
13100
13101        let reader = Reader::open(&path).expect("reopen");
13102        let held = attached(reader.table());
13103        let extents = reader.extents(held[0]).expect("extent table");
13104        assert_eq!(extents.len(), 2, "one byte past the bound is two extents");
13105        assert_eq!(extents[0].length, section::MAX_EXTENT);
13106        assert_eq!(extents[1].length, 1);
13107        assert_eq!(extents[1].first, u64::from(section::MAX_EXTENT));
13108        // And the extent the caller wants is readable on its own, which is the point of the split.
13109        assert_eq!(reader.extent(&extents[1]).expect("the last extent"), vec![0x5a]);
13110        assert_eq!(reader.payload(held[0]).expect("the whole payload").len(), payload.len());
13111
13112        fs::remove_file(&path).expect("clean up");
13113    }
13114
13115    #[test]
13116    fn a_torn_extent_is_refused_rather_than_decoded() {
13117        let path = linked_file("attach_torn", 8);
13118        let payload = a_key_map_payload();
13119        attach(
13120            &path,
13121            "items",
13122            &[section::Attachment {
13123                kind: *section::KEY_MAP,
13124                id: 0,
13125                flags: 0,
13126                header_bytes: 0,
13127                bytes: &payload,
13128            }],
13129        )
13130        .expect("attach");
13131
13132        let reader = Reader::open(&path).expect("reopen");
13133        let extent = reader.extents(&reader.table().sections()[0]).expect("extent table")[0];
13134        let file = OpenOptions::new().write(true).open(&path).expect("reopen to corrupt");
13135        write_at(&file, extent.offset + 7, &[0xff]).expect("flip a byte of the payload");
13136        drop(file);
13137
13138        let reader = Reader::open(&path).expect("the table still opens");
13139        let error = reader
13140            .payload(&reader.table().sections()[0])
13141            .expect_err("a corrupt payload is not handed out");
13142        assert!(error.to_string().contains("checksum"), "{error}");
13143        // And the table is still readable, which is section 3.1: a section that cannot be trusted
13144        // costs the query its shortcut and nothing else.
13145        assert_eq!(reader.read(0, &[0]).expect("the column is untouched").width(), 1);
13146
13147        fs::remove_file(&path).expect("clean up");
13148    }
13149
13150    #[test]
13151    fn attaching_to_a_file_of_the_previous_format_is_refused_rather_than_done() {
13152        // Readable is not writable. A format 22 directory has no section block, and adding one
13153        // without moving the number in the header would leave a file claiming a format it is not.
13154        let path = linked_file("attach_old_format", 8);
13155        let file = OpenOptions::new().write(true).open(&path).expect("reopen to patch");
13156        write_at(&file, 8, &22_u32.to_le_bytes()).expect("stamp the older format");
13157        drop(file);
13158
13159        let payload = a_key_map_payload();
13160        let error = attach(
13161            &path,
13162            "items",
13163            &[section::Attachment {
13164                kind: *section::KEY_MAP,
13165                id: 0,
13166                flags: 0,
13167                header_bytes: 0,
13168                bytes: &payload,
13169            }],
13170        )
13171        .expect_err("format 22 cannot gain a section");
13172        assert!(error.to_string().contains("format 22"), "{error}");
13173        assert!(Reader::open(&path).expect("and the file is untouched").table().rows() == 8);
13174
13175        fs::remove_file(&path).expect("clean up");
13176    }
13177
13178    #[test]
13179    fn a_section_header_longer_than_its_payload_is_refused_at_the_write() {
13180        let path = linked_file("attach_bad_header", 8);
13181        let error = attach(
13182            &path,
13183            "items",
13184            &[section::Attachment {
13185                kind: *section::KEY_MAP,
13186                id: 0,
13187                flags: 0,
13188                header_bytes: 40,
13189                bytes: &[1, 2, 3],
13190            }],
13191        )
13192        .expect_err("a writer's bug stops at the write");
13193        assert!(error.to_string().contains("header is longer"), "{error}");
13194
13195        fs::remove_file(&path).expect("clean up");
13196    }
13197
13198    #[test]
13199    fn attaching_to_a_name_the_file_does_not_hold_says_so() {
13200        let path = linked_file("attach_wrong_name", 8);
13201        let error = attach(&path, "orders", &[]).expect_err("no such table");
13202        assert!(error.to_string().contains("orders"), "{error}");
13203        fs::remove_file(&path).expect("clean up");
13204    }
13205
13206    #[test]
13207    fn the_planner_gets_an_exact_count_for_a_leading_value_of_an_incomplete_synopsis() {
13208        // The case a complete synopsis does not cover, and the one worth the most. 16,000 rows over
13209        // 601 distinct values, 10,000 of them holding a single value and the rest spread ten apiece
13210        // over six hundred more. The writer holds 512 values, so the list is a prefix and most of
13211        // the tail is outside it. The counts inside it are still exact, because the pass recounts
13212        // the candidates that survived it, so `id = 1` is ten thousand rows rather than the
13213        // twenty six a distinct count of 601 would divide its way to.
13214        let path = path("frequency_prefix_for_the_planner");
13215        let mut writer =
13216            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
13217                .expect("new file");
13218        let mut values = vec![Value::Integer(1); 10_000];
13219        for _ in 0..10 {
13220            values.extend((0..600).map(|tail| Value::Integer(1_000 + tail)));
13221        }
13222        // A vector holds 8,192 rows, so this goes in as several parts. The pass that takes the
13223        // synopsis walks the whole column rather than a part, so the counts are the same either way.
13224        for part in values.chunks(8_000) {
13225            let rows = Chunk::new(vec![
13226                Vector::from_values(LogicalType::Integer, part).expect("integers"),
13227            ])
13228            .expect("one column");
13229            writer.append(&rows).expect("a part");
13230        }
13231        writer.finish().expect("commit");
13232        let reader = Reader::open(&path).expect("reopen from disk");
13233        let prefix =
13234            reader.frequency_prefix(0).expect("a readable synopsis").expect("the column has one");
13235        // A prefix and not the whole column, and the writer said how many rows anything left out of
13236        // it can hold.
13237        assert_eq!(prefix.entries.len(), 512);
13238        assert_eq!(prefix.omitted_max, 10);
13239        let common = Common::new(reader);
13240        assert_eq!(common.rows(), 16_000);
13241        let column = common.column("id").expect("the file has that column");
13242        assert_eq!(
13243            common.rows_with(column, &Bound::Int(1)),
13244            Stat::exact(10_000, Provenance::FrequencySynopsis)
13245        );
13246        // In the prefix, because ties go to the smaller value and the prefix reaches 1,510.
13247        assert_eq!(
13248            common.rows_with(column, &Bound::Int(1_100)),
13249            Stat::exact(10, Provenance::FrequencySynopsis)
13250        );
13251        // Outside it, and a prefix says nothing about a value it does not list. Not zero, which is
13252        // what a complete list would say, and the file holds ten rows of this one.
13253        assert_eq!(common.rows_with(column, &Bound::Int(1_550)), Stat::Unknown);
13254        // Not in the file at all, and still nothing rather than a zero. A prefix cannot tell the
13255        // two apart, which is the whole of what it gives up.
13256        assert_eq!(common.rows_with(column, &Bound::Int(9_999)), Stat::Unknown);
13257        // What the prefix left out, which is what turns the unknown above into a number. The 512
13258        // entries account for 15,110 rows, so 890 are left for the 89 values the writer dropped,
13259        // and 890 over 89 is the ten rows each of them really holds.
13260        let remainder = common.remainder(column).expect("the list is a prefix");
13261        assert_eq!(remainder, Remainder { rows: 890, listed: 512, most: 10 });
13262        assert_eq!(remainder.rows / (601 - remainder.listed), 10);
13263        fs::remove_file(&path).expect("clean up");
13264    }
13265
13266    /// A file with no table in it is a file, and opening it says so rather than failing.
13267    #[test]
13268    fn a_file_holding_no_table_commits_and_opens_and_a_table_can_be_added_to_it() {
13269        let path = path("empty");
13270        Writer::empty(&path, &[]).expect("a file with nothing in it");
13271        let catalog = Catalog::open(&path).expect("the empty file opens");
13272        assert_eq!(catalog.len(), 0);
13273        assert!(catalog.is_empty());
13274        assert_eq!(catalog.names().count(), 0);
13275        // The next generation goes over the top of it the way it goes over any other, which is what
13276        // says this is a committed file and not a special case somebody has to know about.
13277        let mut writer =
13278            Writer::open(&path, "items", vec![Field::required("id", LogicalType::Integer)])
13279                .expect("a table goes into the empty file");
13280        writer.append(&sample_ids()).expect("rows");
13281        writer.finish().expect("commit");
13282        let catalog = Catalog::open(&path).expect("the file opens again");
13283        assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
13284        fs::remove_file(&path).expect("clean up");
13285    }
13286
13287    /// A committed table with no rows is a name the next generation takes over, and one with rows
13288    /// is a name it refuses.
13289    ///
13290    /// The refusal is what it always was and it is load bearing: carrying a table that holds rows
13291    /// forward means reading and rewriting its pages, and a writer that quietly wrote a second
13292    /// entry under the same name would leave a file with two tables a reader cannot tell apart. An
13293    /// empty one has no pages and no reader, so there is nothing to carry and nothing to lose, and
13294    /// taking its place is what lets a schema committed by an earlier session be loaded by a stream
13295    /// instead of through memory.
13296    #[test]
13297    fn a_committed_empty_table_gives_up_its_name_and_one_with_rows_does_not() {
13298        let path = path("empty-name");
13299        let field = || vec![Field::required("id", LogicalType::Integer)];
13300        Writer::create(&path, "items", field()).expect("new file").finish().expect("commit");
13301        let catalog = Catalog::open(&path).expect("the file opens");
13302        assert_eq!(catalog.rows().collect::<Vec<_>>(), vec![("items", 0)]);
13303
13304        let mut writer = Writer::open(&path, "items", field()).expect("the empty name is free");
13305        writer.append(&sample_ids()).expect("rows");
13306        writer.finish().expect("commit");
13307        let catalog = Catalog::open(&path).expect("the file opens again");
13308        // One entry and not two. The generation replaced the empty table rather than joining it.
13309        assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
13310        let held = catalog.rows().collect::<Vec<_>>();
13311        assert_eq!(held.len(), 1);
13312        assert!(held[0].1 > 0, "the rows that were appended are the ones the catalog counts");
13313
13314        // The same call against the same name now that it holds rows, which is still refused.
13315        let error = Writer::open(&path, "items", field()).expect_err("a name with rows is taken");
13316        assert!(error.to_string().contains("same name"), "{error}");
13317        fs::remove_file(&path).expect("clean up");
13318    }
13319
13320    /// A view, with everything about it that a reopened catalog has to be able to answer from.
13321    fn sample_view(name: &str) -> ViewEntry {
13322        ViewEntry {
13323            name: name.to_string(),
13324            sql: "SELECT id FROM items WHERE id > 0".to_string(),
13325            statement: format!("CREATE VIEW {name} AS SELECT id FROM items WHERE (id > 0);"),
13326            aliases: vec!["n".to_string()],
13327            columns: vec![Field::new("n", LogicalType::Integer)],
13328        }
13329    }
13330
13331    #[test]
13332    fn a_view_written_into_the_catalog_comes_back_whole() {
13333        let path = path("views");
13334        let mut writer =
13335            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
13336                .expect("new file");
13337        writer.append(&sample_ids()).expect("rows");
13338        writer.with_views(vec![sample_view("v")]).finish().expect("commit");
13339        let catalog = Catalog::open(&path).expect("reopen");
13340        assert_eq!(catalog.views().cloned().collect::<Vec<_>>(), vec![sample_view("v")]);
13341        // The tables are still there and are still read the same way, so the section on the end did
13342        // not move anything in front of it.
13343        assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
13344        fs::remove_file(&path).expect("clean up");
13345    }
13346
13347    /// A writer opened to append a table says nothing about views and must not lose them.
13348    #[test]
13349    fn appending_a_table_carries_the_views_forward() {
13350        let path = path("viewscarry");
13351        let mut writer =
13352            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
13353                .expect("new file");
13354        writer.append(&sample_ids()).expect("rows");
13355        writer.with_views(vec![sample_view("v")]).finish().expect("commit");
13356        let mut writer =
13357            Writer::open(&path, "other", vec![Field::required("id", LogicalType::Integer)])
13358                .expect("a second table");
13359        writer.append(&sample_ids()).expect("rows");
13360        writer.finish().expect("commit");
13361        let catalog = Catalog::open(&path).expect("reopen");
13362        assert_eq!(catalog.views().count(), 1);
13363        assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items", "other"]);
13364        fs::remove_file(&path).expect("clean up");
13365    }
13366
13367    /// The whole point of [`Writer::restate`]: the views change and the pages do not move.
13368    #[test]
13369    fn restating_the_views_leaves_every_table_where_it_was() {
13370        let path = path("restate");
13371        let mut writer =
13372            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
13373                .expect("new file");
13374        writer.append(&sample_ids()).expect("rows");
13375        writer.finish().expect("commit");
13376        let before = fs::metadata(&path).expect("the file is there").len();
13377        Writer::restate(&path, &[sample_view("v"), sample_view("w")]).expect("two views");
13378        let catalog = Catalog::open(&path).expect("reopen");
13379        assert_eq!(catalog.views().count(), 2);
13380        assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
13381        // A catalog on the end and nothing else, so what it grew by is the size of a catalog rather
13382        // than the size of the table.
13383        let after = fs::metadata(&path).expect("the file is there").len();
13384        assert!(after > before, "a generation was written");
13385        assert!(after - before < before, "the table was not written again");
13386        // The rows are still readable through the new generation, which is the part that would go
13387        // wrong if the catalog carried the wrong directory pointers forward.
13388        let reader = Catalog::open(&path).expect("reopen").table("items").expect("the table");
13389        assert_eq!(reader.table().rows, 3);
13390        // And a restate over a restate keeps working, because each one reads the slot that
13391        // checksummed rather than the highest number in the header.
13392        Writer::restate(&path, &[]).expect("no views at all");
13393        assert_eq!(Catalog::open(&path).expect("reopen").views().count(), 0);
13394        fs::remove_file(&path).expect("clean up");
13395    }
13396
13397    /// Two entries under one name is a catalog no lookup can answer, whichever two they are.
13398    #[test]
13399    fn a_view_named_after_a_table_is_refused_when_the_catalog_is_read() {
13400        let bytes = encode_catalog(
13401            &[Entry {
13402                name: "items".to_string(),
13403                fields: vec![Field::required("id", LogicalType::Integer)],
13404                rows: 1,
13405                directory: Page { offset: HEADER, length: 8, hash: 0 },
13406                nonzero: vec![None],
13407                aggregates: vec![None],
13408                distincts: vec![None],
13409                extremes: vec![None],
13410                frequencies: vec![None],
13411            }],
13412            &[sample_view("items")],
13413        )
13414        .expect("it encodes, because encoding does not look");
13415        let error = decode_catalog(&bytes, HEADER + 8).expect_err("and decoding does");
13416        assert!(error.to_string().contains("same name"), "{error}");
13417    }
13418
13419    /// A compressed text page read at some rows is those rows of the page read whole, nulls and
13420    /// all, and a row past the end or rows out of order are refused rather than guessed at.
13421    #[test]
13422    fn a_compressed_text_page_read_at_some_rows_is_those_rows_of_the_whole() {
13423        let rows: usize = 300;
13424        let text: Vec<String> =
13425            (0..rows).map(|row| format!("a street named after number {}", row * 7)).collect();
13426        let values: Vec<&[u8]> = text.iter().map(String::as_bytes).collect();
13427        let mut page = vec![6, 2];
13428        page.extend((0..rows.div_ceil(8)).map(|byte| {
13429            (0..8).filter(|bit| (byte * 8 + bit) % 5 != 3).fold(0_u8, |mask, bit| mask | 1 << bit)
13430        }));
13431        let compressed = string::encode_only(string::Kind::Fsst, &values)
13432            .expect("encoded")
13433            .expect("text this repetitive compresses");
13434        page.extend_from_slice(&compressed);
13435        let whole = decode(&LogicalType::Varchar, rows, &page, None).expect("the whole page");
13436        let positions = [0_u32, 3, 8, 13, 200, 299];
13437        let some =
13438            decode_at(&LogicalType::Varchar, rows, &page, None, &positions).expect("some rows");
13439        assert_eq!(some.len(), positions.len());
13440        for (at, &row) in positions.iter().enumerate() {
13441            assert_eq!(some.value_at(at), whole.value_at(row as usize), "row {row}");
13442        }
13443        assert_eq!(some.value_at(1), Value::Null, "row 3 is null");
13444        assert!(decode_at(&LogicalType::Varchar, rows, &page, None, &[300]).is_err());
13445        assert!(decode_at(&LogicalType::Varchar, rows, &page, None, &[8, 3]).is_err());
13446    }
13447
13448    /// Every column of a part read at some rows is the part read whole and gathered, whatever the
13449    /// page holds.
13450    #[test]
13451    fn a_part_read_at_some_rows_is_the_part_read_whole_and_gathered() {
13452        let path = path("rows");
13453        let mut writer = Writer::create(
13454            &path,
13455            "items",
13456            vec![
13457                Field::required("id", LogicalType::Integer),
13458                Field::new("text", LogicalType::Varchar),
13459            ],
13460        )
13461        .expect("new file");
13462        let rows = 2_000;
13463        let chunk = Chunk::new(vec![
13464            Vector::from_values(
13465                LogicalType::Integer,
13466                &(0..rows).map(Value::Integer).collect::<Vec<_>>(),
13467            )
13468            .expect("integers"),
13469            Vector::from_values(
13470                LogicalType::Varchar,
13471                &(0..rows)
13472                    .map(|row| {
13473                        if row % 7 == 2 {
13474                            Value::Null
13475                        } else {
13476                            Value::Varchar(format!("a comment about order {}", row * 13))
13477                        }
13478                    })
13479                    .collect::<Vec<_>>(),
13480            )
13481            .expect("strings"),
13482        ])
13483        .expect("matching rows");
13484        writer.append(&chunk).expect("one part");
13485        writer.finish().expect("commit");
13486        let reader = Reader::open(&path).expect("reopen from disk");
13487        let positions = [1_u32, 2, 9, 1_000, 1_999];
13488        for whole in [true, false] {
13489            let some = reader.read_rows(0, &[0, 1], &positions, whole).expect("some rows");
13490            let all = reader.read(0, &[0, 1]).expect("the whole part");
13491            assert_eq!(some.len(), positions.len());
13492            for column in 0..2 {
13493                for (at, &row) in positions.iter().enumerate() {
13494                    assert_eq!(some.value_at(at, column), all.value_at(row as usize, column));
13495                }
13496            }
13497        }
13498        assert!(reader.read_rows(0, &[1], &[2_000], true).is_err());
13499    }
13500
13501    #[test]
13502    fn committed_file_reopens_and_reads_only_requested_columns() {
13503        let path = path("reopen");
13504        let mut writer = Writer::create(
13505            &path,
13506            "items",
13507            vec![
13508                Field::required("id", LogicalType::Integer),
13509                Field::new("text", LogicalType::Varchar),
13510            ],
13511        )
13512        .expect("new file");
13513        writer.append(&sample()).expect("first part");
13514        writer.append(&sample()).expect("second part");
13515        writer.finish().expect("commit");
13516        let reader = Reader::open(&path).expect("reopen from disk");
13517        assert_eq!(reader.table().rows(), 6);
13518        // Two appends below the stripe bound are two parts of one stripe, which is the whole point
13519        // of the split: the directory describes the stripe and the scan still reads a part.
13520        assert_eq!(reader.table().stripes().len(), 1);
13521        assert_eq!(reader.parts(), 2);
13522        assert_eq!(reader.part_rows(0), 3);
13523        assert_eq!(reader.part_rows(1), 3);
13524        let text = reader.read(1, &[1]).expect("only text page");
13525        assert_eq!(text.width(), 1);
13526        assert_eq!(text.value_at(1, 0), Value::Null);
13527        assert_eq!(text.value_at(2, 0), Value::Varchar("long text after a slash".into()));
13528        let sparse = reader.read_sparse(1, &[1]).expect("one part without its whole page");
13529        assert_eq!(sparse.width(), 1);
13530        assert_eq!(sparse.value_at(1, 0), Value::Null);
13531        assert_eq!(sparse.value_at(2, 0), Value::Varchar("long text after a slash".into()));
13532        assert!(!reader.skips_codes(0, 1, &[0]).expect("alpha is in the stripe"));
13533        assert!(!reader.skips_codes(0, 1, &[2]).expect("long text is in the stripe"));
13534        assert!(reader.skips_codes(0, 1, &[3]).expect("unknown code is absent"));
13535        let count = reader.read(0, &[]).expect("no page is needed for count");
13536        assert_eq!(count.len(), 3);
13537        assert!(reader.skips(0, &[Probe { column: 0, op: Op::Greater, value: Bound::Int(100) }]));
13538        assert!(!reader.skips(0, &[Probe { column: 0, op: Op::Greater, value: Bound::Int(0) }]));
13539        assert_eq!(reader.top_frequencies(0, 1).expect("valid integer synopsis"), None);
13540        let integers = reader.frequency_prefix(0).expect("valid integer synopsis").expect("kept");
13541        assert_eq!(integers.entries, vec![(Value::Integer(-2), 2), (Value::Integer(4), 2)]);
13542        assert_eq!(integers.omitted_max, 2);
13543        let strings = reader.top_frequencies(1, 1).expect("valid string synopsis").expect("kept");
13544        assert_eq!(strings.len(), 3);
13545        assert!(strings.contains(&(Value::Null, 2)));
13546        assert!(strings.contains(&(Value::Varchar("alpha".into()), 2)));
13547        assert!(strings.contains(&(Value::Varchar("long text after a slash".into()), 2)));
13548        fs::remove_file(path).expect("remove scratch file");
13549    }
13550
13551    /// Two pipeline instances handing over whole runs, which is what makes the native sink safe to
13552    /// instance.
13553    ///
13554    /// The runs arrive in the order the instances finished reading them rather than in source
13555    /// order, and the second one to finish is the one that read the earlier rows. Each run is still
13556    /// a stripe of its own and the table still reads back in source order, which is the whole of
13557    /// what the writer promises about ordering.
13558    #[test]
13559    fn runs_handed_over_out_of_order_still_read_back_in_source_order() {
13560        let path = path("interleaved-runs");
13561        let mut writer =
13562            Writer::create(&path, "interleaved", vec![Field::new("v", LogicalType::BigInt)])
13563                .expect("new file");
13564        for morsel in [2_u64, 0, 3, 1] {
13565            let parts = (0..4_u64)
13566                .map(|chunk| {
13567                    let first = i64::try_from(morsel * 32 + chunk * 8).expect("small");
13568                    let values =
13569                        (0..8_i64).map(|row| Value::BigInt(first + row)).collect::<Vec<_>>();
13570                    let column =
13571                        Vector::from_values(LogicalType::BigInt, &values).expect("a column");
13572                    ((morsel, chunk), Chunk::new(vec![column]).expect("one column"))
13573                })
13574                .collect::<Vec<_>>();
13575            writer.append_stripe(parts).expect("a stripe");
13576        }
13577        writer.finish().expect("commit");
13578
13579        let reader = Reader::open(&path).expect("valid directory");
13580        assert_eq!(reader.table().stripes().len(), 4, "a run is a stripe of its own");
13581        assert_eq!(reader.table().rows(), 128);
13582        for part in 0..16_usize {
13583            let read = reader.read(part, &[0]).expect("a part back");
13584            for row in 0..8_usize {
13585                let want = i64::try_from(part * 8 + row).expect("small");
13586                assert_eq!(read.value_at(row, 0), Value::BigInt(want), "part {part} row {row}");
13587            }
13588        }
13589        fs::remove_file(path).expect("remove scratch file");
13590    }
13591
13592    /// Runs from different callers may interleave and may not overlap, and the commit is what
13593    /// catches an overlap.
13594    #[test]
13595    fn runs_that_overlap_each_other_are_refused_at_commit() {
13596        let path = path("overlapping-runs");
13597        let mut writer =
13598            Writer::create(&path, "overlapping", vec![Field::new("v", LogicalType::BigInt)])
13599                .expect("new file");
13600        let one = |order: (u64, u64)| {
13601            let column =
13602                Vector::from_values(LogicalType::BigInt, &[Value::BigInt(1)]).expect("a column");
13603            (order, Chunk::new(vec![column]).expect("one column"))
13604        };
13605        // The second run sits inside the first rather than after it, which is a thing no instance
13606        // holding its own contiguous run can produce and a thing the file cannot represent.
13607        writer.append_stripe(vec![one((0, 0)), one((0, 2))]).expect("a stripe");
13608        writer.append_stripe(vec![one((0, 1))]).expect("a stripe");
13609        let error = writer.finish().expect_err("the runs overlap");
13610        assert!(error.message().contains("source order"), "{error}");
13611        fs::remove_file(path).expect("remove scratch file");
13612    }
13613
13614    /// A stripe holds [`STRIPE_PARTS`] parts, so a run longer than that is a caller bug rather than
13615    /// something to split, and the writer says so at the door instead of quietly cutting it in two.
13616    #[test]
13617    fn a_run_longer_than_a_stripe_is_refused() {
13618        let path = path("overlong-run");
13619        let mut writer =
13620            Writer::create(&path, "overlong", vec![Field::new("v", LogicalType::BigInt)])
13621                .expect("new file");
13622        let parts = (0..=STRIPE_PARTS)
13623            .map(|at| {
13624                let column = Vector::from_values(LogicalType::BigInt, &[Value::BigInt(1)])
13625                    .expect("a column");
13626                let chunk = Chunk::new(vec![column]).expect("one column");
13627                ((0, u64::try_from(at).expect("small")), chunk)
13628            })
13629            .collect::<Vec<_>>();
13630        let error = writer.append_stripe(parts).expect_err("one part too many");
13631        assert!(error.message().contains("more parts than it holds"), "{error}");
13632        fs::remove_file(path).expect("remove scratch file");
13633    }
13634
13635    /// Parts past the stripe bound start a new stripe, and every part stays addressable on its own.
13636    ///
13637    /// This is the shape the format exists for, so both ends of the split are checked here. The
13638    /// directory holds three stripes rather than a hundred and thirty one, and a read of any one
13639    /// part still answers with that part's rows rather than with its whole stripe's.
13640    #[test]
13641    fn parts_past_the_stripe_bound_start_a_new_stripe() {
13642        let path = path("stripe-bound");
13643        let mut writer = Writer::create(
13644            &path,
13645            "items",
13646            vec![
13647                Field::required("id", LogicalType::Integer),
13648                Field::new("text", LogicalType::Varchar),
13649            ],
13650        )
13651        .expect("new file");
13652        let parts = STRIPE_PARTS * 2 + 3;
13653        for part in 0..parts {
13654            let id = part as i32;
13655            let chunk = Chunk::new(vec![
13656                Vector::from_values(
13657                    LogicalType::Integer,
13658                    &[Value::Integer(id), Value::Integer(-id)],
13659                )
13660                .expect("integers"),
13661                Vector::from_values(
13662                    LogicalType::Varchar,
13663                    &[Value::Varchar(format!("value {part}")), Value::Null],
13664                )
13665                .expect("strings"),
13666            ])
13667            .expect("matching rows");
13668            writer.append(&chunk).expect("one part");
13669        }
13670        writer.finish().expect("commit");
13671
13672        let reader = Reader::open(&path).expect("reopen from disk");
13673        assert_eq!(reader.parts(), parts);
13674        assert_eq!(reader.table().rows(), parts * 2);
13675        assert_eq!(reader.table().stripes().len(), parts.div_ceil(STRIPE_PARTS));
13676        assert_eq!(reader.table().stripes()[0].parts(), STRIPE_PARTS);
13677        assert_eq!(reader.table().stripes()[0].rows(), STRIPE_PARTS * 2);
13678        assert_eq!(reader.table().stripes()[2].parts(), 3);
13679        // Backwards on purpose. The reader keeps four stripes a column, so a scan that walks the
13680        // table the other way is what catches a cache that only ever holds what it just read.
13681        for part in (0..parts).rev() {
13682            let dense = reader.read(part, &[0, 1]).expect("a whole page read");
13683            let sparse = reader.read_sparse(part, &[0, 1]).expect("one part read");
13684            for chunk in [&dense, &sparse] {
13685                assert_eq!(chunk.len(), 2, "part {part} has its own row count");
13686                assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
13687                assert_eq!(chunk.value_at(1, 0), Value::Integer(-(part as i32)));
13688                assert_eq!(chunk.value_at(0, 1), Value::Varchar(format!("value {part}")));
13689                assert_eq!(chunk.value_at(1, 1), Value::Null);
13690            }
13691        }
13692        // The bounds are merged over the stripe, so they answer for the range the whole stripe
13693        // covers and not for the part that was asked about.
13694        let above = [Probe { column: 0, op: Op::Greater, value: Bound::Int(100) }];
13695        assert!(reader.skips(0, &above), "the first stripe stops at 63");
13696        assert!(!reader.skips(STRIPE_PARTS * 2, &above), "the third stripe reaches 130");
13697        fs::remove_file(path).expect("remove scratch file");
13698    }
13699
13700    /// A scattered value in the column that decides `WHERE UserID = ?`.
13701    fn scattered(n: i64) -> i64 {
13702        n.wrapping_mul(-7_046_029_254_386_353_131)
13703    }
13704
13705    /// A part whose sieve does not hold the constant is skipped, and a range would skip none of them.
13706    ///
13707    /// This is ClickBench query 19 in miniature. The values are spread over the whole of `BIGINT`, so
13708    /// every stripe's bounds cover nearly all of it and rule out nothing, and the part that really
13709    /// holds the value is the only one a scan has to read.
13710    #[test]
13711    fn a_part_is_skipped_when_its_sieve_does_not_hold_the_constant() {
13712        let path = path("sieve-skip");
13713        let mut writer =
13714            Writer::create(&path, "hits", vec![Field::required("id", LogicalType::BigInt)])
13715                .expect("new file");
13716        let parts = STRIPE_PARTS + 3;
13717        // Big enough that the filter is worth its bytes. A part of eight numbers packs to under a
13718        // hundred bytes and the smallest filter there is is sixty nine, so a filter over a part
13719        // that small costs about as much to read as the rows do and is no longer written.
13720        let per_part = 128;
13721        for part in 0..parts {
13722            let held: Vec<Value> = (0..per_part)
13723                .map(|row| Value::BigInt(scattered((part * per_part + row) as i64)))
13724                .collect();
13725            let chunk =
13726                Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
13727                    .expect("one column");
13728            writer.append(&chunk).expect("one part");
13729        }
13730        writer.finish().expect("commit");
13731
13732        let reader = Reader::open(&path).expect("reopen from disk");
13733        let probe = |value: i64| Probe {
13734            column: 0,
13735            op: Op::Equal,
13736            value: Bound::Int(i128::from(scattered(value))),
13737        };
13738        for wanted in [0_i64, (per_part + 1) as i64, (parts * per_part - 1) as i64] {
13739            let tests = [probe(wanted)];
13740            let kept: Vec<usize> = (0..parts).filter(|&part| !reader.skips(part, &tests)).collect();
13741            let home = wanted as usize / per_part;
13742            assert!(kept.contains(&home), "the part holding {wanted} is read");
13743            // A filter answers maybe, so a part it keeps need not hold the value. Sixty seven parts
13744            // of a hundred and twenty eight numbers each, at a dozen bits a value, is about one
13745            // stray part across the whole file and that is what this leaves room for.
13746            assert!(kept.len() <= 2, "{wanted} keeps {kept:?}, which is more than one stray part");
13747        }
13748        let absent = [probe((parts * per_part) as i64 + 1)];
13749        let kept = (0..parts).filter(|&part| !reader.skips(part, &absent)).count();
13750        assert!(kept <= 1, "{kept} parts of {parts} kept a value no part holds");
13751        // The same probes against the bounds alone, which is what this replaces. A column of
13752        // scattered numbers has a range per stripe that covers nearly the whole type.
13753        let tests = [probe(0)];
13754        assert!(
13755            reader.table().stripes().iter().all(|stripe| !stripe.zone.skips(&tests)),
13756            "the bounds rule out no stripe at all"
13757        );
13758        fs::remove_file(path).expect("remove scratch file");
13759    }
13760
13761    /// A part whose own bounds rule out an ordered comparison is skipped where the stripe's keep it.
13762    ///
13763    /// This is the shape of ClickBench 24. Each part covers a narrow stretch of the column and the
13764    /// stripe covers all sixty four of them at once, so a comparison that lands inside the stripe
13765    /// rules out none of it and rules out all but a few parts.
13766    #[test]
13767    fn a_part_is_skipped_when_its_own_bounds_rule_out_a_comparison_the_stripe_keeps() {
13768        let path = path("part-range-skip");
13769        let mut writer =
13770            Writer::create(&path, "hits", vec![Field::required("at", LogicalType::BigInt)])
13771                .expect("new file");
13772        let parts = STRIPE_PARTS + 3;
13773        let per_part = 128;
13774        for part in 0..parts {
13775            // Scattered inside the part's own band rather than a run, because a run of
13776            // consecutive numbers encodes to a stride of a few bytes and then the page of ranges
13777            // costs more than reading the column it indexes, which is the case the writer declines.
13778            let held: Vec<Value> = (0..per_part)
13779                .map(|row| {
13780                    Value::BigInt((part * 1_000) as i64 + (scattered(row as i64).rem_euclid(900)))
13781                })
13782                .collect();
13783            let chunk =
13784                Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
13785                    .expect("one column");
13786            writer.append(&chunk).expect("one part");
13787        }
13788        writer.finish().expect("commit");
13789
13790        let reader = Reader::open(&path).expect("reopen from disk");
13791        let under = [Probe { column: 0, op: Op::Less, value: Bound::Int(3_000) }];
13792        let kept: Vec<usize> = (0..parts).filter(|&part| !reader.skips(part, &under)).collect();
13793        assert_eq!(kept, vec![0, 1, 2], "only the three parts that start under three thousand");
13794        // The same question asked of the stripe alone, which is what this replaces.
13795        assert!(!reader.stripe_skips(0, &under), "the stripe reaches from zero and keeps itself");
13796        fs::remove_file(path).expect("remove scratch file");
13797    }
13798
13799    /// The other half of the same page. A part whose own bounds put every row of it inside the
13800    /// filter is waved through, so the comparison never runs on it, where the stripe's bounds reach
13801    /// across every part and can prove nothing.
13802    #[test]
13803    fn a_part_is_waved_through_when_its_own_bounds_pass_a_comparison_the_stripe_cannot() {
13804        let path = path("part-range-certain");
13805        let mut writer =
13806            Writer::create(&path, "hits", vec![Field::required("at", LogicalType::BigInt)])
13807                .expect("new file");
13808        let parts = STRIPE_PARTS + 3;
13809        let per_part = 128;
13810        for part in 0..parts {
13811            let held: Vec<Value> = (0..per_part)
13812                .map(|row| {
13813                    Value::BigInt((part * 1_000) as i64 + (scattered(row as i64).rem_euclid(900)))
13814                })
13815                .collect();
13816            let chunk =
13817                Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
13818                    .expect("one column");
13819            writer.append(&chunk).expect("one part");
13820        }
13821        writer.finish().expect("commit");
13822
13823        let reader = Reader::open(&path).expect("reopen from disk");
13824        let under = [Probe { column: 0, op: Op::Less, value: Bound::Int(3_000) }];
13825        let waved: Vec<usize> = (0..parts).filter(|&part| reader.certain(part, &under)).collect();
13826        assert_eq!(waved, vec![0, 1, 2], "the three parts that end under three thousand");
13827        // The first stripe reaches from zero to past sixty thousand, so it straddles three thousand
13828        // and settles nothing either way. The three yeses above are the parts' own ends talking.
13829        assert!(!reader.stripe_skips(0, &under), "the stripe straddles the comparison");
13830        fs::remove_file(path).expect("remove scratch file");
13831    }
13832
13833    /// The page is worth its bytes on a column with parts to tell apart and is not written on one
13834    /// that has a single part, where the stripe bounds already are the part's.
13835    #[test]
13836    fn a_stripe_of_one_part_writes_no_range_page_and_a_stripe_of_many_does() {
13837        for (parts, wanted) in [(1_usize, false), (STRIPE_PARTS, true)] {
13838            let path = path("part-range-page");
13839            let mut writer =
13840                Writer::create(&path, "hits", vec![Field::required("at", LogicalType::BigInt)])
13841                    .expect("new file");
13842            for part in 0..parts {
13843                let held: Vec<Value> = (0..128)
13844                    .map(|row| {
13845                        Value::BigInt((part * 1_000) as i64 + scattered(row as i64).rem_euclid(900))
13846                    })
13847                    .collect();
13848                let chunk = Chunk::new(vec![
13849                    Vector::from_values(LogicalType::BigInt, &held).expect("numbers"),
13850                ])
13851                .expect("one column");
13852                writer.append(&chunk).expect("one part");
13853            }
13854            writer.finish().expect("commit");
13855            let reader = Reader::open(&path).expect("reopen from disk");
13856            let bytes = reader.layout().columns[0].part_ranges;
13857            assert_eq!(bytes > 0, wanted, "{parts} parts wrote {bytes} bytes of ranges");
13858            fs::remove_file(path).expect("remove scratch file");
13859        }
13860    }
13861
13862    /// A cut down string end is still an end on the side it was, which is the only thing that keeps
13863    /// a shortened bound from turning a skip into a wrong answer.
13864    #[test]
13865    fn a_string_end_that_is_cut_down_still_covers_the_value_it_came_from() {
13866        let long = vec![b'a'; PART_BOUND_BYTES * 2];
13867        let low = shortened(Some(Bound::Bytes(long.clone())), false).expect("a low end");
13868        let high = shortened(Some(Bound::Bytes(long.clone())), true).expect("a high end");
13869        let Bound::Bytes(low) = low else { panic!("a string stays a string") };
13870        let Bound::Bytes(high) = high else { panic!("a string stays a string") };
13871        assert!(low.len() <= PART_BOUND_BYTES && high.len() <= PART_BOUND_BYTES);
13872        assert!(low.as_slice() <= long.as_slice(), "the low end is at or under the value");
13873        assert!(high.as_slice() >= long.as_slice(), "the high end is at or over the value");
13874    }
13875
13876    /// A string of nothing but the largest byte has no prefix that can be stepped up, so the high
13877    /// end is given up rather than claimed too small. No end keeps the part, which is always safe.
13878    #[test]
13879    fn a_string_end_with_no_room_to_step_up_gives_up_the_bound() {
13880        let long = vec![u8::MAX; PART_BOUND_BYTES * 2];
13881        assert_eq!(shortened(Some(Bound::Bytes(long.clone())), true), None);
13882        let low = shortened(Some(Bound::Bytes(long)), false).expect("a low end is still a prefix");
13883        assert_eq!(low, Bound::Bytes(vec![u8::MAX; PART_BOUND_BYTES]));
13884    }
13885
13886    /// What a column is stored as, asked of two files holding the same rows in a different order.
13887    ///
13888    /// This is the question the report exists to answer and it is the one the directory cannot. The
13889    /// two files have the same rows, the same schema and the same number of parts, and the column
13890    /// comes out four times smaller in one of them, because ascending keys delta encode to a few
13891    /// bits a row and shuffled ones do not. Nothing about the file's shape says so. The page header
13892    /// says so, and reading it is what this does.
13893    ///
13894    /// It is q18 on TPC-H in miniature: clustering lineitem by ship date leaves `l_orderkey`
13895    /// ascending inside a partition but sparse, its deltas go from six bits to twelve, and the scan
13896    /// pays for the wider ones.
13897    #[test]
13898    fn what_a_column_is_stored_as_follows_the_order_the_rows_were_written_in() {
13899        let parts = 4;
13900        let per_part = 1024;
13901        let rows = parts * per_part;
13902        let written = |name: &str, keys: &[i64]| {
13903            let path = path(name);
13904            let fields = vec![Field::required("key", LogicalType::BigInt)];
13905            let mut writer = Writer::create(&path, "keys", fields).expect("new file");
13906            for part in 0..parts {
13907                let values: Vec<Value> = keys[part * per_part..(part + 1) * per_part]
13908                    .iter()
13909                    .map(|key| Value::BigInt(*key))
13910                    .collect();
13911                let chunk = Chunk::new(vec![
13912                    Vector::from_values(LogicalType::BigInt, &values).expect("numbers"),
13913                ])
13914                .expect("one column");
13915                writer.append(&chunk).expect("one part");
13916            }
13917            writer.finish().expect("commit");
13918            path
13919        };
13920        // Ascending with a small irregular step, which is what a key column in arrival order looks
13921        // like: an order has one to seven line items, so the key repeats and then moves on by one.
13922        let climbing = |step: &dyn Fn(usize) -> i64| {
13923            let mut key = 0;
13924            (0..rows)
13925                .map(|row| {
13926                    key += step(row);
13927                    key
13928                })
13929                .collect::<Vec<i64>>()
13930        };
13931        let ascending = climbing(&|row| (row % 3) as i64);
13932        // The same rows in the same direction over a range a thousand times wider, which is what a
13933        // partition of a clustered table holds: still ascending, and far enough apart that the
13934        // deltas no longer fit in a handful of bits.
13935        let sparse = climbing(&|row| ((row * 2_654_435_761) % 4096) as i64);
13936        let near_path = written("stored-near", &ascending);
13937        let far_path = written("stored-far", &sparse);
13938
13939        let one = Reader::open(&near_path).expect("reopen from disk");
13940        let other = Reader::open(&far_path).expect("reopen from disk");
13941        let near = one.stored(0).expect("the column is stored");
13942        let far = other.stored(0).expect("the column is stored");
13943        assert_eq!(near.len(), parts, "one row per part");
13944        assert_eq!(far.len(), parts);
13945        // The bytes are the same bytes the directory totals, which is the check that this is
13946        // reading the pages the file really holds rather than some other pages.
13947        let total = |stored: &[StoredPart]| stored.iter().map(|part| part.bytes).sum::<u64>();
13948        assert_eq!(total(&near), one.layout().columns[0].pages);
13949        assert_eq!(total(&far), other.layout().columns[0].pages);
13950        assert!(
13951            total(&near) * 2 < total(&far),
13952            "the sparse keys cost more, {} against {}",
13953            total(&far),
13954            total(&near)
13955        );
13956        // Every part accounted for, in order, with the row it starts at following the one before.
13957        for (at, part) in near.iter().enumerate() {
13958            assert_eq!(part.part, at);
13959            assert_eq!(part.row, at * per_part);
13960            assert_eq!(part.rows, per_part);
13961            let held = &ascending[at * per_part..(at + 1) * per_part];
13962            assert_eq!(part.low, Some(Value::BigInt(held[0])));
13963            assert_eq!(part.high, Some(Value::BigInt(held[per_part - 1])));
13964            assert_eq!(part.nulls, Some(0));
13965        }
13966        // And the encoding is a line of text that names what the encoder chose, which is the whole
13967        // point. Both are a cascade over deltas and the widths inside them are what differ.
13968        assert!(near[0].encoding.contains("DELTA"), "{}", near[0].encoding);
13969        assert!(far[0].encoding.contains("DELTA"), "{}", far[0].encoding);
13970        assert_ne!(near[0].encoding, far[0].encoding);
13971        fs::remove_file(near_path).expect("remove scratch file");
13972        fs::remove_file(far_path).expect("remove scratch file");
13973    }
13974
13975    /// A sieve bigger than the part it indexes is not written, and one smaller than it still is.
13976    ///
13977    /// Both columns hold values spread over the whole of `BIGINT`, so neither gets a bitmap and both
13978    /// reach the filter. They differ in what the part costs to read. `spread` is a thousand distinct
13979    /// numbers and packs to eight kilobytes, so a filter of about thirteen hundred bytes is a good
13980    /// trade. `repeated` is the same thousand rows over four numbers in runs and encodes to
13981    /// almost nothing, but the filter is sized for the rows rather than the values it turns out to
13982    /// hold, so it comes out larger than the data. Reading it to decide whether to read the part spends more than
13983    /// the part, every time, and that is the case this drops.
13984    #[test]
13985    fn a_sieve_larger_than_the_part_it_indexes_is_not_written() {
13986        let path = path("sieve-pays");
13987        let fields = vec![
13988            Field::required("spread", LogicalType::BigInt),
13989            Field::required("repeated", LogicalType::BigInt),
13990        ];
13991        let mut writer = Writer::create(&path, "hits", fields).expect("new file");
13992        let parts = 3;
13993        let per_part = 1024;
13994        for part in 0..parts {
13995            let base = (part * per_part) as i64;
13996            let spread: Vec<Value> =
13997                (0..per_part).map(|row| Value::BigInt(scattered(base + row as i64))).collect();
13998            let repeated: Vec<Value> =
13999                (0..per_part).map(|row| Value::BigInt(scattered((row / 256) as i64))).collect();
14000            let chunk = Chunk::new(vec![
14001                Vector::from_values(LogicalType::BigInt, &spread).expect("numbers"),
14002                Vector::from_values(LogicalType::BigInt, &repeated).expect("numbers"),
14003            ])
14004            .expect("two columns");
14005            writer.append(&chunk).expect("one part");
14006        }
14007        writer.finish().expect("commit");
14008
14009        let reader = Reader::open(&path).expect("reopen from disk");
14010        let layout = reader.layout();
14011        let spread = &layout.columns[0];
14012        let repeated = &layout.columns[1];
14013        assert!(spread.sieves > 0, "a column whose parts are worth a filter keeps one");
14014        assert_eq!(
14015            repeated.sieves, 0,
14016            "a column whose filter costs more than its parts keeps none"
14017        );
14018        // Per part this is the rule itself, so it holds over the column as well: a part without a
14019        // sieve adds to one side of this and to nothing on the other.
14020        for column in &layout.columns {
14021            assert!(
14022                column.sieves < column.pages,
14023                "{} spends {} on sieves over {} of data",
14024                column.name,
14025                column.sieves,
14026                column.pages
14027            );
14028        }
14029        // The filter that was kept still does what it is for.
14030        let absent = [Probe {
14031            column: 0,
14032            op: Op::Equal,
14033            value: Bound::Int(i128::from(scattered((parts * per_part) as i64 + 1))),
14034        }];
14035        assert!((0..parts).all(|part| reader.skips(part, &absent)), "no part holds it");
14036        fs::remove_file(path).expect("remove scratch file");
14037    }
14038
14039    /// A damaged sieve page is a part that gets read, not a query that fails.
14040    ///
14041    /// A sieve is an index over rows that are still there and still correct, so losing one costs
14042    /// time and costs no answers. That is the opposite of the membership index beside it, which is
14043    /// the only thing standing between a string page and a wrong answer.
14044    #[test]
14045    fn a_damaged_sieve_page_is_read_through_rather_than_refused() {
14046        let path = path("sieve-damaged");
14047        let mut writer =
14048            Writer::create(&path, "hits", vec![Field::required("id", LogicalType::BigInt)])
14049                .expect("new file");
14050        let rows = 128;
14051        let held: Vec<Value> = (0..rows).map(|row| Value::BigInt(scattered(row))).collect();
14052        let chunk =
14053            Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
14054                .expect("one column");
14055        writer.append(&chunk).expect("one part");
14056        writer.finish().expect("commit");
14057
14058        let page = Reader::open(&path).expect("reopen").table.stripes[0]
14059            .sieves
14060            .get(0)
14061            .expect("a sieve page");
14062        let mut file = OpenOptions::new().write(true).open(&path).expect("open the sieve page");
14063        file.seek(SeekFrom::Start(page.offset + u64::from(page.length) - 1)).expect("seek");
14064        file.write_all(&[0xff]).expect("damage one byte");
14065        drop(file);
14066
14067        let reader = Reader::open(&path).expect("reopen the damaged file");
14068        let absent =
14069            [Probe { column: 0, op: Op::Equal, value: Bound::Int(i128::from(scattered(99))) }];
14070        assert!(!reader.skips(0, &absent), "a sieve that cannot be read skips nothing");
14071        assert_eq!(
14072            reader.read(0, &[0]).expect("the rows are untouched").len(),
14073            usize::try_from(rows).expect("a small count")
14074        );
14075        fs::remove_file(path).expect("remove scratch file");
14076    }
14077
14078    /// Eight workers over one stripe read it once between them.
14079    ///
14080    /// This is the shape a scan actually has. Parts are handed out in order, so every worker on a
14081    /// column crosses into a stripe within a few parts of the others, and before [`Reader::held`]
14082    /// started sharing the read every one of them read the whole page. On the full ClickBench file
14083    /// that was a `MIN(EventDate), MAX(EventDate)` moving 3.2 GB off the disk to look at 400 MB of
14084    /// column, which is most of what a first touch costs.
14085    ///
14086    /// The workers that lose the race still answer, out of the part reads they do instead, which is
14087    /// what the values below are checking.
14088    #[test]
14089    fn workers_that_want_the_same_stripe_read_it_once() {
14090        let path = path("single-flight");
14091        let mut writer =
14092            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
14093                .expect("new file");
14094        for part in 0..STRIPE_PARTS {
14095            let id = part as i32;
14096            let chunk = Chunk::new(vec![
14097                Vector::from_values(
14098                    LogicalType::Integer,
14099                    &[Value::Integer(id), Value::Integer(-id)],
14100                )
14101                .expect("integers"),
14102            ])
14103            .expect("matching rows");
14104            writer.append(&chunk).expect("one part");
14105        }
14106        writer.finish().expect("commit");
14107
14108        let reader = Reader::open(&path).expect("reopen from disk");
14109        assert_eq!(reader.table().stripes().len(), 1, "one stripe is the point of the test");
14110        let barrier = std::sync::Barrier::new(8);
14111        std::thread::scope(|scope| {
14112            for worker in 0..8 {
14113                let reader = &reader;
14114                let barrier = &barrier;
14115                scope.spawn(move || {
14116                    barrier.wait();
14117                    for part in (worker..STRIPE_PARTS).step_by(8) {
14118                        let chunk = reader.read(part, &[0]).expect("a whole page read");
14119                        assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
14120                        assert_eq!(chunk.value_at(1, 0), Value::Integer(-(part as i32)));
14121                    }
14122                });
14123            }
14124        });
14125        assert_eq!(reader.pages.load(Atomic::Relaxed), 1, "one stripe, one page read, whoever won");
14126        fs::remove_file(path).expect("remove scratch file");
14127    }
14128
14129    /// Opening a file reads the header and the directory, and nothing that depends on the rows.
14130    ///
14131    /// `spec/stats/04-in-memory.md` section 4.2. There are no statistics in the file yet, so this
14132    /// holds today by not having anything to load, and that is exactly why it is worth pinning now.
14133    /// The change that breaks it is the reasonable looking one: summaries are a few hundred bytes,
14134    /// the next query will want them, so read them on the way past. A process that opened the
14135    /// database to run one trivial query pays for all of it and gets nothing.
14136    ///
14137    /// Two files of the same shape and a thousand times the rows in one of them, opened, and the
14138    /// two openings cost the same. The stripe count is held equal so that the directory is the same
14139    /// size in both, which leaves the rows as the only thing that changed. Anything read out of the
14140    /// data would show up here.
14141    #[test]
14142    fn opening_costs_the_same_over_a_thousand_times_the_rows() {
14143        let opened = |label: &str, rows_per_part: i32| {
14144            let path = path(label);
14145            let mut writer =
14146                Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
14147                    .expect("new file");
14148            for part in 0..STRIPE_PARTS * 3 {
14149                // Scrambled rather than sequential, so that the fat file is actually fatter. A run
14150                // of consecutive integers encodes to almost nothing and would leave the two files
14151                // the same size, which would make this test pass for the wrong reason.
14152                let values = (0..rows_per_part)
14153                    .map(|row| {
14154                        Value::Integer((part as i32 * rows_per_part + row).wrapping_mul(2_654_435))
14155                    })
14156                    .collect::<Vec<_>>();
14157                let chunk = Chunk::new(vec![
14158                    Vector::from_values(LogicalType::Integer, &values).expect("integers"),
14159                ])
14160                .expect("matching rows");
14161                writer.append(&chunk).expect("one part");
14162            }
14163            writer.finish().expect("commit");
14164            let reader = Reader::open(&path).expect("reopen from disk");
14165            let size = fs::metadata(&path).expect("the file is there").len();
14166            let out = (reader.reads(), reader.table().stripes().len(), size);
14167            fs::remove_file(path).expect("remove scratch file");
14168            out
14169        };
14170
14171        let (thin, thin_stripes, thin_size) = opened("open-thin", 1);
14172        let (fat, fat_stripes, fat_size) = opened("open-fat", 1000);
14173        assert_eq!(
14174            thin_stripes, fat_stripes,
14175            "the same stripe count is what makes this a fair ask"
14176        );
14177        assert!(
14178            fat_size > thin_size * 50,
14179            "the fat file has to actually be larger, and it is {fat_size} against {thin_size}"
14180        );
14181
14182        assert_eq!(thin.opening.reads, fat.opening.reads, "the same reads either way");
14183        assert_eq!(thin.pages, 0, "opening read a page");
14184        assert_eq!(fat.pages, 0, "opening read a page");
14185        assert_eq!(thin.indexes, 0, "opening read an index");
14186        assert_eq!(fat.indexes, 0, "opening read an index");
14187        // Not exactly equal, because a directory holds offsets and a larger file has larger ones,
14188        // and a handful of bytes of varint is not somebody loading statistics. A factor is.
14189        assert!(
14190            fat.opening.bytes < thin.opening.bytes * 2,
14191            "opening the thin file read {} bytes and the fat one read {}",
14192            thin.opening.bytes,
14193            fat.opening.bytes
14194        );
14195    }
14196
14197    /// The reads a file costs to open are fixed by its shape and not by what ran before.
14198    ///
14199    /// `spec/stats/04-in-memory.md` section 4.3, which is the rule that keeps a plan reproducible:
14200    /// the plan is a function of the data, the generation and the settings, and never of what
14201    /// happened to be in cache. Opening the same file twice in the same process has to cost the
14202    /// same, because a second open that read less would be an open that was about to plan
14203    /// differently.
14204    #[test]
14205    fn two_opens_of_one_file_cost_the_same_and_the_second_is_not_cheaper() {
14206        let path = path("open-twice");
14207        let mut writer =
14208            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
14209                .expect("new file");
14210        for part in 0..STRIPE_PARTS * 3 {
14211            let chunk = Chunk::new(vec![
14212                Vector::from_values(LogicalType::Integer, &[Value::Integer(part as i32)])
14213                    .expect("integers"),
14214            ])
14215            .expect("matching rows");
14216            writer.append(&chunk).expect("one part");
14217        }
14218        writer.finish().expect("commit");
14219
14220        let first = Reader::open(&path).expect("open");
14221        // A whole scan in between, so the operating system's page cache is as warm as it gets and
14222        // anything that consulted it would show up in the second open.
14223        for part in 0..first.parts() {
14224            first.read(part, &[0]).expect("a part");
14225        }
14226        assert!(first.reads().pages > 0, "the scan has to have read something");
14227        let second = Reader::open(&path).expect("open again");
14228
14229        assert_eq!(first.reads().opening, second.reads().opening);
14230        assert_eq!(
14231            second.reads().pages,
14232            0,
14233            "the second open read a page off the back of the first"
14234        );
14235        assert_eq!(second.reads().indexes, 0, "the second open read an index it inherited");
14236        fs::remove_file(path).expect("remove scratch file");
14237    }
14238
14239    /// A scan reads a stripe's index once for the whole scan, not once per part that misses.
14240    ///
14241    /// The page cache holds four stripes and an index used to ride inside it, so a table with more
14242    /// stripes than that read the index again every time a stripe came back around. The index is a
14243    /// few hundred bytes and the page is a quarter of a megabyte, which is why they are now under
14244    /// different budgets. This is the test that keeps them there, since the saving is small enough
14245    /// that nothing in a benchmark would notice it going away again.
14246    #[test]
14247    fn an_index_is_read_once_per_stripe_however_often_the_page_is_evicted() {
14248        let path = path("index-cache");
14249        let mut writer =
14250            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
14251                .expect("new file");
14252        let parts = STRIPE_PARTS * (CACHED_STRIPES_PER_COLUMN + 2);
14253        for part in 0..parts {
14254            let id = part as i32;
14255            let chunk = Chunk::new(vec![
14256                Vector::from_values(LogicalType::Integer, &[Value::Integer(id)]).expect("integers"),
14257            ])
14258            .expect("matching rows");
14259            writer.append(&chunk).expect("one part");
14260        }
14261        writer.finish().expect("commit");
14262
14263        let reader = Reader::open(&path).expect("reopen from disk");
14264        let stripes = reader.table().stripes().len();
14265        assert!(stripes > CACHED_STRIPES_PER_COLUMN, "the page cache has to be too small for this");
14266        // Twice over, so that the second pass finds every page evicted and every index kept.
14267        for _ in 0..2 {
14268            for part in 0..parts {
14269                let chunk = reader.read(part, &[0]).expect("a part");
14270                assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
14271            }
14272        }
14273        assert_eq!(reader.indexes.load(Atomic::Relaxed), stripes, "one index read per stripe");
14274        assert!(
14275            reader.pages.load(Atomic::Relaxed) > stripes,
14276            "the pages are the ones that get read again, which is what makes the index count mean \
14277             something"
14278        );
14279        fs::remove_file(path).expect("remove scratch file");
14280    }
14281
14282    /// A page stays in memory from one scan to the next while the pool has room for it, and a
14283    /// table that is being read takes room from one that is not, down to the floor and no further.
14284    ///
14285    /// This is what the pool is for. Each reader lives as long as its database, so a second query
14286    /// over the same table should find every page it read the first time, and before the pool it
14287    /// found four stripes a column and read the rest off the file again.
14288    #[test]
14289    fn a_pool_keeps_pages_between_scans_and_gives_them_to_the_table_being_read() {
14290        let path = path("page-pool");
14291        let parts = STRIPE_PARTS * (CACHED_STRIPES_PER_COLUMN * 2 + 2);
14292        let fields = || vec![Field::required("id", LogicalType::Integer)];
14293        let mut writer = Writer::create(&path, "a", fields()).expect("new file");
14294        for table in ["a", "b"] {
14295            if table == "b" {
14296                writer = writer.next("b".to_string(), fields()).expect("a second table");
14297            }
14298            for part in 0..parts {
14299                let chunk = Chunk::new(vec![
14300                    Vector::from_values(LogicalType::Integer, &[Value::Integer(part as i32)])
14301                        .expect("integers"),
14302                ])
14303                .expect("matching rows");
14304                writer.append(&chunk).expect("one part");
14305            }
14306        }
14307        writer.finish().expect("commit");
14308
14309        let pool = PagePool::new(usize::MAX);
14310        let catalog = Catalog::open_in(&path, &pool).expect("the file opens");
14311        let (a, b) = (catalog.table("a").expect("a"), catalog.table("b").expect("b"));
14312        let stripes = a.table().stripes().len();
14313        assert!(
14314            stripes > CACHED_STRIPES_PER_COLUMN * 2,
14315            "the floor has to be smaller than a table"
14316        );
14317        let scan = |reader: &Reader| {
14318            for part in 0..parts {
14319                let chunk = reader.read(part, &[0]).expect("a part");
14320                assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
14321            }
14322        };
14323        // The first scan keeps the newest pages of the floor and no more, so the second reads the
14324        // rest again and keeps them, and the third reads nothing.
14325        scan(&a);
14326        assert_eq!(pool.bytes(), 0, "a page read once is not the pool's");
14327        scan(&a);
14328        let twice = stripes * 2 - CACHED_STRIPES_PER_COLUMN;
14329        assert_eq!(a.pages.load(Atomic::Relaxed), twice, "the second scan reads the rest again");
14330        scan(&a);
14331        assert_eq!(a.pages.load(Atomic::Relaxed), twice, "the third scan reads nothing");
14332        let one = pool.bytes();
14333        assert!(one > 0, "the pool counts what the reader holds");
14334
14335        // Room for one table. Reading the other takes the first one's pages down to its floor.
14336        pool.budget.store(one, Atomic::Relaxed);
14337        scan(&b);
14338        scan(&b);
14339        assert_eq!(b.pages.load(Atomic::Relaxed), twice, "a page is never let go while in use");
14340        assert_eq!(a.cache.held[0].load(Atomic::Relaxed), CACHED_STRIPES_PER_COLUMN);
14341        let column = a.cache.columns[0].lock().expect("the column");
14342        let held = column.pages.iter().flatten().count();
14343        assert_eq!(
14344            held,
14345            CACHED_STRIPES_PER_COLUMN + column.passing.len(),
14346            "the count and the slots agree"
14347        );
14348        drop(column);
14349
14350        // A reader that goes takes its pages out of the count with it.
14351        drop((a, b, catalog));
14352        let c = Catalog::open_in(&path, &pool).expect("again").table("a").expect("a");
14353        scan(&c);
14354        scan(&c);
14355        assert!(pool.bytes() <= one, "only what the live reader holds is counted");
14356        fs::remove_file(path).expect("remove scratch file");
14357    }
14358
14359    /// A worker per stripe reads its stripe once, once the cache has been told how many there are.
14360    ///
14361    /// This is the shape a scan has when it hands out a whole stripe per morsel rather than a part.
14362    /// Nobody races for a page any more, but every worker holds a different one for the length of a
14363    /// stripe, so a cache that keeps four pages while eight workers are in eight stripes evicts
14364    /// every one of them before its owner has finished with it, and the owner reads a quarter of a
14365    /// megabyte again for the next part. The barrier is what makes that certain rather than likely:
14366    /// without it a worker can run a whole stripe before the next one starts and never collide.
14367    #[test]
14368    fn a_worker_per_stripe_reads_its_page_once_when_the_cache_was_told_to_expect_it() {
14369        let workers = CACHED_STRIPES_PER_COLUMN + 4;
14370        let path = path("stripe-per-worker");
14371        let mut writer =
14372            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
14373                .expect("new file");
14374        for part in 0..STRIPE_PARTS * workers {
14375            let chunk = Chunk::new(vec![
14376                Vector::from_values(LogicalType::Integer, &[Value::Integer(part as i32)])
14377                    .expect("integers"),
14378            ])
14379            .expect("matching rows");
14380            writer.append(&chunk).expect("one part");
14381        }
14382        writer.finish().expect("commit");
14383
14384        let read = |told: bool| {
14385            let reader = Reader::open(&path).expect("reopen from disk");
14386            assert_eq!(reader.table().stripes().len(), workers, "a stripe per worker");
14387            if told {
14388                reader.keep_stripes(workers);
14389            }
14390            let barrier = std::sync::Barrier::new(workers);
14391            std::thread::scope(|scope| {
14392                for (worker, run) in reader.stripe_parts().into_iter().enumerate() {
14393                    let reader = &reader;
14394                    let barrier = &barrier;
14395                    scope.spawn(move || {
14396                        for part in run {
14397                            barrier.wait();
14398                            let chunk = reader.read(part, &[0]).expect("a part of my own stripe");
14399                            assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
14400                        }
14401                        assert!(worker < workers);
14402                    });
14403                }
14404            });
14405            reader.pages.load(Atomic::Relaxed)
14406        };
14407
14408        assert_eq!(read(true), workers, "one page read per stripe and no more");
14409        assert!(read(false) > workers, "a cache that small is read again on every part");
14410        fs::remove_file(path).expect("remove scratch file");
14411    }
14412
14413    /// A damaged index page is caught before anything decodes a part out of it.
14414    ///
14415    /// The index is the one structure a reader trusts to find bytes with, so it carries a checksum
14416    /// per column section rather than one for the page, and this is what says that check runs.
14417    #[test]
14418    fn a_damaged_index_page_is_an_error() {
14419        let path = path("damaged-index");
14420        let mut writer =
14421            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
14422                .expect("new file");
14423        writer.append(&sample_ids()).expect("first part");
14424        writer.append(&sample_ids()).expect("second part");
14425        writer.finish().expect("commit");
14426
14427        let reader = Reader::open(&path).expect("valid directory");
14428        let index = reader.table.stripes[0].index;
14429        let mut byte = [0; 1];
14430        read_at(&reader.file, index.offset, &mut byte).expect("the first part length");
14431        let mut file = OpenOptions::new().write(true).open(&path).expect("open index page");
14432        file.seek(SeekFrom::Start(index.offset)).expect("index start");
14433        file.write_all(&[!byte[0]]).expect("damage the first part length");
14434        let error = reader.read(1, &[0]).expect_err("a damaged index must not be used");
14435        assert!(error.message().contains("index page section checksum differs"), "{error}");
14436        fs::remove_file(path).expect("remove scratch file");
14437    }
14438
14439    /// Every integer width the format knows about, written and read back.
14440    ///
14441    /// The unsigned ones are the reason ClickBench can be stored at all: `hits` types `EventDate`
14442    /// as `USMALLINT`, and one unsupported column meant the whole table was refused. The extremes
14443    /// are in here on purpose, because a width that round trips through the wrong signedness only
14444    /// goes wrong at the end of its range.
14445    #[test]
14446    fn every_integer_width_round_trips_through_a_page() {
14447        let path = path("integer-widths");
14448        let columns = [
14449            (LogicalType::TinyInt, vec![Value::TinyInt(i8::MIN), Value::TinyInt(i8::MAX)]),
14450            (LogicalType::UTinyInt, vec![Value::UTinyInt(0), Value::UTinyInt(u8::MAX)]),
14451            (LogicalType::SmallInt, vec![Value::SmallInt(i16::MIN), Value::SmallInt(i16::MAX)]),
14452            (LogicalType::USmallInt, vec![Value::USmallInt(0), Value::USmallInt(u16::MAX)]),
14453            (LogicalType::Integer, vec![Value::Integer(i32::MIN), Value::Integer(i32::MAX)]),
14454            (LogicalType::UInteger, vec![Value::UInteger(0), Value::UInteger(u32::MAX)]),
14455            (LogicalType::BigInt, vec![Value::BigInt(i64::MIN), Value::BigInt(i64::MAX)]),
14456            (LogicalType::UBigInt, vec![Value::UBigInt(0), Value::UBigInt(u64::MAX)]),
14457        ];
14458        let fields = columns
14459            .iter()
14460            .enumerate()
14461            .map(|(at, (ty, _))| Field::required(format!("c{at}"), ty.clone()))
14462            .collect::<Vec<_>>();
14463        let vectors = columns
14464            .iter()
14465            .map(|(ty, values)| Vector::from_values(ty.clone(), values).expect("a vector"))
14466            .collect::<Vec<_>>();
14467        let mut writer = Writer::create(&path, "widths", fields).expect("new file");
14468        writer.append(&Chunk::new(vectors).expect("matching rows")).expect("one stripe");
14469        writer.finish().expect("commit");
14470
14471        let reader = Reader::open(&path).expect("reopen from disk");
14472        let wanted = (0..columns.len()).collect::<Vec<_>>();
14473        let read = reader.read(0, &wanted).expect("every column");
14474        assert_eq!(read.len(), 2);
14475        // row at a time: each column has its own type and its own pair of extremes.
14476        for (at, (ty, values)) in columns.iter().enumerate() {
14477            assert_eq!(read.value_at(0, at), values[0], "the low end of {ty}");
14478            assert_eq!(read.value_at(1, at), values[1], "the high end of {ty}");
14479        }
14480        fs::remove_file(path).expect("remove scratch file");
14481    }
14482
14483    /// The rest of the fixed width types, and the byte strings, written and read back.
14484    ///
14485    /// The extremes again, and for a float that means more than the ends of the range. Negative
14486    /// zero and a NaN are the two values that go through an encoder unnoticed and come back
14487    /// different, so they are here on purpose, and the NaN is compared by its bits rather than by
14488    /// `==`, which a NaN fails against itself.
14489    ///
14490    /// A blob is here beside them because it is the same round trip asked of bytes that are not
14491    /// text. The value in it is not UTF-8, so a path that reads a payload as a string on the way
14492    /// past turns this test red rather than turning a user's column into nulls.
14493    #[test]
14494    fn every_other_type_the_format_knows_round_trips_through_a_page() {
14495        let path = path("other-types");
14496        let columns = [
14497            (LogicalType::Float, vec![Value::Float(f32::MIN), Value::Float(-0.0)]),
14498            (LogicalType::Double, vec![Value::Double(f64::MIN), Value::Double(f64::MAX)]),
14499            (LogicalType::HugeInt, vec![Value::HugeInt(i128::MIN), Value::HugeInt(i128::MAX)]),
14500            (LogicalType::UHugeInt, vec![Value::UHugeInt(0), Value::UHugeInt(u128::MAX)]),
14501            (LogicalType::Time, vec![Value::Time(0), Value::Time(86_399_999_999)]),
14502            (LogicalType::TimeTz, vec![Value::TimeTz(-50_400_000_000), Value::TimeTz(0)]),
14503            (
14504                LogicalType::TimestampTz,
14505                vec![Value::TimestampTz(i64::MIN + 1), Value::TimestampTz(i64::MAX)],
14506            ),
14507            (
14508                LogicalType::Interval,
14509                vec![
14510                    Value::Interval { months: i32::MIN, days: i32::MAX, micros: i64::MIN },
14511                    Value::Interval { months: 13, days: -1, micros: 1 },
14512                ],
14513            ),
14514            (
14515                LogicalType::Blob,
14516                vec![Value::Blob(vec![0, 0xff, 0x80, 0xfe]), Value::Blob(Vec::new())],
14517            ),
14518        ];
14519        let fields = columns
14520            .iter()
14521            .enumerate()
14522            .map(|(at, (ty, _))| Field::required(format!("c{at}"), ty.clone()))
14523            .collect::<Vec<_>>();
14524        let vectors = columns
14525            .iter()
14526            .map(|(ty, values)| Vector::from_values(ty.clone(), values).expect("a vector"))
14527            .collect::<Vec<_>>();
14528        let mut writer = Writer::create(&path, "others", fields).expect("new file");
14529        writer.append(&Chunk::new(vectors).expect("matching rows")).expect("one stripe");
14530        writer.finish().expect("commit");
14531
14532        let reader = Reader::open(&path).expect("reopen from disk");
14533        let wanted = (0..columns.len()).collect::<Vec<_>>();
14534        let read = reader.read(0, &wanted).expect("every column");
14535        assert_eq!(read.len(), 2);
14536        for (at, (ty, values)) in columns.iter().enumerate() {
14537            assert_eq!(read.value_at(0, at), values[0], "the low end of {ty}");
14538            assert_eq!(read.value_at(1, at), values[1], "the high end of {ty}");
14539        }
14540        // A float keeps its sign through a zero, which `==` says nothing about because negative
14541        // zero and zero compare equal.
14542        let Value::Float(zero) = read.value_at(1, 0) else { panic!("a float stays a float") };
14543        assert!(zero.is_sign_negative(), "a negative zero came back as {zero}");
14544
14545        fs::remove_file(path).expect("remove scratch file");
14546    }
14547
14548    /// A NaN is still a NaN after a trip through a page.
14549    ///
14550    /// Apart from the other floats because it cannot be asserted the same way. A NaN is not equal
14551    /// to itself, so a comparison against the value that was written passes for every NaN and for
14552    /// nothing else, which is the one assertion that would not catch a page that lost it.
14553    #[test]
14554    fn a_nan_survives_being_written_down() {
14555        let path = path("nan");
14556        let nan = Vector::from_values(LogicalType::Double, &[Value::Double(f64::NAN)])
14557            .expect("a NaN vector");
14558        let mut writer =
14559            Writer::create(&path, "nan", vec![Field::required("d", LogicalType::Double)])
14560                .expect("new file");
14561        writer.append(&Chunk::new(vec![nan]).expect("one column")).expect("one stripe");
14562        writer.finish().expect("commit");
14563        let read = Reader::open(&path).expect("reopen").read(0, &[0]).expect("the column");
14564        let Value::Double(back) = read.value_at(0, 0) else { panic!("a double stays a double") };
14565        assert!(back.is_nan(), "a NaN came back as {back}");
14566        fs::remove_file(path).expect("remove scratch file");
14567    }
14568
14569    /// A uuid and a bit string, which have no `Value` arm of their own and are checked as bits.
14570    ///
14571    /// A uuid is the 128 bit lane and a bit string is bytes, and neither of them reads back as
14572    /// anything in `Value` today, so asking for a value here would compare two nulls and pass
14573    /// whatever the file held. The data underneath is what the storage promise is about, so that is
14574    /// what this reads.
14575    #[test]
14576    fn a_uuid_and_a_bit_string_come_back_as_the_bits_that_went_in() {
14577        let path = path("uuid-and-bit");
14578        let uuids = vec![0_i128, i128::MIN, -1];
14579        let mut bits = StringColumn::new();
14580        for value in [&b"\x02\xff"[..], &b""[..], &b"\x00\x01\x02\x03\x04\x05"[..]] {
14581            bits.push_bytes(value);
14582        }
14583        let expected = bits.clone();
14584        let fields =
14585            vec![Field::required("u", LogicalType::Uuid), Field::required("b", LogicalType::Bit)];
14586        let vectors = vec![
14587            Vector::flat(LogicalType::Uuid, Data::Int128(uuids.clone().into())).expect("uuids"),
14588            Vector::flat(LogicalType::Bit, Data::Varlen(bits)).expect("bit strings"),
14589        ];
14590        let mut writer = Writer::create(&path, "ids", fields).expect("new file");
14591        writer.append(&Chunk::new(vectors).expect("matching rows")).expect("one stripe");
14592        writer.finish().expect("commit");
14593
14594        let reader = Reader::open(&path).expect("reopen from disk");
14595        let read = reader.read(0, &[0, 1]).expect("both columns").flatten().expect("flat");
14596        let Some(Data::Int128(back)) = read.column(0).expect("the uuids").data() else {
14597            panic!("a uuid column is the 128 bit lane")
14598        };
14599        assert_eq!(back.as_slice(), uuids.as_slice());
14600        let Some(Data::Varlen(back)) = read.column(1).expect("the bits").data() else {
14601            panic!("a bit column is bytes")
14602        };
14603        for row in 0..expected.len() {
14604            assert_eq!(back.bytes(row), expected.bytes(row), "row {row} of the bit column");
14605        }
14606        fs::remove_file(path).expect("remove scratch file");
14607    }
14608
14609    /// Counting a run at once has to leave the candidate table exactly where counting its rows one
14610    /// at a time would, including once the table is full and a run is turned away row by row.
14611    #[test]
14612    fn a_run_counted_at_once_leaves_the_candidates_a_row_at_a_time_would() {
14613        let mut rows: Vec<Option<u64>> = Vec::new();
14614        let mut state = 0x2545_f491_4f6c_dd1d_u64;
14615        for index in 0..400_000_u64 {
14616            state ^= state << 13;
14617            state ^= state >> 7;
14618            state ^= state << 17;
14619            let times = 1 + (state % 7) as usize;
14620            let bits = match state % 11 {
14621                0 => None,
14622                1..=3 => Some(state % 16),
14623                _ => Some(index.wrapping_mul(0x9e37_79b9_7f4a_7c15)),
14624            };
14625            rows.extend(std::iter::repeat_n(bits, times));
14626        }
14627        let mut by_row = Candidates::default();
14628        for &bits in &rows {
14629            by_row.add(bits, 1);
14630        }
14631        let mut by_run = Candidates::default();
14632        let mut run = Run::default();
14633        let mut runs = 0_usize;
14634        for &bits in &rows {
14635            if let Some((bits, times)) = run.push(bits) {
14636                by_run.add(bits, times);
14637                runs += 1;
14638            }
14639        }
14640        if let Some((bits, times)) = run.take() {
14641            by_run.add(bits, times);
14642        }
14643        assert!(runs < rows.len() / 2, "the rows came in runs");
14644        assert!(by_row.decrements > 0, "the table filled and turned values away");
14645        assert_eq!(sorted_candidates(&by_run), sorted_candidates(&by_row));
14646        assert_eq!(by_run.nulls, by_row.nulls);
14647        assert_eq!(by_run.decrements, by_row.decrements);
14648    }
14649
14650    fn sorted_candidates(candidates: &Candidates) -> Vec<(u64, u32)> {
14651        let mut pairs = candidates.pairs().collect::<Vec<_>>();
14652        pairs.sort_unstable();
14653        assert_eq!(pairs.len(), candidates.held, "the count of held slots drifted");
14654        pairs
14655    }
14656
14657    /// The Misra-Gries table as it was written over a `HashMap`, kept as the oracle the open
14658    /// addressed one has to agree with.
14659    #[derive(Default)]
14660    struct MapCandidates {
14661        counts: HashMap<u64, u32>,
14662        nulls: u32,
14663        decrements: u64,
14664    }
14665
14666    impl MapCandidates {
14667        fn add(&mut self, bits: Option<u64>, mut times: u32) {
14668            while times > 0 {
14669                let held = match bits {
14670                    Some(bits) => self.counts.get_mut(&bits),
14671                    None if self.nulls != 0 => Some(&mut self.nulls),
14672                    None => None,
14673                };
14674                if let Some(count) = held {
14675                    *count = count.saturating_add(times);
14676                    return;
14677                }
14678                if self.counts.len() + usize::from(self.nulls != 0) < FREQUENCY_CANDIDATES {
14679                    match bits {
14680                        Some(bits) => {
14681                            self.counts.insert(bits, times);
14682                        }
14683                        None => self.nulls = times,
14684                    }
14685                    return;
14686                }
14687                self.counts.retain(|_, count| {
14688                    *count -= 1;
14689                    *count != 0
14690                });
14691                self.nulls = self.nulls.saturating_sub(1);
14692                self.decrements = self.decrements.saturating_add(1);
14693                times -= 1;
14694            }
14695        }
14696    }
14697
14698    /// Near unique values, a few heavy ones, nulls, and runs, through enough rows that the table
14699    /// fills, grows through every size and is decremented many times over. Both tables have to hold
14700    /// the same candidates with the same counts at the end, and at points along the way.
14701    #[test]
14702    fn the_open_addressed_candidates_agree_with_the_map_they_replaced() {
14703        for seed in [0x2545_f491_4f6c_dd1d_u64, 0x9e37_79b9_7f4a_7c15, 7] {
14704            let mut table = Candidates::default();
14705            let mut oracle = MapCandidates::default();
14706            let mut state = seed;
14707            for index in 0..300_000_u64 {
14708                state ^= state << 13;
14709                state ^= state >> 7;
14710                state ^= state << 17;
14711                let bits = match state % 13 {
14712                    0 => None,
14713                    1..=4 => Some(state % 40),
14714                    5 => Some((index % 1000) * 1_000_000),
14715                    _ => Some(state),
14716                };
14717                let times = 1 + (state >> 60) as u32 % 3;
14718                table.add(bits, times);
14719                oracle.add(bits, times);
14720                if index % 50_000 == 0 {
14721                    let mut expected =
14722                        oracle.counts.iter().map(|(&b, &c)| (b, c)).collect::<Vec<_>>();
14723                    expected.sort_unstable();
14724                    assert_eq!(sorted_candidates(&table), expected, "seed {seed} row {index}");
14725                }
14726            }
14727            let mut expected = oracle.counts.iter().map(|(&b, &c)| (b, c)).collect::<Vec<_>>();
14728            expected.sort_unstable();
14729            assert_eq!(sorted_candidates(&table), expected, "seed {seed}");
14730            assert_eq!(table.nulls, oracle.nulls, "seed {seed}");
14731            assert_eq!(table.decrements, oracle.decrements, "seed {seed}");
14732            assert!(table.decrements > 0, "seed {seed} never filled the table");
14733            for &(bits, _) in &expected {
14734                assert!(table.position(bits).is_some(), "seed {seed} lost {bits}");
14735            }
14736        }
14737    }
14738
14739    #[test]
14740    fn numeric_frequency_candidates_keep_bounded_row_ordinals() {
14741        let path = path("frequency-ordinals");
14742        let mut writer =
14743            Writer::create(&path, "items", vec![Field::required("id", LogicalType::BigInt)])
14744                .expect("new file");
14745        let mut values = Vec::new();
14746        for leader in 0..10_i64 {
14747            values.extend(std::iter::repeat_n(leader, 100));
14748        }
14749        values.extend(1_000_i64..41_000);
14750        for part in values.chunks(1_024) {
14751            let vector = Vector::flat(LogicalType::BigInt, Data::Int64(part.to_vec().into()))
14752                .expect("big integers");
14753            writer.append(&Chunk::new(vec![vector]).expect("one column")).expect("one stripe");
14754        }
14755        writer.finish().expect("commit");
14756
14757        let reader = Reader::open(&path).expect("reopen from disk");
14758        let occurrences =
14759            reader.frequency_occurrences(0).expect("valid metadata").expect("bounded ordinals");
14760        assert!(occurrences.omitted_max < 100);
14761        assert!(occurrences.ordinals.len() <= FREQUENCY_ORDINALS);
14762        assert_eq!(occurrences.anchor_indices.len(), occurrences.ordinals.len());
14763        assert!(occurrences.ordinals.windows(2).all(|pair| pair[0] < pair[1]));
14764        assert_eq!(&occurrences.ordinals[..1_000], &(0_u64..1_000).collect::<Vec<_>>());
14765        assert_eq!(
14766            &occurrences.anchor_indices[..1_000]
14767                .iter()
14768                .map(|&entry| occurrences.anchors[entry as usize].clone())
14769                .collect::<Vec<_>>(),
14770            &(0_i64..10)
14771                .flat_map(|leader| std::iter::repeat_n(Value::BigInt(leader), 100))
14772                .collect::<Vec<_>>()
14773        );
14774        fs::remove_file(path).expect("remove scratch file");
14775    }
14776
14777    #[test]
14778    fn numeric_frequencies_count_nulls_and_values_past_the_top_of_bigint() {
14779        // Ten leaders, then more unique values than the candidate table holds, so the first pass
14780        // has to decrement and the counts come from the recount. The unsigned leaders sit above
14781        // `i64::MAX`, where reading the bits as signed would give a different value, and the signed
14782        // ones are negative, where reading them as unsigned would.
14783        let path = path("frequency-bits");
14784        let mut writer = Writer::create(
14785            &path,
14786            "items",
14787            vec![Field::new("u", LogicalType::UBigInt), Field::new("s", LogicalType::BigInt)],
14788        )
14789        .expect("new file");
14790        let mut rows = Vec::new();
14791        let mut leaders = Vec::new();
14792        for leader in 0..10_u64 {
14793            let count = 300 - leader * 10;
14794            let (unsigned, signed) = if leader == 0 {
14795                (Value::Null, Value::Null)
14796            } else {
14797                (Value::UBigInt(u64::MAX - leader), Value::BigInt(-(leader as i64)))
14798            };
14799            rows.extend(std::iter::repeat_n((unsigned.clone(), signed.clone()), count as usize));
14800            leaders.push(((unsigned, count), (signed, count)));
14801        }
14802        rows.extend((1_000..41_000_u64).map(|id| (Value::UBigInt(id), Value::BigInt(id as i64))));
14803        for part in rows.chunks(1_024) {
14804            let unsigned = part.iter().map(|(value, _)| value.clone()).collect::<Vec<_>>();
14805            let signed = part.iter().map(|(_, value)| value.clone()).collect::<Vec<_>>();
14806            let chunk = Chunk::new(vec![
14807                Vector::from_values(LogicalType::UBigInt, &unsigned).expect("unsigned"),
14808                Vector::from_values(LogicalType::BigInt, &signed).expect("signed"),
14809            ])
14810            .expect("matching columns");
14811            writer.append(&chunk).expect("rows");
14812        }
14813        writer.finish().expect("commit");
14814
14815        let reader = Reader::open(&path).expect("reopen from disk");
14816        for column in 0..2 {
14817            let prefix =
14818                reader.frequency_prefix(column).expect("valid metadata").expect("a synopsis");
14819            let wanted = leaders
14820                .iter()
14821                .map(|(unsigned, signed)| if column == 0 { unsigned } else { signed })
14822                .cloned()
14823                .collect::<Vec<_>>();
14824            assert_eq!(&prefix.entries[..10], &wanted[..], "column {column}");
14825            assert!(prefix.omitted_max < 210, "column {column}");
14826            assert_eq!(
14827                reader.distinct_values(column).expect("valid metadata"),
14828                Some(9 + 40_000),
14829                "column {column}"
14830            );
14831        }
14832        fs::remove_file(path).expect("remove scratch file");
14833    }
14834
14835    #[test]
14836    fn a_narrow_column_takes_its_frequencies_from_the_tally_and_they_match_the_rows() {
14837        // Every column here has fewer distinct values than the tally holds, so the close takes its
14838        // counts from the gather rather than reading the pages back. The types are the ones whose
14839        // bits could come out wrong on that road: a negative tiny integer that has to be sign
14840        // extended, an unsigned one past the top of `INTEGER`, a date and a timestamp. A null every
14841        // thirteenth row checks that the nulls come from the pass and not from the list.
14842        let path = path("frequency-tally");
14843        let types = [
14844            LogicalType::TinyInt,
14845            LogicalType::UInteger,
14846            LogicalType::Date,
14847            LogicalType::Timestamp,
14848        ];
14849        let value = |ty: &LogicalType, at: i64| match ty {
14850            LogicalType::TinyInt => Value::TinyInt((at % 250 - 125) as i8),
14851            LogicalType::UInteger => Value::UInteger(u32::MAX - at as u32),
14852            LogicalType::Date => Value::Date(19_000 - at as i32),
14853            _ => Value::Timestamp(1_700_000_000_000_000 - at * 1_000_003),
14854        };
14855        let fields = types
14856            .iter()
14857            .enumerate()
14858            .map(|(at, ty)| Field::new(format!("c{at}"), ty.clone()))
14859            .collect::<Vec<_>>();
14860        let mut writer = Writer::create(&path, "items", fields).expect("new file");
14861        let mut rows = Vec::new();
14862        for at in 0..250_i64 {
14863            for _ in 0..=(at % 37) {
14864                rows.push(if rows.len() % 13 == 0 { None } else { Some(at) });
14865            }
14866        }
14867        for part in rows.chunks(1_000) {
14868            let columns = types
14869                .iter()
14870                .map(|ty| {
14871                    let values = part
14872                        .iter()
14873                        .map(|row| row.map_or(Value::Null, |at| value(ty, at)))
14874                        .collect::<Vec<_>>();
14875                    Vector::from_values(ty.clone(), &values).expect("a column")
14876                })
14877                .collect();
14878            writer.append(&Chunk::new(columns).expect("matching columns")).expect("rows");
14879        }
14880        writer.finish().expect("commit");
14881
14882        let reader = Reader::open(&path).expect("reopen from disk");
14883        for (column, ty) in types.iter().enumerate() {
14884            let mut counts = HashMap::<Option<i64>, u64>::new();
14885            for row in &rows {
14886                *counts.entry(*row).or_default() += 1;
14887            }
14888            let wanted = counts
14889                .into_iter()
14890                .map(|(row, count)| (row.map_or(Value::Null, |at| value(ty, at)), count))
14891                .collect::<Vec<_>>();
14892            let prefix =
14893                reader.frequency_prefix(column).expect("valid metadata").expect("a synopsis");
14894            assert_eq!(prefix.entries.len(), 2, "column {column}");
14895            assert!(prefix.omitted_max > 0, "column {column}");
14896            for (value, count) in &prefix.entries {
14897                let held =
14898                    wanted.iter().find(|(wanted, _)| wanted == value).map(|(_, count)| count);
14899                assert_eq!(held, Some(count), "column {column} value {value:?}");
14900            }
14901            assert!(prefix.entries.windows(2).all(|pair| pair[0].1 >= pair[1].1));
14902            assert_eq!(
14903                reader.distinct_values(column).expect("valid metadata"),
14904                Some(wanted.len() as u64 - 1),
14905                "column {column}"
14906            );
14907        }
14908        fs::remove_file(path).expect("remove scratch file");
14909    }
14910
14911    #[test]
14912    fn distinct_counts_are_exact_either_side_of_a_full_candidate_table() {
14913        // The count comes from the candidate table while it has room and from the set once it
14914        // fills, so the sizes around the fill, with and without a null taking a place, are where a
14915        // value could be counted twice or missed. Zero is in every column because the set keeps it
14916        // apart from the other values, and every value comes back later to be counted again.
14917        let edge = FREQUENCY_CANDIDATES as i64;
14918        for distinct in [0, 1, 7, edge - 2, edge - 1, edge, edge + 1, edge + 2, 3 * edge] {
14919            for with_null in [false, true] {
14920                let path = path("distinct-edge");
14921                let mut writer =
14922                    Writer::create(&path, "items", vec![Field::new("id", LogicalType::BigInt)])
14923                        .expect("new file");
14924                let mut values = Vec::new();
14925                for round in 0..2 {
14926                    for value in 0..distinct {
14927                        let repeat = if round == 0 { 1 + (value % 3) as usize } else { 1 };
14928                        values.extend(std::iter::repeat_n(
14929                            Value::BigInt(value * 7_919 % distinct),
14930                            repeat,
14931                        ));
14932                        if with_null && value % 1_000 == 0 {
14933                            values.push(Value::Null);
14934                        }
14935                    }
14936                }
14937                if with_null {
14938                    values.push(Value::Null);
14939                }
14940                for part in values.chunks(1_024) {
14941                    let chunk = Chunk::new(vec![
14942                        Vector::from_values(LogicalType::BigInt, part).expect("ids"),
14943                    ])
14944                    .expect("one column");
14945                    writer.append(&chunk).expect("rows");
14946                }
14947                writer.finish().expect("commit");
14948                let reader = Reader::open(&path).expect("reopen from disk");
14949                assert_eq!(
14950                    reader.distinct_values(0).expect("valid metadata"),
14951                    Some(distinct as u64),
14952                    "{distinct} values, null {with_null}"
14953                );
14954                fs::remove_file(path).expect("remove scratch file");
14955            }
14956        }
14957    }
14958
14959    #[test]
14960    fn narrow_nonzero_count_matches_the_full_reader_across_stripes() {
14961        let path = path("quick-nonzero");
14962        let mut writer = Writer::create(
14963            &path,
14964            "items",
14965            vec![Field::new("label", LogicalType::Varchar), Field::new("id", LogicalType::Integer)],
14966        )
14967        .expect("create");
14968        for ids in [
14969            &[Value::Integer(0), Value::Null, Value::Integer(3)][..],
14970            &[Value::Integer(0), Value::Integer(7), Value::Null][..],
14971        ] {
14972            let labels = vec![Value::Varchar("same".into()); ids.len()];
14973            writer
14974                .append(
14975                    &Chunk::new(vec![
14976                        Vector::from_values(LogicalType::Varchar, &labels).expect("labels"),
14977                        Vector::from_values(LogicalType::Integer, ids).expect("ids"),
14978                    ])
14979                    .expect("chunk"),
14980                )
14981                .expect("append");
14982        }
14983        writer.finish().expect("finish");
14984        let catalog = Catalog::open(&path).expect("catalog");
14985        assert_eq!(catalog.entries[0].nonzero, vec![None, None]);
14986        assert_eq!(catalog.entries[0].aggregates, vec![None, Some((10, 4))]);
14987        assert_eq!(catalog.entries[0].distincts, vec![Some(1), Some(3)]);
14988        assert_eq!(catalog.exact_numeric_frequencies("items", 1).expect("frequencies"), None);
14989        let prefix = catalog
14990            .table("items")
14991            .expect("reader")
14992            .frequency_prefix(1)
14993            .expect("valid metadata")
14994            .expect("partial frequencies");
14995        assert_eq!(prefix.entries, vec![(Value::Null, 2), (Value::Integer(0), 2)]);
14996        assert_eq!(prefix.omitted_max, 1);
14997        assert_eq!(catalog.distinct_count("items", 1).expect("distinct count"), Some(3));
14998        assert_eq!(
14999            catalog.integer_extremes("items", 1).expect("extremes"),
15000            Some(IntegerExtremes::Values { low: 0, high: 7 })
15001        );
15002        assert_eq!(
15003            catalog.aggregate_sums("items", &[1]).expect("catalog sums"),
15004            Some(CertifiedSums { columns: vec![(10, 4)], rows: 6 })
15005        );
15006        assert_eq!(catalog.nonzero_count("items", 1).expect("quick count"), Some(2));
15007        let mut legacy = catalog.clone();
15008        Arc::make_mut(&mut legacy.entries)[0].nonzero[1] = Some(999);
15009        assert_eq!(legacy.nonzero_count("items", 1).expect("ignore legacy count"), Some(2));
15010        Arc::make_mut(&mut legacy.entries)[0].frequencies[1] = None;
15011        assert_eq!(legacy.nonzero_count("items", 1).expect("directory fallback"), Some(2));
15012        Writer::certify_counts(&path).expect("recertify");
15013        assert_eq!(
15014            Catalog::open(&path).expect("reopen").nonzero_count("items", 1).expect("count"),
15015            Some(2)
15016        );
15017        assert_eq!(
15018            Catalog::open(&path).expect("reopen").aggregate_sums("items", &[1]).expect("sums"),
15019            Some(CertifiedSums { columns: vec![(10, 4)], rows: 6 })
15020        );
15021        assert_eq!(
15022            Catalog::open(&path).expect("reopen").distinct_count("items", 1).expect("distinct"),
15023            Some(3)
15024        );
15025        assert_eq!(
15026            Catalog::open(&path).expect("reopen").integer_extremes("items", 1).expect("ends"),
15027            Some(IntegerExtremes::Values { low: 0, high: 7 })
15028        );
15029        assert_eq!(
15030            Catalog::open(&path)
15031                .expect("reopen")
15032                .exact_numeric_frequencies("items", 1)
15033                .expect("frequencies"),
15034            None
15035        );
15036        assert_eq!(catalog.table("items").expect("reader").null_count(1).expect("nulls"), 2);
15037        fs::remove_file(path).expect("remove scratch file");
15038    }
15039
15040    #[test]
15041    fn numeric_string_pair_leaders_are_certified_in_the_directory() {
15042        let path = path("pair-frequencies");
15043        let mut pairs = Vec::new();
15044        pairs.extend(std::iter::repeat_n((1_i64, "alpha".to_string()), 100));
15045        pairs.extend(std::iter::repeat_n((1_i64, "beta".to_string()), 50));
15046        pairs.extend(std::iter::repeat_n((2_i64, "gamma".to_string()), 40));
15047        pairs.extend((1_000_i64..1_600).map(|id| (id, format!("tail {id}"))));
15048        let mut writer = Writer::create(
15049            &path,
15050            "items",
15051            vec![
15052                Field::required("id", LogicalType::BigInt),
15053                Field::required("phrase", LogicalType::Varchar),
15054            ],
15055        )
15056        .expect("new file");
15057        for part in pairs.chunks(1_024) {
15058            let ids = part.iter().map(|(id, _)| Value::BigInt(*id)).collect::<Vec<_>>();
15059            let phrases =
15060                part.iter().map(|(_, phrase)| Value::Varchar(phrase.clone())).collect::<Vec<_>>();
15061            writer
15062                .append(
15063                    &Chunk::new(vec![
15064                        Vector::from_values(LogicalType::BigInt, &ids).expect("ids"),
15065                        Vector::from_values(LogicalType::Varchar, &phrases).expect("phrases"),
15066                    ])
15067                    .expect("matching columns"),
15068                )
15069                .expect("rows");
15070        }
15071        writer.finish().expect("commit");
15072
15073        let reader = Reader::open(&path).expect("reopen from disk");
15074        assert!(
15075            reader.table.pair_frequencies.is_empty(),
15076            "no query-specific pair result is stored"
15077        );
15078        fs::remove_file(path).expect("remove scratch file");
15079    }
15080
15081    #[test]
15082    fn legacy_group_answers_are_ignored() {
15083        let path = path("legacy-group-answers");
15084        let mut writer = Writer::create(
15085            &path,
15086            "items",
15087            vec![
15088                Field::required("id", LogicalType::BigInt),
15089                Field::required("text", LogicalType::Varchar),
15090            ],
15091        )
15092        .expect("new file");
15093        writer
15094            .append(
15095                &Chunk::new(vec![
15096                    Vector::from_values(LogicalType::BigInt, &[Value::BigInt(1)]).expect("id"),
15097                    Vector::from_values(LogicalType::Varchar, &[Value::Varchar("x".into())])
15098                        .expect("text"),
15099                ])
15100                .expect("row"),
15101            )
15102            .expect("append");
15103        writer.finish().expect("commit");
15104        let mut reader = Reader::open(&path).expect("reopen");
15105        let table = Arc::make_mut(&mut reader.table);
15106        table.pair_frequencies.push(PairFrequencySummary {
15107            first: 0,
15108            second: 1,
15109            entries: vec![PairFrequencyEntry { first_entry: 0, second: Some(0), count: 999 }],
15110            omitted_max: 0,
15111        });
15112        table.host_groups = Some(host::HostSummary {
15113            column: 1,
15114            omitted_max: 0,
15115            entries: vec![host::HostEntry {
15116                host: "fake.test".into(),
15117                count: 999,
15118                bytes_sum: 999,
15119                minimum: "x".into(),
15120            }],
15121        });
15122        assert_eq!(reader.top_pair_frequencies(0, 1, 1).expect("legacy pair"), None);
15123        assert_eq!(reader.host_groups(1, 1).expect("legacy host"), None);
15124        fs::remove_file(path).expect("remove scratch file");
15125    }
15126
15127    /// The bug this is here for cost a 43 GB ClickBench table and an hour of reloading it. The
15128    /// format went from 11 to 12, every binary built after that said "magic or major version is
15129    /// unsupported" about the file, and there was no way to tell from the message whether the path
15130    /// was wrong, the file was truncated, or it was ours and simply older. The number this build
15131    /// wants is the whole answer and it was the one thing the message did not carry.
15132    #[test]
15133    fn a_file_from_another_format_says_which_format_it_is() {
15134        let older = path("older-format");
15135        let mut writer =
15136            Writer::create(&older, "items", vec![Field::new("id", LogicalType::Integer)])
15137                .expect("new file");
15138        let chunk = Chunk::new(vec![
15139            Vector::flat(LogicalType::Integer, Data::Int32(vec![1, 2, 3].into()))
15140                .expect("integers"),
15141        ])
15142        .expect("chunk");
15143        writer.append(&chunk).expect("page written");
15144        writer.finish().expect("commit");
15145
15146        // A format below the whole readable set, rather than `FORMAT - 1`, because the set has
15147        // more than one member now: format 22 is deliberately still readable, so the version that
15148        // has to be refused is the one under the oldest one accepted.
15149        let unreadable =
15150            READABLE.iter().copied().min().expect("at least one format is readable") - 1;
15151        let mut file = OpenOptions::new().write(true).open(&older).expect("open for the header");
15152        file.seek(SeekFrom::Start(8)).expect("the version follows the magic");
15153        file.write_all(&unreadable.to_le_bytes()).expect("write an older version");
15154        drop(file);
15155        let complaint = Reader::open(&older).expect_err("an older format is refused").to_string();
15156        assert!(complaint.contains(&format!("format {unreadable}")), "{complaint}");
15157        assert!(complaint.contains(&format!("format {FORMAT}")), "{complaint}");
15158
15159        let mut file = OpenOptions::new().write(true).open(&older).expect("open for the header");
15160        file.seek(SeekFrom::Start(0)).expect("the magic is first");
15161        file.write_all(b"NOTRUDB!").expect("write another engine's magic");
15162        drop(file);
15163        let complaint = Reader::open(&older).expect_err("a foreign file is refused").to_string();
15164        assert!(complaint.contains("magic"), "{complaint}");
15165        assert!(!complaint.contains("format"), "a version has nothing to do with it: {complaint}");
15166        fs::remove_file(older).expect("remove scratch file");
15167    }
15168
15169    #[test]
15170    fn an_unfinished_or_damaged_file_does_not_answer_with_partial_rows() {
15171        let unfinished = path("unfinished");
15172        let mut writer =
15173            Writer::create(&unfinished, "items", vec![Field::new("id", LogicalType::Integer)])
15174                .expect("new file");
15175        let chunk = Chunk::new(vec![
15176            Vector::flat(LogicalType::Integer, Data::Int32(vec![1, 2, 3].into()))
15177                .expect("integers"),
15178        ])
15179        .expect("chunk");
15180        writer.append(&chunk).expect("page written");
15181        drop(writer);
15182        assert!(Reader::open(&unfinished).is_err(), "no directory was committed");
15183        fs::remove_file(unfinished).expect("remove scratch file");
15184
15185        let damaged = path("damaged");
15186        let mut writer =
15187            Writer::create(&damaged, "items", vec![Field::new("id", LogicalType::Integer)])
15188                .expect("new file");
15189        writer.append(&chunk).expect("page written");
15190        writer.finish().expect("commit");
15191        let reader = Reader::open(&damaged).expect("valid directory");
15192        let mut file =
15193            OpenOptions::new().write(true).open(&damaged).expect("open for a damaged page");
15194        file.seek(SeekFrom::Start(HEADER + 1)).expect("inside first page");
15195        file.write_all(&[255]).expect("damage one byte");
15196        assert!(reader.read(0, &[0]).is_err(), "page checksum rejects corruption");
15197        fs::remove_file(damaged).expect("remove scratch file");
15198    }
15199
15200    #[test]
15201    fn damaged_lazy_dictionary_payload_is_an_error() {
15202        let path = path("damaged-dictionary");
15203        let mut writer = Writer::create(
15204            &path,
15205            "items",
15206            vec![
15207                Field::required("id", LogicalType::Integer),
15208                Field::new("text", LogicalType::Varchar),
15209            ],
15210        )
15211        .expect("new file");
15212        writer.append(&sample()).expect("stripe written");
15213        writer.finish().expect("commit");
15214
15215        let reader = Reader::open(&path).expect("valid directory");
15216        let dictionary = reader.table.dictionaries[1].expect("string dictionary page");
15217        // Read the count out of the page rather than writing it here, so that adding something
15218        // else to the index does not silently turn this into a test that damages the index.
15219        let mut header = [0; DICTIONARY_HEADER];
15220        read_at(&reader.file, dictionary.offset, &mut header).expect("dictionary header");
15221        // The first block's start is the first word after the offsets, since the blocks are written
15222        // during the load and are wherever the writer was when each was encoded.
15223        let count = u32::from_le_bytes(header[0..4].try_into().expect("four bytes")) as usize;
15224        let width = u32::from_le_bytes(header[12..16].try_into().expect("four bytes"));
15225        assert_ne!(width & DICTIONARY_SCATTERED, 0, "the blocks say where they are");
15226        let bits = (width & !DICTIONARY_FLAGS) as usize;
15227        let mut start = [0; 8];
15228        let at = dictionary.offset + (DICTIONARY_HEADER + offset_bytes(count, bits)) as u64;
15229        read_at(&reader.file, at, &mut start).expect("the first block's start");
15230        let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
15231        file.seek(SeekFrom::Start(u64::from_le_bytes(start))).expect("inside dictionary payload");
15232        file.write_all(&[255]).expect("damage dictionary payload");
15233
15234        let chunk = reader.read(0, &[1]).expect("code page and dictionary index remain valid");
15235        let error =
15236            chunk.validate_external().expect_err("payload corruption must reach the caller");
15237        assert!(error.message().contains("payload checksum differs"), "{error}");
15238        fs::remove_file(path).expect("remove scratch file");
15239    }
15240
15241    /// A column whose values are all different is written without a dictionary, and one whose
15242    /// values repeat keeps it.
15243    ///
15244    /// The two columns go in the same table and hold the same number of rows, so the only thing
15245    /// separating them is how much of the first stripe was a value it had not seen before. Both have
15246    /// to read back the values that were written, because the decision is about cost and nothing
15247    /// else. The file size is the other half of it: a column written without a dictionary goes
15248    /// through the string cascade instead, so dropping the dictionary must not turn into storing the
15249    /// column raw.
15250    #[test]
15251    fn a_column_of_all_different_values_is_written_without_a_dictionary() {
15252        let path = path("dictionary-decide");
15253        let rows = 20_000;
15254        // Long enough that storing it raw would show, and different in every row.
15255        let unique =
15256            |row: usize| format!("{row:09} a value that appears exactly once in the table");
15257        // The same values in the same shape, each one used forty times over.
15258        let repeated = |row: usize| unique(row / 40);
15259        let mut writer = Writer::create(
15260            &path,
15261            "items",
15262            vec![
15263                Field::required("unique", LogicalType::Varchar),
15264                Field::required("repeated", LogicalType::Varchar),
15265            ],
15266        )
15267        .expect("new file");
15268        for part in (0..rows).step_by(1_000) {
15269            let span = part..(part + 1_000).min(rows);
15270            let left = span.clone().map(|row| Value::Varchar(unique(row))).collect::<Vec<_>>();
15271            let right = span.map(|row| Value::Varchar(repeated(row))).collect::<Vec<_>>();
15272            writer
15273                .append(
15274                    &Chunk::new(vec![
15275                        Vector::from_values(LogicalType::Varchar, &left).expect("strings"),
15276                        Vector::from_values(LogicalType::Varchar, &right).expect("strings"),
15277                    ])
15278                    .expect("two columns"),
15279                )
15280                .expect("a part");
15281        }
15282        writer.finish().expect("commit");
15283
15284        let reader = Reader::open(&path).expect("reopen from disk");
15285        assert!(
15286            reader.table.dictionaries[0].is_none(),
15287            "a column with no repeats has nothing to say twice"
15288        );
15289        assert!(
15290            reader.table.dictionaries[1].is_some(),
15291            "a column whose values come round again keeps its dictionary"
15292        );
15293        let mut first = 0;
15294        for part in 0..reader.parts() {
15295            let chunk = reader.read(part, &[0, 1]).expect("a part");
15296            for row in 0..chunk.len() {
15297                assert_eq!(chunk.value_at(row, 0), Value::Varchar(unique(first + row)));
15298                assert_eq!(chunk.value_at(row, 1), Value::Varchar(repeated(first + row)));
15299            }
15300            first += chunk.len();
15301        }
15302        assert_eq!(first, rows, "every row was read back");
15303        let raw = (0..rows).map(|row| unique(row).len()).sum::<usize>();
15304        let size = fs::metadata(&path).expect("the file is there").len() as usize;
15305        assert!(size < raw, "a column without a dictionary is still encoded: {size} against {raw}");
15306        fs::remove_file(path).expect("remove scratch file");
15307    }
15308
15309    /// A payload of many blocks reads and checks every block of it.
15310    ///
15311    /// The test above has a dictionary of three values, which is one block, so it says nothing
15312    /// about a reader finding the right block among many. This one has thirty two thousand values,
15313    /// which is thirty two blocks, and it reads a value out of the first block and a value out of
15314    /// the last and then damages the last and asks for it again.
15315    ///
15316    /// Forty thousand rows over those thirty two thousand values, because a column the writer finds
15317    /// to be all distinct does not get a dictionary at all and there would be nothing here to test.
15318    /// Four rows in five holding a value the stripe has not seen before is a column that keeps one.
15319    /// The repeats are put at the front so that the values still arrive in order after them, which
15320    /// is what keeps the last part of the table on the last block of the payload.
15321    #[test]
15322    fn a_dictionary_over_many_blocks_checks_every_block_of_it() {
15323        let path = path("dictionary-blocks");
15324        let value = |row: usize| {
15325            let row = row.saturating_sub(8_000);
15326            format!("{row:07} a value long enough to be worth a payload block")
15327        };
15328        let parts = 40;
15329        let per_part = 1000;
15330        let mut writer =
15331            Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
15332                .expect("new file");
15333        for part in 0..parts {
15334            let values = (0..per_part)
15335                .map(|row| Value::Varchar(value(part * per_part + row)))
15336                .collect::<Vec<_>>();
15337            let chunk = Chunk::new(vec![
15338                Vector::from_values(LogicalType::Varchar, &values).expect("strings"),
15339            ])
15340            .expect("matching rows");
15341            writer.append(&chunk).expect("a part");
15342        }
15343        writer.finish().expect("commit");
15344
15345        let reader = Reader::open(&path).expect("reopen from disk");
15346        let dictionary = reader.table.dictionaries[0].expect("string dictionary page");
15347        assert!(
15348            parts * per_part > TEXT_PAYLOAD_VALUES * 4,
15349            "the dictionary has to be several blocks for this to be testing anything"
15350        );
15351        for part in [0, parts - 1] {
15352            let chunk = reader.read(part, &[0]).expect("a part");
15353            chunk.validate_external().expect("every payload block checks out");
15354            assert_eq!(chunk.value_at(0, 0), Value::Varchar(value(part * per_part)));
15355        }
15356
15357        // The last block is wherever the writer was when it was encoded, which the index says.
15358        let mut header = [0; DICTIONARY_HEADER];
15359        read_at(&reader.file, dictionary.offset, &mut header).expect("dictionary header");
15360        let count = u32::from_le_bytes(header[0..4].try_into().expect("four bytes")) as usize;
15361        let blocks = u32::from_le_bytes(header[8..12].try_into().expect("four bytes")) as usize;
15362        let width = u32::from_le_bytes(header[12..16].try_into().expect("four bytes"));
15363        let bits = (width & !DICTIONARY_FLAGS) as usize;
15364        let mut place = [0; 16];
15365        let at = DICTIONARY_HEADER + offset_bytes(count, bits) + (blocks - 1) * 16;
15366        read_at(&reader.file, dictionary.offset + at as u64, &mut place).expect("its place");
15367        let start = u64::from_le_bytes(place[..8].try_into().expect("eight bytes"));
15368        let length = u64::from_le_bytes(place[8..].try_into().expect("eight bytes"));
15369        let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
15370        file.seek(SeekFrom::Start(start + length - 4)).expect("the last bytes of the last block");
15371        file.write_all(&[255]).expect("damage the last payload block");
15372        let reader = Reader::open(&path).expect("the directory and the index are untouched");
15373        let chunk = reader.read(parts - 1, &[0]).expect("the code page remains valid");
15374        let error = chunk.validate_external().expect_err("the damage must reach the caller");
15375        assert!(error.message().contains("payload checksum differs"), "{error}");
15376        fs::remove_file(path).expect("remove scratch file");
15377    }
15378
15379    /// Values of different lengths read back where the offsets say they do.
15380    ///
15381    /// The offsets are packed at one width for the column, they are relative to the payload block a
15382    /// value lands in, and they go in runs of half a block, so there are two boundaries where the
15383    /// arithmetic could be off by one and neither shows up on values that are all the same length.
15384    /// This writes 5,000 values whose lengths cycle through a wide range and reads every one back,
15385    /// so the first value of a block, the last value of a run and the last value of a block are all
15386    /// covered several times over. An empty value is in the cycle because a zero length span is the
15387    /// case the reader short circuits.
15388    ///
15389    /// Six thousand rows over those 5,000 values, because a column the writer finds to be all
15390    /// distinct is written without a dictionary and then there are no packed offsets to be off by
15391    /// one in.
15392    #[test]
15393    fn values_of_different_lengths_read_back_out_of_packed_offsets() {
15394        let path = path("dictionary-offsets");
15395        let value = |row: usize| {
15396            let row = row % 5_000;
15397            if row % 511 == 3 { String::new() } else { "x".repeat(row % 97) + &format!("{row:05}") }
15398        };
15399        let rows = 6_000;
15400        let mut writer =
15401            Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
15402                .expect("new file");
15403        let values = (0..rows).map(|row| Value::Varchar(value(row))).collect::<Vec<_>>();
15404        for part in values.chunks(1_000) {
15405            let chunk =
15406                Chunk::new(vec![Vector::from_values(LogicalType::Varchar, part).expect("strings")])
15407                    .expect("matching rows");
15408            writer.append(&chunk).expect("a part");
15409        }
15410        writer.finish().expect("commit");
15411
15412        let reader = Reader::open(&path).expect("reopen from disk");
15413        assert!(
15414            rows > TEXT_PAYLOAD_VALUES * 4,
15415            "the dictionary has to be several blocks for this to be testing anything"
15416        );
15417        for part in 0..rows / 1_000 {
15418            let chunk = reader.read(part, &[0]).expect("a part");
15419            for row in 0..1_000 {
15420                let row = part * 1_000 + row;
15421                assert_eq!(
15422                    chunk.value_at(row % 1_000, 0),
15423                    Value::Varchar(value(row)),
15424                    "value {row}"
15425                );
15426            }
15427        }
15428        // The lengths a vector at a time, twice over, because the first pass is what makes the
15429        // table of ends worth building and the second is read out of the lengths worked out of it.
15430        for _ in 0..2 {
15431            for part in 0..rows / 1_000 {
15432                let chunk = reader.read(part, &[0]).expect("a part");
15433                let mut lens = vec![0_i64; 1_000];
15434                let column = chunk.column(0).expect("one column");
15435                assert!(column.try_bytes_lens(&mut lens).expect("lengths"), "a stored column");
15436                for (row, &len) in lens.iter().enumerate() {
15437                    let row = part * 1_000 + row;
15438                    assert_eq!(len as usize, value(row).len(), "the length of value {row}");
15439                }
15440            }
15441        }
15442        fs::remove_file(path).expect("remove scratch file");
15443    }
15444
15445    /// Lengths start again at every block, and ends that go backwards inside one give no table.
15446    #[test]
15447    fn lengths_restart_at_each_block_and_refuse_ends_that_go_backwards() {
15448        let mut ends: Vec<u32> = (1..=TEXT_PAYLOAD_VALUES as u32).map(|at| at * 2).collect();
15449        ends.extend([3, 3, 10]);
15450        let Some(Lengths::Narrow(lens)) = lengths_of(&ends) else { panic!("short ordered ends") };
15451        assert!(lens[..TEXT_PAYLOAD_VALUES].iter().all(|&len| len == 2));
15452        assert_eq!(&lens[TEXT_PAYLOAD_VALUES..], &[3, 0, 7]);
15453        // One value longer than sixteen bits keeps every length at four bytes.
15454        let long = [5, 70_005, 70_006];
15455        let Some(Lengths::Wide(lens)) = lengths_of(&long) else { panic!("long ordered ends") };
15456        assert_eq!(lens, [5, 70_000, 1]);
15457        let mut read = Vec::new();
15458        Lengths::Wide(lens).extend_at(&[1, 9, 0], &mut read);
15459        assert_eq!(read, [70_000, 0, 5], "a position past the end is no length");
15460        ends.push(9);
15461        assert!(lengths_of(&ends).is_none());
15462    }
15463
15464    /// Every worker of a scan wants the dictionary at the same moment and one of them fetches it.
15465    ///
15466    /// Asking a `OnceLock` whether it holds something answers the question a worker that already has
15467    /// the dictionary is asking and not the one a worker without it is asking, which is whether
15468    /// somebody is already on their way with it. Sixteen workers that all miss will all read the
15469    /// page, all verify it and all decode it, and fifteen will drop the result. Nothing about that
15470    /// is incorrect, which is why it went unnoticed, and it showed up as ClickBench 38 getting
15471    /// slower when the scan in front of it got faster and stopped staggering the arrivals.
15472    ///
15473    /// The barrier is what makes the test about that rather than about luck. Without it the first
15474    /// thread is usually finished before the last one starts and the count is one either way.
15475    #[test]
15476    fn a_global_dictionary_is_opened_once_however_many_workers_ask_at_once() {
15477        let path = path("dictionary-once");
15478        let parts = 8;
15479        let per_part = 500;
15480        let value =
15481            |row: usize| format!("{row:07} a value long enough to be worth a payload block");
15482        let mut writer =
15483            Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
15484                .expect("new file");
15485        for part in 0..parts {
15486            let values = (0..per_part)
15487                .map(|row| Value::Varchar(value(part * per_part + row)))
15488                .collect::<Vec<_>>();
15489            let chunk = Chunk::new(vec![
15490                Vector::from_values(LogicalType::Varchar, &values).expect("strings"),
15491            ])
15492            .expect("matching rows");
15493            writer.append(&chunk).expect("a part");
15494        }
15495        writer.finish().expect("commit");
15496
15497        let reader = Reader::open(&path).expect("reopen from disk");
15498        assert!(reader.table.dictionaries[0].is_some(), "the column has to have one to share");
15499        assert_eq!(reader.reads().dictionaries, 0, "opening the file does not open a dictionary");
15500
15501        let workers = 16;
15502        let gate = std::sync::Barrier::new(workers);
15503        std::thread::scope(|scope| {
15504            for worker in 0..workers {
15505                let reader = reader.clone();
15506                let gate = &gate;
15507                scope.spawn(move || {
15508                    gate.wait();
15509                    let chunk = reader.read(worker % parts, &[0]).expect("a part");
15510                    assert_eq!(
15511                        chunk.value_at(0, 0),
15512                        Value::Varchar(value((worker % parts) * per_part))
15513                    );
15514                });
15515            }
15516        });
15517
15518        assert_eq!(reader.reads().dictionaries, 1, "sixteen workers, one dictionary, one open");
15519        fs::remove_file(path).expect("remove scratch file");
15520    }
15521
15522    /// The sorted order sits outside the index the page checksum covers, because a query that
15523    /// never searches a dictionary should not read it, so it carries its own checksums and this is
15524    /// what says they are checked. A search that trusted a damaged order would give a wrong answer
15525    /// rather than a slow one.
15526    #[test]
15527    fn a_damaged_sorted_order_is_an_error() {
15528        let path = path("damaged-order");
15529        let mut writer = Writer::create(
15530            &path,
15531            "items",
15532            vec![
15533                Field::required("id", LogicalType::Integer),
15534                Field::new("text", LogicalType::Varchar),
15535            ],
15536        )
15537        .expect("new file");
15538        writer.append(&sample()).expect("stripe written");
15539        writer.finish().expect("commit");
15540
15541        let reader = Reader::open(&path).expect("valid directory");
15542        let page = reader.table.dictionaries[1].expect("string dictionary page");
15543        let mut header = [0; DICTIONARY_HEADER];
15544        read_at(&reader.file, page.offset, &mut header).expect("dictionary header");
15545        let index_len = dictionary_index_len(&header);
15546        let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
15547        file.seek(SeekFrom::Start(page.offset + index_len)).expect("the first head");
15548        file.write_all(&[255]).expect("damage the order");
15549
15550        let dictionary = reader.dictionary(1).expect("read").expect("a string column has one");
15551        let error = dictionary.compare_rank(0, b"anything").expect_err("a damaged order is caught");
15552        assert!(error.message().contains("rank checksum differs"), "{error}");
15553        fs::remove_file(path).expect("remove scratch file");
15554    }
15555
15556    /// Codes stay in first appearance order and the sorted order is written beside them, so a
15557    /// reader can put the values back in order without the writer having had to know them all
15558    /// before it handed out the first code.
15559    #[test]
15560    fn a_global_dictionary_carries_the_sorted_order_of_its_values() {
15561        // Chosen so the sort cannot be decided on the first eight bytes alone. Three values share
15562        // a nine byte prefix, one is a prefix of another, and one is empty.
15563        let spellings = ["overlong1z", "b", "", "overlong1a", "overlong", "ab", "a", "overlong1"];
15564        let path = path("dictionary-order");
15565        let mut writer =
15566            Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
15567                .expect("new file");
15568        writer
15569            .append(
15570                &Chunk::new(vec![
15571                    Vector::from_values(
15572                        LogicalType::Varchar,
15573                        &spellings.map(|text| Value::Varchar(text.into())),
15574                    )
15575                    .expect("strings"),
15576                ])
15577                .expect("one column"),
15578            )
15579            .expect("stripe written");
15580        writer.finish().expect("commit");
15581
15582        let reader = Reader::open(&path).expect("valid directory");
15583        let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
15584        let count = dictionary.ranks().expect("a v10 file stores one");
15585        assert_eq!(count, spellings.len(), "every distinct value has a rank");
15586        let order = (0..count)
15587            .map(|rank| dictionary.code_at_rank(rank).expect("a code"))
15588            .collect::<Vec<_>>();
15589        let mut seen = order.clone();
15590        seen.sort_unstable();
15591        assert_eq!(seen, (0..spellings.len() as u32).collect::<Vec<_>>(), "a permutation of codes");
15592
15593        let ranked = order
15594            .iter()
15595            .map(|&code| {
15596                dictionary.try_bytes_at(code as usize).expect("read").expect("a value").to_vec()
15597            })
15598            .collect::<Vec<_>>();
15599        let mut expected = spellings.map(|text| text.as_bytes().to_vec()).to_vec();
15600        expected.sort();
15601        assert_eq!(ranked, expected, "rank order is value order");
15602
15603        // What a search asks, on the values themselves rather than through a kernel, so that a
15604        // file whose heads disagree with its bytes is caught here rather than as a wrong answer.
15605        for (rank, value) in expected.iter().enumerate() {
15606            assert_eq!(
15607                dictionary.compare_rank(rank, value).expect("compare"),
15608                Ordering::Equal,
15609                "rank {rank} is its own value"
15610            );
15611            if rank > 0 {
15612                assert_eq!(
15613                    dictionary.compare_rank(rank - 1, value).expect("compare"),
15614                    Ordering::Less,
15615                    "rank {rank} follows the one before it"
15616                );
15617            }
15618        }
15619        fs::remove_file(path).expect("remove scratch file");
15620    }
15621
15622    /// Five text columns of different sizes close at the same time, and each comes back with its
15623    /// own values in its own order.
15624    ///
15625    /// The sizes differ so that the columns are taken in an order that is not the column order, and
15626    /// the values of each column are spelled with its number so that one column's page written in
15627    /// another's place would read back as the wrong strings rather than the right ones by chance.
15628    #[test]
15629    fn text_columns_closed_at_once_each_keep_their_own_dictionary() {
15630        let sizes = [300_usize, 5_000, 40, 2_000, 1_200];
15631        let path = path("dictionaries-at-once");
15632        let fields = (0..sizes.len())
15633            .map(|column| Field::new(format!("text{column}"), LogicalType::Varchar))
15634            .collect::<Vec<_>>();
15635        let mut writer = Writer::create(&path, "items", fields).expect("new file");
15636        let rows = 10_000_usize;
15637        for start in (0..rows).step_by(1_024) {
15638            let columns = sizes
15639                .iter()
15640                .enumerate()
15641                .map(|(column, &size)| {
15642                    let values = (start..(start + 1_024).min(rows))
15643                        .map(|row| Value::Varchar(format!("c{column}-{:05}", (row * 7919) % size)))
15644                        .collect::<Vec<_>>();
15645                    Vector::from_values(LogicalType::Varchar, &values).expect("strings")
15646                })
15647                .collect::<Vec<_>>();
15648            writer.append(&Chunk::new(columns).expect("five columns")).expect("stripe written");
15649        }
15650        writer.finish().expect("commit");
15651
15652        let reader = Reader::open(&path).expect("valid directory");
15653        for (column, &size) in sizes.iter().enumerate() {
15654            let dictionary =
15655                reader.dictionary(column).expect("read").expect("a string column has one");
15656            let count = dictionary.ranks().expect("a v10 file stores one");
15657            assert_eq!(count, size, "column {column} has its own distinct count");
15658            let ranked = (0..count)
15659                .map(|rank| {
15660                    let code = dictionary.code_at_rank(rank).expect("a code");
15661                    dictionary.try_bytes_at(code as usize).expect("read").expect("a value").to_vec()
15662                })
15663                .collect::<Vec<_>>();
15664            let expected = (0..size)
15665                .map(|value| format!("c{column}-{value:05}").into_bytes())
15666                .collect::<Vec<_>>();
15667            assert_eq!(ranked, expected, "column {column} ranks its own values in order");
15668        }
15669        fs::remove_file(path).expect("remove scratch file");
15670    }
15671
15672    /// A dictionary large enough to be decoded and sorted on several threads ranks the way one small
15673    /// enough for one thread does.
15674    ///
15675    /// Seventy thousand values over sixty nine blocks, in no order and each four times over so the
15676    /// column is worth a dictionary, written and ranked in the close.
15677    /// Some share a long prefix and some differ only in the last byte, so the buckets of the sort cut
15678    /// through runs of values that agree for a long way.
15679    #[test]
15680    fn a_large_dictionary_ranks_in_value_order() {
15681        let path = path("dictionary-large-rank");
15682        let value = |row: u64| {
15683            let mixed = row.wrapping_mul(0x9e37_79b9_7f4a_7c15) >> 40;
15684            match row % 3 {
15685                0 => format!("https://example.com/a/long/shared/path/{mixed:08}"),
15686                1 => format!("{mixed}"),
15687                _ => format!("x{}", row % 1000).repeat(1 + (row % 4) as usize) + &row.to_string(),
15688            }
15689        };
15690        let distinct = 70_000;
15691        let parts = 4 * distinct / 1000;
15692        let per_part = 1000;
15693        let mut writer =
15694            Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
15695                .expect("new file");
15696        for part in 0..parts {
15697            let values = (0..per_part)
15698                .map(|row| Value::Varchar(value((part * per_part + row) / 4)))
15699                .collect::<Vec<_>>();
15700            let chunk = Chunk::new(vec![
15701                Vector::from_values(LogicalType::Varchar, &values).expect("strings"),
15702            ])
15703            .expect("matching rows");
15704            writer.append(&chunk).expect("a part");
15705        }
15706        writer.finish().expect("commit");
15707
15708        let reader = Reader::open(&path).expect("reopen from disk");
15709        let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
15710        let count = dictionary.ranks().expect("a ranked dictionary");
15711        assert_eq!(count, distinct as usize, "every distinct value has a rank");
15712        assert!(count >= PARALLEL_SORT_MIN, "too few values to be sorted on more than one thread");
15713        let ranked = (0..count)
15714            .map(|rank| {
15715                let code = dictionary.code_at_rank(rank).expect("a code");
15716                dictionary.try_bytes_at(code as usize).expect("read").expect("a value").to_vec()
15717            })
15718            .collect::<Vec<_>>();
15719        let mut expected = (0..distinct).map(|row| value(row).into_bytes()).collect::<Vec<_>>();
15720        expected.sort();
15721        assert_eq!(ranked, expected, "rank order is value order");
15722        fs::remove_file(path).expect("remove scratch file");
15723    }
15724
15725    /// A string column's synopsis is turned into values without keeping the blocks it went through.
15726    ///
15727    /// Three thousand values, every fifth of them four times over, so the synopsis is a prefix of
15728    /// five hundred and twelve codes spread over all three payload blocks. Reading it used to leave
15729    /// all three decoded for as long as the reader lived. It leaves none of them now, and the second
15730    /// read answers out of what the first remembered.
15731    /// A directory read out of the file a window at a time is the directory read whole.
15732    ///
15733    /// The windows here are far smaller than any field is long, so every kind of field is split
15734    /// across a refill somewhere, and a bound is offered to its codec short more than once. The
15735    /// synopses are left in the file, and each one read back from where it was left is the one the
15736    /// whole read decoded.
15737    #[test]
15738    fn a_directory_read_a_window_at_a_time_is_the_directory_read_whole() {
15739        let path = path("windowed-directory");
15740        let fields = vec![
15741            Field::required("id", LogicalType::BigInt),
15742            Field::required("word", LogicalType::Varchar),
15743            Field::new("score", LogicalType::Double),
15744        ];
15745        let mut writer = Writer::create(&path, "items", fields).expect("new file");
15746        for part in 0..70_i64 {
15747            let ids = (0..100).map(|row| Value::BigInt(part * 100 + row % 7)).collect::<Vec<_>>();
15748            let words = (0..100)
15749                .map(|row| Value::Varchar(format!("word {}", row % 13)))
15750                .collect::<Vec<_>>();
15751            let scores = (0..100)
15752                .map(|row| if row % 4 == 0 { Value::Null } else { Value::Double(row as f64) })
15753                .collect::<Vec<_>>();
15754            let chunk = Chunk::new(vec![
15755                Vector::from_values(LogicalType::BigInt, &ids).expect("integers"),
15756                Vector::from_values(LogicalType::Varchar, &words).expect("strings"),
15757                Vector::from_values(LogicalType::Double, &scores).expect("doubles"),
15758            ])
15759            .expect("three columns");
15760            writer.append(&chunk).expect("a part");
15761        }
15762        writer.finish().expect("commit");
15763
15764        let catalog = Catalog::open(&path).expect("reopen");
15765        let entry = catalog.entries.first().expect("one table").directory;
15766        let (offset, length) = (entry.offset, entry.length as usize);
15767        let mut bytes = vec![0; length];
15768        read_at(&catalog.file, offset, &mut bytes).expect("the directory");
15769        assert_eq!(file_checksum(&catalog.file, offset, length).expect("checksum"), entry.hash);
15770        let whole = decode_directory(&bytes, catalog.size).expect("whole");
15771        assert!(whole.stripes.len() > 1, "the table should span stripes");
15772        for size in [1, 7, 33, 4_096] {
15773            let mut cursor = Cursor::over(&catalog.file, offset, length);
15774            cursor.window.as_mut().expect("a window").size = size;
15775            let windowed = read_directory(cursor, catalog.size, Some(offset)).expect("windowed");
15776            assert_eq!(format!("{:?}", windowed.stripes), format!("{:?}", whole.stripes));
15777            assert_eq!(format!("{:?}", windowed.fields), format!("{:?}", whole.fields));
15778            let mut stored = 0;
15779            for (column, (left, held)) in
15780                windowed.frequencies.iter().zip(&whole.frequencies).enumerate()
15781            {
15782                match (left, held) {
15783                    (None, None) => {}
15784                    (
15785                        Some(super::Frequencies::Stored { span, values }),
15786                        Some(super::Frequencies::Held(summary)),
15787                    ) => {
15788                        let mut one = vec![0; span.length as usize];
15789                        read_at(&catalog.file, span.offset, &mut one).expect("a synopsis");
15790                        let read = decode_summary(
15791                            &mut Cursor::new(&one),
15792                            &whole.fields[column],
15793                            whole.rows,
15794                            *values,
15795                        )
15796                        .expect("a valid synopsis")
15797                        .expect("one is there");
15798                        assert_eq!(format!("{read:?}"), format!("{summary:?}"));
15799                        stored += 1;
15800                    }
15801                    other => panic!("column {column} came back as {other:?}"),
15802                }
15803            }
15804            assert!(stored >= 2, "only {stored} synopses were left in the file");
15805        }
15806        let reader = catalog.table("items").expect("the table");
15807        assert!(reader.frequency_summaries[1].get().is_none());
15808        assert!(reader.top_frequencies(1, 1).expect("a readable synopsis").is_some());
15809        let first = reader.frequency_summaries[1].get().expect("decoded synopsis");
15810        let clone = reader.clone();
15811        assert!(clone.top_frequencies(1, 1).expect("cached synopsis").is_some());
15812        assert!(Arc::ptr_eq(first, clone.frequency_summaries[1].get().expect("same synopsis")));
15813        fs::remove_file(path).expect("remove scratch file");
15814    }
15815
15816    #[test]
15817    fn a_checksum_carried_across_reads_is_the_checksum_of_the_whole() {
15818        let path = path("file-checksum");
15819        let bytes = (0..200_000_u32)
15820            .map(|at| (at.wrapping_mul(2_654_435_761) >> 13) as u8)
15821            .collect::<Vec<_>>();
15822        fs::write(&path, &bytes).expect("scratch file");
15823        let file = File::open(&path).expect("open");
15824        for (offset, length) in [
15825            (0, 0),
15826            (3, 1),
15827            (5, 31),
15828            (0, 32),
15829            (9, 33),
15830            (1, 65_536),
15831            (7, 65_567),
15832            (0, 200_000),
15833            (11, 131_101),
15834        ] {
15835            let whole = checksum(&bytes[offset..offset + length]);
15836            assert_eq!(
15837                file_checksum(&file, offset as u64, length).expect("read"),
15838                whole,
15839                "{offset} {length}"
15840            );
15841        }
15842        fs::remove_file(path).expect("remove scratch file");
15843    }
15844
15845    #[test]
15846    fn a_string_synopsis_is_read_without_keeping_the_dictionary_blocks() {
15847        let path = path("synopsis-keeps-no-block");
15848        let spelled = |index: usize| Value::Varchar(format!("phrase {index:05}"));
15849        let mut values = (0..3_000).map(spelled).collect::<Vec<_>>();
15850        for _ in 0..3 {
15851            values.extend((0..3_000).step_by(5).map(spelled));
15852        }
15853        let mut writer =
15854            Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
15855                .expect("new file");
15856        for part in values.chunks(1_024) {
15857            writer
15858                .append(
15859                    &Chunk::new(vec![
15860                        Vector::from_values(LogicalType::Varchar, part).expect("strings"),
15861                    ])
15862                    .expect("one column"),
15863                )
15864                .expect("a part");
15865        }
15866        writer.finish().expect("commit");
15867
15868        let reader = Reader::open(&path).expect("reopen from disk");
15869        let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
15870        let resting = dictionary.footprint();
15871        let prefix = reader.frequency_prefix(0).expect("a readable synopsis").expect("one");
15872        assert_eq!(prefix.entries.len(), 512);
15873        for (value, count) in &prefix.entries {
15874            let Value::Varchar(text) = value else { panic!("a string column gave {value:?}") };
15875            let index = text["phrase ".len()..].parse::<usize>().expect("a spelled number");
15876            assert_eq!((index % 5, *count), (0, 4), "{text} came back with {count}");
15877        }
15878        assert_eq!(dictionary.footprint(), resting, "reading the synopsis kept a decoded block");
15879        let again = reader.frequency_prefix(0).expect("a readable synopsis").expect("one");
15880        assert_eq!(again.entries, prefix.entries);
15881        fs::remove_file(path).expect("remove scratch file");
15882    }
15883
15884    /// `length` over a stored column keeps a count a value rather than the blocks it counted.
15885    ///
15886    /// Reading the bytes a row at a time keeps every block it touches, so a scan of `length` over a
15887    /// whole column used to end up holding the column decoded. The counts are what is kept now, and
15888    /// they have to be the counts of characters rather than bytes, which is why the values here are
15889    /// not ASCII.
15890    #[test]
15891    fn character_lengths_are_counted_without_keeping_the_dictionary_blocks() {
15892        let path = path("character-lengths");
15893        let spellings = (0..2_500)
15894            .map(|index| Value::Varchar(format!("héllo {index:05} {}", "ü".repeat(index % 30))))
15895            .collect::<Vec<_>>();
15896        let mut writer =
15897            Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
15898                .expect("new file");
15899        for part in spellings.chunks(1_024) {
15900            writer
15901                .append(
15902                    &Chunk::new(vec![
15903                        Vector::from_values(LogicalType::Varchar, part).expect("strings"),
15904                    ])
15905                    .expect("one column"),
15906                )
15907                .expect("a part");
15908        }
15909        writer.finish().expect("commit");
15910
15911        let reader = Reader::open(&path).expect("reopen from disk");
15912        let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
15913        let resting = dictionary.footprint();
15914        let mut lens = Vec::new();
15915        assert!(dictionary.try_chars_lens(&mut lens).expect("counted"), "a stored source counts");
15916        let counted = dictionary.footprint() - resting;
15917        let blocks = dictionary.len().div_ceil(TEXT_PAYLOAD_VALUES);
15918        assert!(
15919            counted <= blocks * TEXT_PAYLOAD_VALUES * size_of::<u32>(),
15920            "counting kept {counted} bytes, more than a count a value"
15921        );
15922        let expected = (0..dictionary.len())
15923            .map(|code| {
15924                let bytes = dictionary.try_bytes_at(code).expect("read").expect("a value");
15925                i64::try_from(std::str::from_utf8(bytes).expect("utf-8").chars().count())
15926                    .expect("small")
15927            })
15928            .collect::<Vec<_>>();
15929        assert_eq!(lens, expected, "a count is the number of characters, not of bytes");
15930        let mut again = Vec::new();
15931        assert!(dictionary.try_chars_lens(&mut again).expect("counted"));
15932        assert_eq!(again, lens, "the kept counts answer the second time");
15933        fs::remove_file(path).expect("remove scratch file");
15934    }
15935
15936    /// Writes one column of strings whose code is where they sit in `spellings`, and reopens it.
15937    fn stored_spellings(label: &str, spellings: &[String]) -> (PathBuf, Reader) {
15938        let path = path(label);
15939        let values = spellings.iter().map(|text| Value::Varchar(text.clone())).collect::<Vec<_>>();
15940        let mut writer =
15941            Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
15942                .expect("new file");
15943        for part in values.chunks(1_024) {
15944            writer
15945                .append(
15946                    &Chunk::new(vec![
15947                        Vector::from_values(LogicalType::Varchar, part).expect("strings"),
15948                    ])
15949                    .expect("one column"),
15950                )
15951                .expect("a part");
15952        }
15953        writer.finish().expect("commit");
15954        let reader = Reader::open(&path).expect("reopen from disk");
15955        (path, reader)
15956    }
15957
15958    /// Codes that go all over a dictionary of `len` values, and every seventh row null.
15959    ///
15960    /// The shape of a vector a scan hands out: its codes are in row order, which lands them in
15961    /// every block of the dictionary in no order at all, so a read of the whole vector has to put
15962    /// them in block order itself to read each block once.
15963    fn scattered_rows(len: usize) -> (Vec<u32>, Vec<bool>) {
15964        let codes = (0..len)
15965            .map(|row| u32::try_from(row * 7_919 % len).expect("a small dictionary"))
15966            .collect::<Vec<_>>();
15967        let valid = (0..len).map(|row| row % 7 != 3).collect::<Vec<_>>();
15968        (codes, valid)
15969    }
15970
15971    /// `length` over a vector with nulls keeps the counts and not the blocks, the same as over one
15972    /// without.
15973    ///
15974    /// The whole vector count used to be taken only when no row was null, and every other vector
15975    /// went a row at a time through the bytes, which keeps every block it reads. A column with a
15976    /// null in each vector was held decoded after one `length` over it.
15977    #[test]
15978    fn character_lengths_with_nulls_are_counted_without_keeping_the_dictionary_blocks() {
15979        let spellings = (0..2_500)
15980            .map(|index| format!("héllo {index:05} {}", "ü".repeat(index % 30)))
15981            .collect::<Vec<_>>();
15982        let (path, reader) = stored_spellings("character-lengths-nulls", &spellings);
15983        let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
15984        let (codes, valid) = scattered_rows(spellings.len());
15985        let rows = Vector::dictionary_over(codes.clone(), Arc::clone(&dictionary))
15986            .expect("every code is inside")
15987            .with_validity(Validity::from_run(&valid));
15988
15989        let resting = dictionary.footprint();
15990        let lens = rudb_kernels::call("length", &[&rows], &LogicalType::BigInt, None)
15991            .expect("length reads");
15992        let counted = dictionary.footprint() - resting;
15993        let blocks = dictionary.len().div_ceil(TEXT_PAYLOAD_VALUES);
15994        assert!(
15995            counted <= blocks * TEXT_PAYLOAD_VALUES * size_of::<u32>(),
15996            "length over a vector with nulls kept {counted} bytes, more than a count a value"
15997        );
15998        let expected = (0..rows.len())
15999            .map(|row| match valid[row] {
16000                true => Value::BigInt(
16001                    i64::try_from(spellings[codes[row] as usize].chars().count()).expect("small"),
16002                ),
16003                false => Value::Null,
16004            })
16005            .collect::<Vec<_>>();
16006        let answers = (0..lens.len()).map(|row| lens.value_at(row)).collect::<Vec<_>>();
16007        assert_eq!(answers, expected, "a count of characters where a row has one, null elsewhere");
16008        fs::remove_file(path).expect("remove scratch file");
16009    }
16010
16011    /// `lower`, `upper` and `substring` read a stored dictionary a block at a time and keep none of
16012    /// it while the column is at its budget, until reading without keeping stops being cheap.
16013    ///
16014    /// The three used to read a row at a time through the bytes, which keeps every block a row lands
16015    /// in for as long as the table is open. They read the whole vector in one visit now, and the
16016    /// dictionary here is opened with a budget of zero so that what a visit would keep under the
16017    /// budget of a running database is what the test sees dropped. After a column's worth of blocks
16018    /// has been decoded and dropped the visit keeps what it reads, which is what bounds its cost on
16019    /// a scan whose codes keep coming back to every block, and the end of the test holds it to that.
16020    #[test]
16021    fn string_kernels_read_a_stored_dictionary_without_keeping_its_blocks() {
16022        let spellings = (0..2_500)
16023            .map(|index| format!("HéLLo {index:05} {}", "Üß".repeat(index % 30)))
16024            .collect::<Vec<_>>();
16025        let (path, reader) = stored_spellings("string-kernels", &spellings);
16026        let page = reader.table.dictionaries[0].expect("a string column has one");
16027        let starved =
16028            open_global_dictionary(Arc::clone(&reader.file), page, &LogicalType::Varchar, 0)
16029                .expect("a dictionary opens whatever it may keep");
16030        let starved = Arc::new(starved);
16031        let (codes, valid) = scattered_rows(spellings.len());
16032        let rows = Vector::dictionary_over(codes.clone(), Arc::clone(&starved))
16033            .expect("every code is inside")
16034            .with_validity(Validity::from_run(&valid));
16035        let expected = |each: &dyn Fn(&str) -> String| {
16036            (0..rows.len())
16037                .map(|row| match valid[row] {
16038                    true => Value::Varchar(each(&spellings[codes[row] as usize])),
16039                    false => Value::Null,
16040                })
16041                .collect::<Vec<_>>()
16042        };
16043        let answers =
16044            |vector: &Vector| (0..vector.len()).map(|row| vector.value_at(row)).collect::<Vec<_>>();
16045
16046        // What a visit may add is the table of where every value ends, four bytes a value, which
16047        // reading every value this often makes worth building. A block is tens of bytes a value.
16048        let resting = starved.footprint();
16049        let ends = spellings.len() * size_of::<u32>();
16050        let lowered = rudb_kernels::call("lower", &[&rows], &LogicalType::Varchar, None)
16051            .expect("lower reads");
16052        assert_eq!(answers(&lowered), expected(&|text| text.to_lowercase()), "lower");
16053        assert!(starved.footprint() <= resting + ends, "lower kept a block it read");
16054
16055        let start = Vector::constant(LogicalType::BigInt, Value::BigInt(3), rows.len());
16056        let length = Vector::constant(LogicalType::BigInt, Value::BigInt(9), rows.len());
16057        let cut =
16058            rudb_kernels::call("substring", &[&rows, &start, &length], &LogicalType::Varchar, None)
16059                .expect("substring reads");
16060        let cut_of = |text: &str| text.chars().skip(2).take(9).collect::<String>();
16061        assert_eq!(answers(&cut), expected(&cut_of), "substring");
16062        assert!(starved.footprint() <= resting + ends, "substring kept a block it read");
16063
16064        // Every block has been read twice now and dropped the second time as well, which is a
16065        // column's worth dropped for want of a budget, so the next visit keeps what it reads.
16066        let raised = rudb_kernels::call("upper", &[&rows], &LogicalType::Varchar, None)
16067            .expect("upper reads");
16068        assert_eq!(answers(&raised), expected(&|text| text.to_uppercase()), "upper");
16069        let payload = spellings.iter().map(String::len).sum::<usize>();
16070        assert!(
16071            starved.footprint() >= resting + payload,
16072            "a visit that has dropped a column's worth of blocks keeps what it reads"
16073        );
16074        let again = rudb_kernels::call("upper", &[&rows], &LogicalType::Varchar, None)
16075            .expect("upper reads kept blocks");
16076        assert_eq!(answers(&again), answers(&raised), "the kept blocks answer the same");
16077        fs::remove_file(path).expect("remove scratch file");
16078    }
16079
16080    /// A sweep of the dictionary reads every value, and the second sweep keeps what it read, up to
16081    /// the budget.
16082    ///
16083    /// The point of the sweep is the resident size rather than the answer, so both are checked
16084    /// here. The first sweep keeps nothing, because a process that runs one statement never reads
16085    /// a block twice. A dictionary this small is well under [`TEXT_KEEP_BUDGET`], so the second
16086    /// sweep keeps everything and a third decodes nothing, which is what makes a session asking the
16087    /// same question again cost what it should. The ceiling is the other half of it and it has its own
16088    /// test below, because a ceiling that never binds is not a ceiling anybody checked.
16089    #[test]
16090    fn a_dictionary_sweep_reads_every_value_and_keeps_it_under_the_budget() {
16091        let path = path("dictionary-sweep");
16092        // Two thousand five hundred distinct values is two whole payload blocks and a part of a
16093        // third, so the sweep has to be called more than once and the last call has to stop short.
16094        let spellings = (0..2_500)
16095            .map(|index| Value::Varchar(format!("value {index:08} {}", "x".repeat(index % 40))))
16096            .collect::<Vec<_>>();
16097        let mut writer =
16098            Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
16099                .expect("new file");
16100        // A chunk is a part and a part is at most 1,024 rows, so the values go in three of them.
16101        // The dictionary is table wide and does not care where a value was written.
16102        for part in spellings.chunks(1_024) {
16103            writer
16104                .append(
16105                    &Chunk::new(vec![
16106                        Vector::from_values(LogicalType::Varchar, part).expect("strings"),
16107                    ])
16108                    .expect("one column"),
16109                )
16110                .expect("stripe written");
16111        }
16112        writer.finish().expect("commit");
16113
16114        let reader = Reader::open(&path).expect("valid directory");
16115        let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
16116        assert_eq!(dictionary.len(), spellings.len(), "every value is distinct");
16117        for first in [0, TEXT_PAYLOAD_VALUES, TEXT_PAYLOAD_VALUES * 2] {
16118            assert!(dictionary.text_block_might_contain(first, b"value").expect("signature"));
16119            assert!(!dictionary.text_block_might_contain(first, b"google").expect("signature"));
16120        }
16121
16122        let resting = dictionary.footprint();
16123        let sweep = || {
16124            let mut swept: Vec<Vec<u8>> = Vec::new();
16125            let mut at = 0;
16126            let mut calls = 0;
16127            while at < dictionary.len() {
16128                let stopped = dictionary
16129                    .sweep_text(at, dictionary.len(), &mut |index: usize, text: &[u8]| {
16130                        assert_eq!(index, swept.len(), "a sweep hands its values over in order");
16131                        swept.push(text.to_vec());
16132                        Ok(())
16133                    })
16134                    .expect("a sweep reads");
16135                assert!(stopped > at, "a sweep moves");
16136                at = stopped;
16137                calls += 1;
16138            }
16139            assert_eq!(calls, 3, "a sweep hands over one block at a time");
16140            swept
16141        };
16142        let swept = sweep();
16143        assert_eq!(dictionary.footprint(), resting, "a first sweep keeps nothing it decoded");
16144        assert_eq!(sweep(), swept, "a second sweep reads what the first did");
16145        let after = dictionary.footprint();
16146        assert!(after > resting, "a second sweep under the budget keeps what it decoded");
16147
16148        let read = (0..dictionary.len())
16149            .map(|code| dictionary.try_bytes_at(code).expect("read").expect("a value").to_vec())
16150            .collect::<Vec<_>>();
16151        assert_eq!(swept, read, "a sweep answers what a point read answers");
16152        // A read per value is about what makes the unpacked ends worth building, so whether they
16153        // are built here depends on how many reads the sweep made on the way. They are the one thing
16154        // allowed to grow, by four bytes a value, and nothing of the payload is.
16155        let grown = dictionary.footprint() - after;
16156        assert!(
16157            grown == 0 || grown == dictionary.len() * size_of::<u32>(),
16158            "a point read of a kept block decodes nothing, and {grown} bytes grew"
16159        );
16160        fs::remove_file(path).expect("remove scratch file");
16161    }
16162
16163    #[test]
16164    fn a_narrow_signature_of_an_older_file_answers_by_its_own_width() {
16165        let path = path("narrow-substring-signature");
16166        let blocks = [&b"https://google.com/"[..], b"https://example.org/", b"mail.google.com"];
16167        let mut grams = Vec::new();
16168        for text in blocks {
16169            let mut bits = vec![0_u8; NARROW_GRAM_BYTES];
16170            for gram in text.windows(4) {
16171                for bit in gram_bits(gram, NARROW_GRAM_BYTES) {
16172                    bits[bit / 8] |= 1 << (bit % 8);
16173                }
16174            }
16175            grams.extend(bits);
16176        }
16177        fs::write(&path, &grams).expect("scratch file");
16178        let file = File::open(&path).expect("open scratch file");
16179        let signatures = NativeGrams {
16180            start: 0,
16181            length: grams.len(),
16182            width: NARROW_GRAM_BYTES,
16183            hash: checksum(&grams),
16184            verdicts: Mutex::new(Vec::new()),
16185        };
16186        let verdict = signatures.verdicts(&file, b"google").expect("signatures read");
16187        assert_eq!(&verdict[..], &[true, false, true], "one verdict a block, at the narrow width");
16188        assert!(signatures.footprint() > 0, "a verdict is remembered");
16189        let again = signatures.verdicts(&file, b"google").expect("remembered");
16190        assert!(Arc::ptr_eq(&verdict, &again), "a second question about a literal reads nothing");
16191
16192        let damaged = NativeGrams {
16193            hash: signatures.hash ^ 1,
16194            verdicts: Mutex::new(Vec::new()),
16195            ..signatures
16196        };
16197        let error = damaged.verdicts(&file, b"google").expect_err("a damaged region is refused");
16198        assert!(error.to_string().contains("substring signatures checksum differs"), "{error}");
16199        fs::remove_file(path).expect("remove scratch file");
16200    }
16201
16202    #[test]
16203    fn a_damaged_substring_signature_is_checked_only_when_used() {
16204        let path = path("damaged-substring-signature");
16205        let mut writer =
16206            Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
16207                .expect("new file");
16208        let rows = [Value::Varchar("google".into()), Value::Varchar("example".into())];
16209        writer
16210            .append(
16211                &Chunk::new(vec![
16212                    Vector::from_values(LogicalType::Varchar, &rows).expect("strings"),
16213                ])
16214                .expect("one column"),
16215            )
16216            .expect("stripe written");
16217        writer.finish().expect("commit");
16218
16219        let reader = Reader::open(&path).expect("valid directory");
16220        let page = reader.table.dictionaries[0].expect("string dictionary page");
16221        let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
16222        file.seek(SeekFrom::Start(page.offset + u64::from(page.length) - 1))
16223            .expect("last signature byte");
16224        file.write_all(&[255]).expect("damage signature");
16225        let reader = Reader::open(&path).expect("the directory is still valid");
16226        let dictionary = reader.dictionary(0).expect("index is still valid").expect("dictionary");
16227        let error = dictionary
16228            .text_block_might_contain(0, b"goog")
16229            .expect_err("a used signature checks its own checksum");
16230        assert!(error.message().contains("substring signatures checksum differs"), "{error}");
16231        fs::remove_file(path).expect("remove scratch file");
16232    }
16233
16234    /// A sweep over a block whose second run of offsets is short reads the same values as a point
16235    /// read does.
16236    ///
16237    /// The sweep decodes the offsets of a whole run at a time rather than a value at a time, and a
16238    /// run holds half a block, so the count it asks for is the run length everywhere but at the end
16239    /// of the dictionary. Two thousand five hundred values, which is what the test above writes,
16240    /// never puts a short run second in its block: the last block there begins on a run boundary and
16241    /// holds one run. Two thousand eight hundred does, so the last block is a whole run of five
16242    /// hundred and twelve followed by two hundred and forty, and an off by one in either the count
16243    /// asked for or the slice taken out of the answer shows up as a wrong value or a refusal.
16244    #[test]
16245    fn a_sweep_over_a_block_with_a_short_second_run_reads_what_a_point_read_reads() {
16246        let path = path("dictionary-sweep-short-run");
16247        let spellings = (0..2_800)
16248            .map(|index| Value::Varchar(format!("value {index:08} {}", "x".repeat(index % 40))))
16249            .collect::<Vec<_>>();
16250        let mut writer =
16251            Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
16252                .expect("new file");
16253        for part in spellings.chunks(1_024) {
16254            writer
16255                .append(
16256                    &Chunk::new(vec![
16257                        Vector::from_values(LogicalType::Varchar, part).expect("strings"),
16258                    ])
16259                    .expect("one column"),
16260                )
16261                .expect("stripe written");
16262        }
16263        writer.finish().expect("commit");
16264
16265        let reader = Reader::open(&path).expect("valid directory");
16266        let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
16267        assert_eq!(dictionary.len(), spellings.len(), "every value is distinct");
16268        let last = dictionary.len() % TEXT_PAYLOAD_VALUES;
16269        assert!(last > TEXT_OFFSET_RUN, "the last block has to reach into a second run of offsets");
16270        assert!(last < TEXT_PAYLOAD_VALUES, "and that second run has to be short of a whole one");
16271
16272        let mut swept: Vec<Vec<u8>> = Vec::new();
16273        let mut at = 0;
16274        while at < dictionary.len() {
16275            let stopped = dictionary
16276                .sweep_text(at, dictionary.len(), &mut |index: usize, text: &[u8]| {
16277                    assert_eq!(index, swept.len(), "a sweep hands its values over in order");
16278                    swept.push(text.to_vec());
16279                    Ok(())
16280                })
16281                .expect("a sweep reads");
16282            assert!(stopped > at, "a sweep moves");
16283            at = stopped;
16284        }
16285        let read = (0..dictionary.len())
16286            .map(|code| dictionary.try_bytes_at(code).expect("read").expect("a value").to_vec())
16287            .collect::<Vec<_>>();
16288        assert_eq!(swept, read, "a sweep answers what a point read answers");
16289        fs::remove_file(path).expect("remove scratch file");
16290    }
16291
16292    /// The unpacked ends answer what the packed ends answer, on both sides of the switch.
16293    ///
16294    /// A column asked for one offset at a time reads them out of the packed form until the reads
16295    /// are worth a table and out of the table after that, so every value here is read twice and the
16296    /// two passes are compared against the spellings and against each other. Two thousand eight
16297    /// hundred values is two payload blocks and a bit, which puts the switch in the middle of the
16298    /// first pass and means the pass straddles a block boundary, where the start of a value is zero
16299    /// rather than the end of the value before it.
16300    #[test]
16301    fn the_unpacked_ends_answer_what_the_packed_ends_answer() {
16302        let path = path("dictionary-unpacked-ends");
16303        let spellings = (0..2_800)
16304            .map(|index| Value::Varchar(format!("value {index:08} {}", "x".repeat(index % 40))))
16305            .collect::<Vec<_>>();
16306        let mut writer =
16307            Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
16308                .expect("new file");
16309        for part in spellings.chunks(1_024) {
16310            writer
16311                .append(
16312                    &Chunk::new(vec![
16313                        Vector::from_values(LogicalType::Varchar, part).expect("strings"),
16314                    ])
16315                    .expect("one column"),
16316                )
16317                .expect("stripe written");
16318        }
16319        writer.finish().expect("commit");
16320
16321        let reader = Reader::open(&path).expect("valid directory");
16322        let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
16323        assert_eq!(dictionary.len(), spellings.len(), "every value is distinct");
16324        let wanted = (0..spellings.len())
16325            .map(|index| format!("value {index:08} {}", "x".repeat(index % 40)).into_bytes())
16326            .collect::<Vec<_>>();
16327
16328        let pass = |what: &str| {
16329            for (index, value) in wanted.iter().enumerate() {
16330                let len = dictionary.try_bytes_len_at(index).expect("read").expect("a value");
16331                assert_eq!(len, value.len(), "{what} has the wrong length at {index}");
16332                let bytes = dictionary.try_bytes_at(index).expect("read").expect("a value");
16333                assert_eq!(bytes, value.as_slice(), "{what} has the wrong value at {index}");
16334            }
16335        };
16336        pass("the first pass");
16337        pass("the second pass");
16338
16339        // The whole vector in one call, over the text and through codes into it, which is how a
16340        // scan of a stored column hands it out. The codes run backwards and repeat so that they are
16341        // neither the positions nor in order.
16342        let lens = wanted.iter().map(|value| value.len() as i64).collect::<Vec<_>>();
16343        let mut whole = vec![0i64; wanted.len()];
16344        assert!(dictionary.try_bytes_lens(&mut whole).expect("read"), "the text answers whole");
16345        assert_eq!(whole, lens, "a vector of lengths answers what a length at a time answers");
16346        let codes = (0..4_000_u32).map(|row| (7 * (4_000 - row)) % 2_800).collect::<Vec<_>>();
16347        let coded = Vector::dictionary_over(codes.clone(), dictionary).expect("codes in range");
16348        let mut through = vec![0i64; codes.len()];
16349        assert!(coded.try_bytes_lens(&mut through).expect("read"), "the codes answer whole");
16350        for (row, &code) in codes.iter().enumerate() {
16351            assert_eq!(through[row], lens[code as usize], "row {row} reads code {code}");
16352            let one = coded.try_bytes_len_at(row).expect("read").expect("a value");
16353            assert_eq!(through[row], one as i64, "row {row} a row at a time");
16354        }
16355
16356        // A handful of codes over a column nobody has read yet is short of the table, so the same
16357        // call answers out of the packed ends instead, and has to answer the same.
16358        let fresh = Reader::open(&path).expect("valid directory");
16359        let untouched = fresh.dictionary(0).expect("read").expect("a string column has one");
16360        let few = vec![2_799_u32, 0, 1_024, 1_023, 511, 512];
16361        let coded = Vector::dictionary_over(few.clone(), untouched).expect("in range");
16362        let mut short = vec![0i64; few.len()];
16363        assert!(coded.try_bytes_lens(&mut short).expect("read"), "the codes answer whole");
16364        let expected = few.iter().map(|&code| lens[code as usize]).collect::<Vec<_>>();
16365        assert_eq!(short, expected, "the packed ends answer what the table answers");
16366        fs::remove_file(path).expect("remove scratch file");
16367    }
16368
16369    /// Narrowing a page takes what fits and refuses the page for anything that does not.
16370    ///
16371    /// The edges of the range on both sides and one step past each of them, for every type, because
16372    /// checking a page separately from converting it is only right if the check refuses exactly what
16373    /// `TryFrom` would have refused, and off by one there is a file that reads back a different
16374    /// number than it was given. The check is a bit pattern rather than a comparison, so it is not
16375    /// the shape a reader would guess from the bounds, which is why all six are here. The empty page
16376    /// is here because a check written the obvious way starts with the extremes the wrong way round
16377    /// and refuses it.
16378    #[test]
16379    fn narrowing_a_page_takes_what_fits_and_refuses_what_does_not() {
16380        assert_eq!(fit::<i8>(&[]).expect("an empty page fits anything"), Vec::<i8>::new());
16381        assert_eq!(fit::<i8>(&[-128, 0, 127]).expect("the edges fit"), vec![-128_i8, 0, 127]);
16382        fit::<i8>(&[128]).expect_err("one past the top does not fit");
16383        fit::<i8>(&[-129]).expect_err("one past the bottom does not fit");
16384        assert_eq!(fit::<u8>(&[0, 255]).expect("the edges fit"), vec![0_u8, 255]);
16385        fit::<u8>(&[256]).expect_err("one past the top does not fit");
16386        fit::<u8>(&[-1]).expect_err("a negative does not fit an unsigned page");
16387        assert_eq!(
16388            fit::<i16>(&[-32_768, 0, 32_767]).expect("the edges fit"),
16389            vec![-32_768_i16, 0, 32_767]
16390        );
16391        fit::<i16>(&[32_768]).expect_err("one past the top does not fit");
16392        fit::<i16>(&[-32_769]).expect_err("one past the bottom does not fit");
16393        assert_eq!(fit::<u16>(&[0, 65_535]).expect("the edges fit"), vec![0_u16, 65_535]);
16394        fit::<u16>(&[65_536]).expect_err("one past the top does not fit");
16395        fit::<u16>(&[-1]).expect_err("a negative does not fit an unsigned page");
16396        assert_eq!(
16397            fit::<i32>(&[i64::from(i32::MIN), 0, i64::from(i32::MAX)]).expect("the edges fit"),
16398            vec![i32::MIN, 0, i32::MAX]
16399        );
16400        fit::<i32>(&[i64::from(i32::MAX) + 1]).expect_err("one past the top does not fit");
16401        fit::<i32>(&[i64::from(i32::MIN) - 1]).expect_err("one past the bottom does not fit");
16402        assert_eq!(
16403            fit::<u32>(&[0, 4_294_967_295]).expect("the edges fit"),
16404            vec![0_u32, 4_294_967_295]
16405        );
16406        fit::<u32>(&[4_294_967_296]).expect_err("one past the top does not fit");
16407        fit::<u32>(&[-1]).expect_err("a negative does not fit an unsigned page");
16408
16409        // One value in a page that fits is still a page that does not, which is the thing an or
16410        // into an accumulator could get wrong in a way a page of one value would never show.
16411        fit::<i8>(&[0, 1, 2, 128, 3]).expect_err("one bad value spoils the page");
16412    }
16413
16414    /// The residue says yes to exactly what `TryFrom` says yes to.
16415    ///
16416    /// The edges above are the cases anyone would think to write down. This is the argument that
16417    /// there are no others, made by asking both questions about every value either narrow type could
16418    /// have an opinion about, and then about the values around the wide edges and the ends of an
16419    /// `i64`, which a range that size cannot reach.
16420    #[test]
16421    fn the_residue_agrees_with_a_checked_conversion_everywhere() {
16422        for value in -70_000_i64..70_000 {
16423            assert_eq!(fit::<i8>(&[value]).is_ok(), i8::try_from(value).is_ok(), "{value} as i8");
16424            assert_eq!(fit::<u8>(&[value]).is_ok(), u8::try_from(value).is_ok(), "{value} as u8");
16425            assert_eq!(fit::<i16>(&[value]).is_ok(), i16::try_from(value).is_ok(), "{value} i16");
16426            assert_eq!(fit::<u16>(&[value]).is_ok(), u16::try_from(value).is_ok(), "{value} u16");
16427        }
16428        let wide = [i64::MIN, i64::MIN + 1, i64::from(i32::MIN), 0, i64::from(u32::MAX), i64::MAX];
16429        for edge in wide {
16430            for step in -2_i64..=2 {
16431                let value = edge.saturating_add(step);
16432                assert_eq!(
16433                    fit::<i32>(&[value]).is_ok(),
16434                    i32::try_from(value).is_ok(),
16435                    "{value} as i32"
16436                );
16437                assert_eq!(
16438                    fit::<u32>(&[value]).is_ok(),
16439                    u32::try_from(value).is_ok(),
16440                    "{value} as u32"
16441                );
16442            }
16443        }
16444    }
16445
16446    /// All three block layouts come back as the same values in the same order.
16447    ///
16448    /// Blocks outside the page are what every file this build writes holds. Blocks that say where
16449    /// they are but sit inside the page behind the order are format 26, and blocks behind one
16450    /// another with only their ends recorded are older still. Nothing in the writer produces the
16451    /// last two any more, so the only way to find out whether the reader still understands those
16452    /// files is to write them here. The
16453    /// bytes go straight into a file with no directory around them, because what is under test is
16454    /// [`open_global_dictionary`], which is handed a page and a file and asks the directory for
16455    /// nothing.
16456    ///
16457    /// Three thousand values so that there are three payload blocks and a partial fourth, which is
16458    /// what makes the last block the one place where a length and an end disagree about what they
16459    /// are counting.
16460    #[test]
16461    fn a_dictionary_reads_the_same_whether_its_blocks_say_where_they_are() {
16462        let spellings = (0..3_000)
16463            .map(|index| format!("value {index:08} {}", "y".repeat(index % 40)))
16464            .collect::<Vec<_>>();
16465        let mut read = Vec::new();
16466        for layout in ["outside", "inside", "behind"] {
16467            let mut dictionary = GlobalDictionary::new();
16468            for text in &spellings {
16469                dictionary.code(text).expect("a code for every spelling");
16470            }
16471            dictionary.finish_blocks().expect("the last block encodes");
16472            let order = dictionary.ranked(None).expect("a sorted order");
16473            // Where the blocks go if they start at `from` and follow one another.
16474            let laid = |from: u64| {
16475                let mut at = from;
16476                dictionary
16477                    .blocks
16478                    .iter()
16479                    .map(|block| {
16480                        let place =
16481                            Placed { start: at, length: block.len() as u64, hash: checksum(block) };
16482                        at += block.len() as u64;
16483                        place
16484                    })
16485                    .collect::<Vec<_>>()
16486            };
16487            let payload = dictionary.blocks.concat();
16488            let scattered = layout != "behind";
16489            let (bytes, encoded, offset, length) = if layout == "outside" {
16490                let mut bytes = vec![0; HEADER as usize];
16491                bytes.extend_from_slice(&payload);
16492                let encoded = encode_global_dictionary(&dictionary, &order, &laid(HEADER), true)
16493                    .expect("an encoding");
16494                let offset = bytes.len() as u64;
16495                bytes.extend_from_slice(&encoded.index);
16496                bytes.extend_from_slice(&encoded.ranks);
16497                bytes.extend_from_slice(&encoded.grams);
16498                let length = encoded.index.len() + encoded.ranks.len() + encoded.grams.len();
16499                (bytes, encoded, offset, length)
16500            } else {
16501                // The index is the same length wherever the blocks are, so a first pass says where
16502                // the page ends and the second writes the places that follow it.
16503                let first = encode_global_dictionary(&dictionary, &order, &laid(0), scattered)
16504                    .expect("an encoding");
16505                let body = (first.index.len() + first.ranks.len() + first.grams.len()) as u64;
16506                let encoded = encode_global_dictionary(&dictionary, &order, &laid(body), scattered)
16507                    .expect("an encoding");
16508                let mut bytes = encoded.index.clone();
16509                bytes.extend_from_slice(&encoded.ranks);
16510                bytes.extend_from_slice(&encoded.grams);
16511                bytes.extend_from_slice(&payload);
16512                let length = bytes.len();
16513                (bytes, encoded, 0, length)
16514            };
16515            let path = path(&format!("blocks-{layout}"));
16516            fs::write(&path, &bytes).expect("the dictionary is written on its own");
16517            let file = Arc::new(File::open(&path).expect("it opens again"));
16518            let page = Page {
16519                offset,
16520                length: u32::try_from(length).expect("a test dictionary is small"),
16521                hash: checksum(&encoded.index),
16522            };
16523            let opened =
16524                open_global_dictionary(file, page, &LogicalType::Varchar, TEXT_KEEP_BUDGET)
16525                    .expect("a dictionary laid out either way opens");
16526            let mut swept: Vec<Vec<u8>> = Vec::new();
16527            let mut at = 0;
16528            while at < opened.len() {
16529                at = opened
16530                    .sweep_text(at, opened.len(), &mut |_index: usize, text: &[u8]| {
16531                        swept.push(text.to_vec());
16532                        Ok(())
16533                    })
16534                    .expect("a sweep reads");
16535            }
16536            fs::remove_file(&path).expect("clean up");
16537            read.push(swept);
16538        }
16539        let wanted =
16540            spellings.iter().map(|text| text.as_bytes().to_vec()).collect::<Vec<Vec<u8>>>();
16541        assert_eq!(read[0], wanted, "the blocks outside the page hold the values");
16542        assert_eq!(read[1], read[0], "the blocks inside the page hold the same values");
16543        assert_eq!(read[2], read[0], "the blocks behind one another hold the same values");
16544    }
16545
16546    /// A dictionary at its budget sweeps without keeping, and still answers what it answered.
16547    ///
16548    /// The budget is a quarter of a gigabyte in a running database, which is a fine size for a real
16549    /// column and no size at all for a test, so this opens the same dictionary a second time with a
16550    /// budget of zero. That is the shape of the hundred million row case: `URL` fills the budget
16551    /// somewhere in the middle of itself and everything past that point is read and dropped, which
16552    /// costs the decode again and holds none of it.
16553    #[test]
16554    fn a_dictionary_at_its_budget_sweeps_without_keeping() {
16555        let path = path("dictionary-budget");
16556        let spellings = (0..2_500)
16557            .map(|index| Value::Varchar(format!("value {index:08} {}", "y".repeat(index % 40))))
16558            .collect::<Vec<_>>();
16559        let mut writer =
16560            Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
16561                .expect("new file");
16562        for part in spellings.chunks(1_024) {
16563            writer
16564                .append(
16565                    &Chunk::new(vec![
16566                        Vector::from_values(LogicalType::Varchar, part).expect("strings"),
16567                    ])
16568                    .expect("one column"),
16569                )
16570                .expect("stripe written");
16571        }
16572        writer.finish().expect("commit");
16573
16574        let reader = Reader::open(&path).expect("valid directory");
16575        let page = reader.table.dictionaries[0].expect("a string column has one");
16576        let file = Arc::clone(&reader.file);
16577        let starved = open_global_dictionary(file, page, &LogicalType::Varchar, 0)
16578            .expect("a dictionary opens whatever it may keep");
16579
16580        let resting = starved.footprint();
16581        let mut swept: Vec<Vec<u8>> = Vec::new();
16582        let mut at = 0;
16583        while at < starved.len() {
16584            at = starved
16585                .sweep_text(at, starved.len(), &mut |_index: usize, text: &[u8]| {
16586                    swept.push(text.to_vec());
16587                    Ok(())
16588                })
16589                .expect("a sweep reads");
16590        }
16591        assert_eq!(swept.len(), spellings.len(), "a starved sweep still reads every value");
16592        assert_eq!(starved.footprint(), resting, "and keeps no block it decoded");
16593
16594        let generous = reader.dictionary(0).expect("read").expect("a string column has one");
16595        let read = (0..generous.len())
16596            .map(|code| generous.try_bytes_at(code).expect("read").expect("a value").to_vec())
16597            .collect::<Vec<_>>();
16598        assert_eq!(swept, read, "a starved sweep answers what a point read answers");
16599        fs::remove_file(path).expect("remove scratch file");
16600    }
16601
16602    #[test]
16603    fn damaged_membership_cannot_skip_a_string_page() {
16604        let path = path("damaged-membership");
16605        let mut writer = Writer::create(
16606            &path,
16607            "items",
16608            vec![
16609                Field::required("id", LogicalType::Integer),
16610                Field::new("text", LogicalType::Varchar),
16611            ],
16612        )
16613        .expect("new file");
16614        writer.append(&sample()).expect("stripe written");
16615        writer.finish().expect("commit");
16616
16617        let reader = Reader::open(&path).expect("valid directory");
16618        let membership = reader.table.stripes[0].memberships.get(1).expect("string membership");
16619        let mut file = OpenOptions::new().write(true).open(&path).expect("open membership page");
16620        file.seek(SeekFrom::Start(membership.offset)).expect("membership start");
16621        file.write_all(&[255]).expect("damage membership");
16622        let error = reader.skips_codes(0, 1, &[3]).expect_err("corruption must not skip rows");
16623        assert!(error.message().contains("membership page checksum differs"), "{error}");
16624        fs::remove_file(path).expect("remove scratch file");
16625    }
16626
16627    #[test]
16628    fn membership_delta_stream_is_sorted_exact_and_bounded() {
16629        let unique = unique_codes(&[900, 4, 4, 72, 9, u32::MAX]);
16630        assert_eq!(unique, [4, 9, 72, 900, u32::MAX]);
16631        let encoded = encode_membership(&unique);
16632        assert_eq!(
16633            decode_membership(&encoded).expect("valid membership"),
16634            [4, 9, 72, 900, u32::MAX]
16635        );
16636        // A stripe's index is the union of its parts', so a code in two of them is in it once and
16637        // the result is still one ascending run of deltas.
16638        let merged = merged_codes(vec![vec![4, 900], vec![9, 900, u32::MAX], vec![72]]);
16639        assert_eq!(merged, [4, 9, 72, 900, u32::MAX]);
16640        assert_eq!(
16641            decode_membership(&encode_membership(&merged)).expect("valid membership"),
16642            unique
16643        );
16644        assert!(decode_membership(&[1, 0x80]).is_err(), "a truncated varint is invalid");
16645        assert!(
16646            decode_membership(&[1, 0xff, 0xff, 0xff, 0xff, 0x10]).is_err(),
16647            "a value past u32 is invalid"
16648        );
16649    }
16650
16651    #[test]
16652    fn a_global_dictionary_may_be_larger_than_one_column_page() {
16653        let dictionary = Page {
16654            offset: HEADER,
16655            length: u32::try_from(MAX_PAGE + 1).expect("the page bound fits on disk"),
16656            hash: 0,
16657        };
16658        let table = Table {
16659            name: "items".to_owned(),
16660            fields: vec![Field::new("text", LogicalType::Varchar)],
16661            stripes: Vec::new(),
16662            rows: 0,
16663            dictionaries: vec![Some(dictionary)],
16664            dictionary_payloads: Vec::new(),
16665            demoted: Vec::new(),
16666            distincts: vec![None],
16667            frequencies: vec![None],
16668            pair_frequencies: Vec::new(),
16669            frequency_texts: Vec::new(),
16670            host_groups: None,
16671            clustering: None,
16672            generation: 1,
16673            sections: Vec::new(),
16674        };
16675        let directory = encode_directory(&table).expect("directory");
16676        let file_size = dictionary.offset + u64::from(dictionary.length) + 1;
16677
16678        let decoded = decode_directory(&directory, file_size).expect("large lazy dictionary");
16679        assert_eq!(decoded.dictionaries[0].expect("dictionary").length, dictionary.length);
16680    }
16681
16682    #[test]
16683    fn a_column_with_one_value_everywhere_costs_almost_nothing_a_row() {
16684        let path = path("constant-codes");
16685        let mut writer =
16686            Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
16687                .expect("new file");
16688        let empty = vec![Value::Varchar(String::new()); 1024];
16689        for _ in 0..4 {
16690            let column = Vector::from_values(LogicalType::Varchar, &empty).expect("strings");
16691            writer.append(&Chunk::new(vec![column]).expect("one column")).expect("a part");
16692        }
16693        writer.finish().expect("commit");
16694
16695        let reader = Reader::open(&path).expect("valid directory");
16696        let pages = reader.layout().columns.first().expect("one column").pages;
16697        // This column used to cost four bytes a row, 16,384 of them, the same as a column of four
16698        // thousand distinct URLs would. The cascade calls each part a constant, so what is left is
16699        // a tag, a count and the value, and the row count stops being what drives the number.
16700        assert!(pages < 256, "{pages} bytes of pages for 4,096 rows of one value");
16701        let read = reader.read(3, &[0]).expect("the last part back");
16702        assert_eq!(read.value_at(0, 0), Value::Varchar(String::new()));
16703        assert_eq!(read.value_at(1023, 0), Value::Varchar(String::new()));
16704        fs::remove_file(path).expect("remove scratch file");
16705    }
16706
16707    #[test]
16708    fn a_cascade_value_too_wide_for_its_column_is_refused_rather_than_cut() {
16709        // What a damaged page looks like from here: the cascade decoded, so the bytes are not
16710        // truncated, but the values do not belong to the column the directory says they do.
16711        let over = vec![i64::from(i32::MAX) + 1];
16712        let error = narrowed(&LogicalType::Integer, over).expect_err("a page that disagrees");
16713        assert!(format!("{error}").contains("not of its type"), "{error}");
16714        assert!(narrowed(&LogicalType::BigInt, vec![i64::MIN]).is_ok(), "bigint holds all of i64");
16715        assert!(narrowed(&LogicalType::Varchar, vec![0]).is_err(), "strings are not integers");
16716    }
16717
16718    #[test]
16719    fn a_code_stream_the_cascade_cannot_shrink_is_left_alone() {
16720        // A shift register rather than a run, because an arithmetic run is the one wide shape the
16721        // cascade does shrink. This is what a column with tens of millions of distinct values hands
16722        // over: full width codes with no order to them.
16723        let mut state: u32 = 0x9e37_79b9;
16724        let spread: Vec<u32> = (0..1024)
16725            .map(|_| {
16726                state ^= state << 13;
16727                state ^= state >> 17;
16728                state ^= state << 5;
16729                state
16730            })
16731            .collect();
16732        assert_eq!(encoded_codes(&spread).expect("no failure"), None);
16733        let near: Vec<u32> = (0..1024).collect();
16734        let coded = encoded_codes(&near).expect("no failure").expect("counting up is packable");
16735        assert!(coded.len() < near.len() * 4, "{} bytes for a run of 1,024", coded.len());
16736    }
16737
16738    /// The columns of a stripe are encoded on whichever thread got to them, so the one thing that
16739    /// must not depend on which thread that was is the file. Two writes of the same rows are
16740    /// compared byte for byte rather than value for value, because a dictionary that two columns
16741    /// somehow shared would still read back correctly and would hand out its codes in the order the
16742    /// threads happened to run in, which is exactly what this is here to catch.
16743    #[test]
16744    fn two_writes_of_the_same_rows_give_the_same_bytes() {
16745        fn written(path: &PathBuf) {
16746            let fields = (0..40)
16747                .map(|column| {
16748                    let ty =
16749                        if column % 4 == 0 { LogicalType::Varchar } else { LogicalType::BigInt };
16750                    Field::new(format!("c{column}"), ty)
16751                })
16752                .collect::<Vec<_>>();
16753            let mut writer = Writer::create(path, "wide", fields).expect("new file");
16754            for part in 0..70_u64 {
16755                let columns = (0..40)
16756                    .map(|column| {
16757                        let values = (0..64_u64)
16758                            .map(|row| {
16759                                let seed = part.wrapping_mul(31).wrapping_add(row);
16760                                if column % 4 == 0 {
16761                                    Value::Varchar(format!("v{}", seed % 17))
16762                                } else {
16763                                    Value::BigInt(i64::try_from(seed % 97).expect("small"))
16764                                }
16765                            })
16766                            .collect::<Vec<_>>();
16767                        let ty = if column % 4 == 0 {
16768                            LogicalType::Varchar
16769                        } else {
16770                            LogicalType::BigInt
16771                        };
16772                        Vector::from_values(ty, &values).expect("a column")
16773                    })
16774                    .collect::<Vec<_>>();
16775                writer.append(&Chunk::new(columns).expect("forty columns")).expect("a part");
16776            }
16777            writer.finish().expect("commit");
16778        }
16779
16780        let first = path("repeatable-one");
16781        let second = path("repeatable-two");
16782        written(&first);
16783        written(&second);
16784        let left = fs::read(&first).expect("the first file");
16785        let right = fs::read(&second).expect("the second file");
16786        assert_eq!(left.len(), right.len(), "two writes of the same rows differ in length");
16787        assert!(left == right, "two writes of the same rows differ in their bytes");
16788
16789        // And the rows are still there, since a pair of identically wrong files would pass the
16790        // comparison above on its own.
16791        let reader = Reader::open(&first).expect("valid directory");
16792        assert_eq!(reader.table().rows(), 70 * 64);
16793        let read = reader.read(0, &[0, 1]).expect("the first part back");
16794        assert_eq!(read.value_at(0, 0), Value::Varchar("v0".to_owned()));
16795        assert_eq!(read.value_at(0, 1), Value::BigInt(0));
16796        fs::remove_file(first).expect("remove scratch file");
16797        fs::remove_file(second).expect("remove scratch file");
16798    }
16799
16800    /// Three tables of different shapes in one file, read back by name.
16801    fn three_tables(path: &PathBuf) {
16802        let writer = Writer::create(
16803            path,
16804            "region",
16805            vec![
16806                Field::new("r_key", LogicalType::Integer),
16807                Field::new("r_name", LogicalType::Varchar),
16808            ],
16809        )
16810        .expect("new file");
16811        let mut writer = writer;
16812        writer
16813            .append(
16814                &Chunk::new(vec![
16815                    Vector::from_values(
16816                        LogicalType::Integer,
16817                        &[Value::Integer(0), Value::Integer(1)],
16818                    )
16819                    .expect("keys"),
16820                    Vector::from_values(
16821                        LogicalType::Varchar,
16822                        &[Value::Varchar("AFRICA".to_owned()), Value::Varchar("ASIA".to_owned())],
16823                    )
16824                    .expect("names"),
16825                ])
16826                .expect("two columns"),
16827            )
16828            .expect("a part");
16829        let mut writer = writer
16830            .next("empty", vec![Field::new("nothing", LogicalType::BigInt)])
16831            .expect("a second table");
16832        writer
16833            .append(
16834                &Chunk::new(vec![
16835                    Vector::from_values(LogicalType::BigInt, &[Value::BigInt(7)]).expect("a row"),
16836                ])
16837                .expect("one column"),
16838            )
16839            .expect("a part");
16840        let mut writer =
16841            writer.next("wide", vec![Field::new("n", LogicalType::BigInt)]).expect("a third table");
16842        for part in 0..70_i64 {
16843            let values = (0..64).map(|row| Value::BigInt(part * 64 + row)).collect::<Vec<_>>();
16844            writer
16845                .append(
16846                    &Chunk::new(vec![
16847                        Vector::from_values(LogicalType::BigInt, &values).expect("a column"),
16848                    ])
16849                    .expect("one column"),
16850                )
16851                .expect("a part");
16852        }
16853        writer.finish().expect("commit");
16854    }
16855
16856    #[test]
16857    fn three_tables_in_one_file_read_back_by_name() {
16858        let file = path("three-tables");
16859        three_tables(&file);
16860        let catalog = Catalog::open(&file).expect("a committed catalog");
16861        assert_eq!(catalog.names().collect::<Vec<_>>(), ["region", "empty", "wide"]);
16862
16863        let region = catalog.table("region").expect("the first table");
16864        assert_eq!(region.table().rows(), 2);
16865        assert_eq!(
16866            region.read(0, &[1]).expect("names").value_at(1, 0),
16867            Value::Varchar("ASIA".to_owned())
16868        );
16869
16870        let wide = catalog.table("wide").expect("the third table");
16871        assert_eq!(wide.table().rows(), 70 * 64);
16872        assert_eq!(wide.read(0, &[0]).expect("the first part").value_at(0, 0), Value::BigInt(0));
16873
16874        // The middle table is reached without the one after it having been touched, which is what
16875        // a directory per table buys over one directory of everything.
16876        let empty = catalog.table("empty").expect("the second table");
16877        assert_eq!(empty.table().rows(), 1);
16878        assert_eq!(empty.read(0, &[0]).expect("the row").value_at(0, 0), Value::BigInt(7));
16879
16880        fs::remove_file(file).expect("remove scratch file");
16881    }
16882
16883    #[test]
16884    fn a_name_the_file_does_not_hold_is_an_error_rather_than_the_first_table() {
16885        let file = path("three-tables-missing");
16886        three_tables(&file);
16887        let catalog = Catalog::open(&file).expect("a committed catalog");
16888        let error = catalog.table("nation").expect_err("no such table");
16889        assert!(error.message().contains("nation"), "{}", error.message());
16890        fs::remove_file(file).expect("remove scratch file");
16891    }
16892
16893    #[test]
16894    fn a_file_of_three_tables_will_not_open_as_one() {
16895        let file = path("three-tables-unnamed");
16896        three_tables(&file);
16897        let error = Reader::open(&file).expect_err("more than one table");
16898        assert!(error.message().contains("more than one table"), "{}", error.message());
16899        fs::remove_file(file).expect("remove scratch file");
16900    }
16901
16902    /// One column per storage width, because the width is what decides how many bytes a row costs.
16903    #[test]
16904    fn decimals_of_every_storage_width_round_trip() {
16905        let file = path("decimals");
16906        let widths = [(4_u8, 2_u8), (9, 2), (18, 4), (38, 6)];
16907        let fields = widths
16908            .iter()
16909            .enumerate()
16910            .map(|(index, (width, scale))| {
16911                Field::new(
16912                    format!("d{index}"),
16913                    LogicalType::decimal(*width, *scale).expect("a decimal type"),
16914                )
16915            })
16916            .collect::<Vec<_>>();
16917        let mut writer = Writer::create(&file, "money", fields).expect("new file");
16918        let rows: [i128; 3] = [-1234, 0, 999];
16919        let columns = widths
16920            .iter()
16921            .map(|(width, scale)| {
16922                let values = rows
16923                    .iter()
16924                    .map(|unscaled| Value::Decimal {
16925                        unscaled: *unscaled,
16926                        width: *width,
16927                        scale: *scale,
16928                    })
16929                    .collect::<Vec<_>>();
16930                Vector::from_values(
16931                    LogicalType::decimal(*width, *scale).expect("a decimal type"),
16932                    &values,
16933                )
16934                .expect("a decimal column")
16935            })
16936            .collect::<Vec<_>>();
16937        writer.append(&Chunk::new(columns).expect("four columns")).expect("a part");
16938        writer.finish().expect("commit");
16939
16940        let reader = Reader::open(&file).expect("a committed file");
16941        for (index, (width, scale)) in widths.iter().enumerate() {
16942            assert_eq!(
16943                reader.table().fields()[index].ty,
16944                LogicalType::decimal(*width, *scale).expect("a decimal type"),
16945                "column {index} came back as another type"
16946            );
16947            let column = reader.read(0, &[index]).expect("the column");
16948            for (row, unscaled) in rows.iter().enumerate() {
16949                assert_eq!(
16950                    column.value_at(row, 0),
16951                    Value::Decimal { unscaled: *unscaled, width: *width, scale: *scale },
16952                    "column {index} row {row}"
16953                );
16954            }
16955        }
16956        fs::remove_file(file).expect("remove scratch file");
16957    }
16958
16959    #[test]
16960    fn two_tables_of_one_name_are_refused_before_anything_is_committed() {
16961        let file = path("two-of-a-name");
16962        let writer = Writer::create(&file, "t", vec![Field::new("a", LogicalType::BigInt)])
16963            .expect("new file");
16964        let error = writer
16965            .next("t", vec![Field::new("a", LogicalType::BigInt)])
16966            .expect_err("the same name twice");
16967        assert!(error.message().contains("same name"), "{}", error.message());
16968        fs::remove_file(file).expect("remove scratch file");
16969    }
16970
16971    #[test]
16972    fn integer_tally_counts_encoded_rows_and_declines_null_parts() {
16973        let file = path("integer-tally");
16974        let mut writer =
16975            Writer::create(&file, "events", vec![Field::new("source", LogicalType::SmallInt)])
16976                .expect("new file");
16977        let mut values = vec![Value::SmallInt(0); 1024];
16978        values[7] = Value::SmallInt(3);
16979        values[99] = Value::SmallInt(-2);
16980        values[1001] = Value::SmallInt(3);
16981        let column = Vector::from_values(LogicalType::SmallInt, &values).expect("integer values");
16982        writer.append(&Chunk::new(vec![column]).expect("one column")).expect("first part");
16983        values[0] = Value::Null;
16984        let column = Vector::from_values(LogicalType::SmallInt, &values).expect("nullable values");
16985        writer.append(&Chunk::new(vec![column]).expect("one column")).expect("second part");
16986        writer.finish().expect("commit");
16987
16988        let reader = Reader::open(&file).expect("read file");
16989        assert_eq!(
16990            reader.integer_tally(0, 0).expect("valid part"),
16991            Some(vec![(-2, 1), (0, 1021), (3, 2)])
16992        );
16993        assert!(reader.integer_tally(1, 0).expect("valid null part").is_none());
16994        let catalog = Catalog::open(&file).expect("catalog");
16995        assert_eq!(
16996            catalog.integer_tally("events", 0).expect("nullable column"),
16997            Some(vec![(-2, 2), (0, 2041), (3, 4)])
16998        );
16999        fs::remove_file(file).expect("remove scratch file");
17000    }
17001
17002    #[test]
17003    fn catalog_tallies_one_integer_column_without_opening_the_whole_table() {
17004        let file = path("catalog-integer-tally");
17005        let mut writer = Writer::create(
17006            &file,
17007            "events",
17008            vec![
17009                Field::new("noise", LogicalType::SmallInt),
17010                Field::new("source", LogicalType::SmallInt),
17011            ],
17012        )
17013        .expect("new file");
17014        let noise = vec![Value::SmallInt(9); 1024];
17015        let mut source = vec![Value::SmallInt(0); 1024];
17016        source[7] = Value::SmallInt(3);
17017        source[99] = Value::SmallInt(-2);
17018        let chunk = Chunk::new(vec![
17019            Vector::from_values(LogicalType::SmallInt, &noise).expect("noise"),
17020            Vector::from_values(LogicalType::SmallInt, &source).expect("source"),
17021        ])
17022        .expect("two columns");
17023        writer.append(&chunk).expect("append");
17024        writer.finish().expect("commit");
17025
17026        let catalog = Catalog::open(&file).expect("catalog");
17027        assert_eq!(
17028            catalog.integer_tally("events", 1).expect("selected column"),
17029            Some(vec![(-2, 1), (0, 1022), (3, 1)])
17030        );
17031        assert_eq!(
17032            catalog.integer_tally("events", 0).expect("other column"),
17033            Some(vec![(9, 1024)])
17034        );
17035        fs::remove_file(file).expect("remove scratch file");
17036    }
17037
17038    #[test]
17039    fn opening_the_catalog_reads_no_table_directory() {
17040        let file = path("catalog-only");
17041        three_tables(&file);
17042        let catalog = Catalog::open(&file).expect("a committed catalog");
17043        // The header and one slot, and nothing under it. The third table's directory covers seventy
17044        // stripes and reading it here would be the whole point of the two levels thrown away.
17045        assert_eq!(catalog.opening.reads, 2, "opening the catalog read more than the slot");
17046        assert_eq!(catalog.names().len(), 3);
17047        fs::remove_file(file).expect("remove scratch file");
17048    }
17049
17050    /// The checksum answers what it has always answered, at every length its branches split on.
17051    ///
17052    /// This is a compatibility test rather than a correctness one. Nothing about the hash has to be
17053    /// any particular function, but a file already on disk carries the answers the version that
17054    /// wrote it gave, so a change here is a change that makes every stored file fail to verify. The
17055    /// lengths are the ones the code makes decisions about: nothing, under a block, a block exactly,
17056    /// a block and a word, a word and a half word, and a half word and a byte.
17057    ///
17058    /// The empty answer is the published xxHash64 vector for an empty input at seed zero, which is
17059    /// also a check that this is the function it says it is.
17060    #[test]
17061    fn the_checksum_answers_what_it_has_always_answered() {
17062        let bytes: Vec<u8> =
17063            (0..1000_u32).map(|at| (at.wrapping_mul(31).wrapping_add(7) % 251) as u8).collect();
17064        for (length, expected) in [
17065            (0, 0xef46_db37_51d8_e999),
17066            (1, 0xa96c_7f0c_e858_bbb7),
17067            (3, 0x56e6_9576_32a4_87f9),
17068            (4, 0xc60d_15b1_e3ff_8f04),
17069            (5, 0x8088_1585_8624_dd4e),
17070            (7, 0xafbe_fc3d_6c6f_9a8e),
17071            (8, 0x3da5_c7aa_2696_83e0),
17072            (9, 0x465e_c429_b13c_3892),
17073            (15, 0xdee8_9d8a_065a_6233),
17074            (16, 0x1330_489a_7767_9c80),
17075            (31, 0x3391_303d_485e_846e),
17076            (32, 0x40b7_aff7_5d45_bbc8),
17077            (33, 0x4997_cae4_951c_17a5),
17078            (39, 0x5807_28fd_5c14_5739),
17079            (40, 0xf95c_f6f5_c08a_3d3b),
17080            (63, 0x2944_b4da_fc69_b206),
17081            (64, 0xbb76_f6ef_19bd_5a1b),
17082            (65, 0x814e_0c65_4a9f_d640),
17083            (127, 0x00de_aab1_31cf_f89b),
17084            (1000, 0x9e33_00c1_cde3_c58d),
17085        ] {
17086            assert_eq!(checksum(&bytes[..length]), expected, "the checksum of {length} bytes");
17087        }
17088        assert_eq!(checksum(b"the quick brown fox jumps over the lazy dog"), 0xed71_4233_c5a9_a792);
17089    }
17090    /// A declared order survives the file, and a table that declared none stays as it was.
17091    ///
17092    /// The second half is the one worth a test. The clustering section is written only when there
17093    /// is a declaration, so a file of two tables where one is clustered exercises both the present
17094    /// and the absent branch of the decoder in one directory, which is where a length bug would
17095    /// show up as one table reading the other's bytes.
17096    #[test]
17097    fn a_declared_order_comes_back_out_of_the_file() {
17098        let path = path("clustered");
17099        let shipped = vec![
17100            Field::new("key", LogicalType::BigInt),
17101            Field::new("line", LogicalType::Integer),
17102            Field::new("shipdate", LogicalType::Date),
17103        ];
17104        let plain = vec![Field::new("a", LogicalType::Integer)];
17105        let stage_zero = Clustering::new(vec![2, 0, 1], Width::Month, &shipped).expect("valid");
17106
17107        let mut writer = Writer::create(&path, "lineitem", shipped)
17108            .expect("new file")
17109            .declare(stage_zero.clone())
17110            .expect("the columns are the table's");
17111        let column = |ty: LogicalType, values: &[Value]| {
17112            Vector::from_values(ty, values).expect("the values match the type")
17113        };
17114        writer
17115            .append(
17116                &Chunk::new(vec![
17117                    column(
17118                        LogicalType::BigInt,
17119                        &[Value::BigInt(0), Value::BigInt(1), Value::BigInt(2), Value::BigInt(3)],
17120                    ),
17121                    column(
17122                        LogicalType::Integer,
17123                        &[
17124                            Value::Integer(1),
17125                            Value::Integer(1),
17126                            Value::Integer(1),
17127                            Value::Integer(1),
17128                        ],
17129                    ),
17130                    column(
17131                        LogicalType::Date,
17132                        &[Value::Date(0), Value::Date(1), Value::Date(2), Value::Date(3)],
17133                    ),
17134                ])
17135                .expect("three columns"),
17136            )
17137            .expect("four rows");
17138        let mut writer = writer.next("nation", plain).expect("a second table");
17139        writer
17140            .append(
17141                &Chunk::new(vec![column(LogicalType::Integer, &[Value::Integer(7)])])
17142                    .expect("one column"),
17143            )
17144            .expect("one row");
17145        writer.finish().expect("commit");
17146
17147        let catalog = Catalog::open(&path).expect("reopen");
17148        let lineitem = catalog.table("lineitem").expect("the clustered table");
17149        assert_eq!(lineitem.table().clustering(), Some(&stage_zero));
17150        let nation = catalog.table("nation").expect("the plain table");
17151        assert_eq!(nation.table().clustering(), None, "nobody declared one here");
17152
17153        // And the rows are still the rows, because the section goes on the end of the directory
17154        // and the easy way to break that is to leave the cursor somewhere the next read trusts.
17155        assert_eq!(lineitem.table().rows(), 4);
17156        assert_eq!(nation.table().rows(), 1);
17157        fs::remove_file(&path).ok();
17158    }
17159
17160    /// A declaration naming a column the table does not have is refused where it is made.
17161    #[test]
17162    fn a_declaration_off_the_end_of_the_table_never_reaches_the_file() {
17163        let path = path("clustered-bad");
17164        let writer = Writer::create(&path, "items", vec![Field::new("a", LogicalType::Integer)])
17165            .expect("new file");
17166        let four =
17167            (0..4).map(|at| Field::new(format!("c{at}"), LogicalType::Integer)).collect::<Vec<_>>();
17168        let wrong = Clustering::new(vec![3], Width::Exact, &four).expect("valid against four");
17169        assert!(writer.declare(wrong).is_err(), "the table has one column, not four");
17170        fs::remove_file(&path).ok();
17171    }
17172
17173    /// The sorted order is the byte order, whatever the values do before they differ.
17174    ///
17175    /// A block handed out of the writer's lock to be encoded, and given back in whatever order the
17176    /// stripes happen to finish in, is the same block with the same signature as one encoded in
17177    /// place, and lands in the same position.
17178    #[test]
17179    fn blocks_handed_out_and_given_back_out_of_order_are_the_blocks_encoded_in_place() {
17180        let values = (0..PAYLOAD_SAMPLE_BLOCKS * TEXT_PAYLOAD_VALUES * 2 + 100)
17181            .map(|at| format!("http://example{}.test/page/{at:06}", at % 7))
17182            .collect::<Vec<_>>();
17183        let filled = || {
17184            let mut dictionary = GlobalDictionary::new();
17185            for value in &values {
17186                dictionary.code(value).expect("a code for every value");
17187            }
17188            dictionary.settle().expect("a shape");
17189            dictionary
17190        };
17191        let mut in_place = filled();
17192        in_place.finish_blocks().expect("every block encodes");
17193
17194        let mut handed = filled();
17195        let out = handed.hand_out(3);
17196        assert_eq!(out.len(), PAYLOAD_SAMPLE_BLOCKS * 2, "every sealed block goes out");
17197        assert!(handed.waiting.is_empty(), "and none is left to be encoded under the lock");
17198        for job in out.iter().rev() {
17199            assert_eq!(job.place().0, 3, "a block goes back to the column it came from");
17200            handed.take_back(job.place().1, job.encode().expect("encodes")).expect("taken back");
17201        }
17202        assert!(handed.early.is_empty(), "nothing is waiting on a gap");
17203        handed.finish_blocks().expect("the last block encodes");
17204
17205        assert_eq!(handed.blocks, in_place.blocks, "the same blocks in the same order");
17206        assert_eq!(handed.grams, in_place.grams, "with the same signatures");
17207    }
17208
17209    /// A block given back twice is a bug in whoever gave it, and is said rather than written twice.
17210    #[test]
17211    fn a_block_given_back_twice_is_refused() {
17212        let mut dictionary = GlobalDictionary::new();
17213        for at in 0..PAYLOAD_SAMPLE_BLOCKS * TEXT_PAYLOAD_VALUES {
17214            dictionary.code(&format!("value {at}")).expect("a code");
17215        }
17216        dictionary.settle().expect("a shape");
17217        let out = dictionary.hand_out(0);
17218        let last = out.last().expect("blocks went out");
17219        let at = last.place().1;
17220        dictionary.take_back(at, last.encode().expect("encodes")).expect("taken back once");
17221        assert!(dictionary.take_back(at, last.encode().expect("encodes")).is_err());
17222    }
17223
17224    /// The values here are the shape the sort is built for and the shape a comparison sort is worst
17225    /// at: a common scheme, a handful of hosts, and a path that only decides the pair thirty bytes
17226    /// in. They also cover what the bucketing has to get right at the edges, which is a value that
17227    /// has run out where another carries on, the empty value, and enough entries to take the range
17228    /// down through several passes and out the bottom into the comparison that finishes it.
17229    #[test]
17230    fn the_dictionary_order_is_the_byte_order_however_deep_the_values_agree() {
17231        let mut values = vec![String::new(), "http://".to_owned()];
17232        for host in 0..7 {
17233            for path in 0..30 {
17234                values.push(format!("http://example{host}.test/page/{path:04}/index.html"));
17235                values.push(format!("http://example{host}.test/page/{path:04}"));
17236            }
17237        }
17238        values.push("http://example0.test/page/0000/index.htmlx".to_owned());
17239
17240        let mut dictionary = GlobalDictionary::new();
17241        for value in &values {
17242            dictionary.code(value).expect("a code for every value");
17243        }
17244        dictionary.finish_blocks().expect("the last block encodes");
17245        let ranked = dictionary.ranked(None).expect("a sorted order");
17246        assert_eq!(ranked.len(), values.len(), "one entry a distinct value");
17247
17248        let spellings = dictionary_values(&dictionary);
17249        let seen = ranked
17250            .iter()
17251            .map(|&(_, code)| {
17252                String::from_utf8(spellings[code as usize].clone()).expect("text in, text out")
17253            })
17254            .collect::<Vec<_>>();
17255        let mut wanted = values.clone();
17256        wanted.sort_unstable();
17257        assert_eq!(seen, wanted, "the order is the order the bytes give");
17258
17259        for &(carried, code) in &ranked {
17260            let value = &spellings[code as usize];
17261            assert_eq!(carried, head(value), "the head belongs to the value it is filed with");
17262        }
17263    }
17264
17265    /// Picking the commonest entries leaves exactly what sorting all of them and cutting left.
17266    ///
17267    /// The counts here are deliberately full of ties, including a tie that straddles the cut, which
17268    /// is where a partition and a sort can disagree if the comparison they are given is not total.
17269    #[test]
17270    fn the_commonest_entries_are_the_ones_a_full_sort_would_have_kept() {
17271        let entry =
17272            |value: u32, count: u64| FrequencyEntry { value: FrequencyValue::Code(value), count };
17273        let mut all = (0..FREQUENCY_ENTRIES as u32 * 3)
17274            .map(|code| entry(code, u64::from(code % 7) + 1))
17275            .collect::<Vec<_>>();
17276        all.push(FrequencyEntry { value: FrequencyValue::Null, count: 4 });
17277
17278        let mut sorted = all.clone();
17279        sorted.sort_unstable_by(|left, right| {
17280            right.count.cmp(&left.count).then_with(|| frequency_order(left.value, right.value))
17281        });
17282        let wanted_omitted = sorted[FREQUENCY_ENTRIES].count;
17283        sorted.truncate(FREQUENCY_ENTRIES);
17284
17285        let mut picked = all.clone();
17286        let omitted = keep_most_frequent(&mut picked);
17287        assert_eq!(omitted, wanted_omitted, "the largest count that did not make the cut");
17288        assert_eq!(picked.len(), FREQUENCY_ENTRIES, "the cut is where it says it is");
17289        assert!(
17290            picked
17291                .iter()
17292                .zip(&sorted)
17293                .all(|(one, two)| one.value == two.value && one.count == two.count),
17294            "the same entries in the same order"
17295        );
17296
17297        let mut short = all[..FREQUENCY_ENTRIES - 1].to_vec();
17298        let omitted = keep_most_frequent(&mut short);
17299        assert_eq!(omitted, 0, "nothing is omitted when everything fits");
17300        assert!(short.windows(2).all(|pair| pair[0].count >= pair[1].count), "still in order");
17301    }
17302
17303    /// A dictionary too small to bucket, and one with nothing in it, come back in order too.
17304    #[test]
17305    fn a_short_dictionary_sorts_without_a_bucketing_pass() {
17306        let empty = GlobalDictionary::new();
17307        assert!(empty.ranked(None).expect("an empty order").is_empty(), "nothing in, nothing out");
17308
17309        let mut dictionary = GlobalDictionary::new();
17310        for value in ["pear", "apple", "", "apples", "app"] {
17311            dictionary.code(value).expect("a code for every value");
17312        }
17313        dictionary.finish_blocks().expect("the one block encodes");
17314        let spellings = dictionary_values(&dictionary);
17315        let seen = dictionary
17316            .ranked(None)
17317            .expect("a sorted order")
17318            .iter()
17319            .map(|&(_, code)| spellings[code as usize].clone())
17320            .collect::<Vec<_>>();
17321        let wanted: Vec<Vec<u8>> =
17322            [&b""[..], b"app", b"apple", b"apples", b"pear"].iter().map(|v| v.to_vec()).collect();
17323        assert_eq!(seen, wanted, "shorter first where one runs out inside another");
17324    }
17325
17326    /// A demoted dictionary gives back what it kept for looking values up, the load profile is told,
17327    /// and it refuses any value after that.
17328    #[test]
17329    fn a_demoted_dictionary_holds_less_and_takes_no_more_values() {
17330        let profile = LoadProfile::begin("demoted");
17331        let mut dictionary = GlobalDictionary::new();
17332        for value in 0..50_000 {
17333            dictionary.code(&format!("https://example.com/page/{value}")).expect("a code");
17334        }
17335        let (_, grown) = dictionary.recharge(Some(&profile));
17336        assert_eq!(profile.held(), grown, "the profile holds what the dictionary does");
17337
17338        dictionary.demote();
17339        let (before, after) = dictionary.recharge(Some(&profile));
17340        assert_eq!(before, grown);
17341        // What stays is the ends, the counts and the blocks not yet written, which a load writes
17342        // as it goes, so here the drop is the hash tables and the check hashes.
17343        assert!(after < grown - grown / 4, "the lookup is let go of: {after} of {grown}");
17344        assert_eq!(profile.held(), after, "the profile was told about the drop");
17345        assert!(dictionary.code("one more").is_err(), "a demoted dictionary takes no values");
17346
17347        dictionary.demote();
17348        assert_eq!(
17349            dictionary.recharge(Some(&profile)),
17350            (after, after),
17351            "demoting twice is a no-op"
17352        );
17353        assert_eq!(dictionary.values(), 50_000, "the values coded before stay");
17354    }
17355}