Skip to main content

rudb_native/
lib.rs

1//! Rudb's single-file columnar snapshot format.
2//!
3//! A committed directory names independently readable column pages. It has two levels: a catalog
4//! directory naming every table in the file, which is what a footer slot points at and what opening
5//! a database reads, and one directory per table under it holding that table's stripes, pages and
6//! statistics. One slot write publishes all of them, so a commit is atomic across tables.
7//!
8//! This version handles scalar columns; the file header has two generation slots so an unfinished
9//! replacement directory cannot hide the last complete one. See
10//! `spec/storage-v3/12-many-tables-in-one-file.md`.
11//!
12//! # Parts and stripes
13//!
14//! A part is one appended chunk, which is a thousand rows, and it is the unit a scan decodes and
15//! hands to the pipeline. A stripe is sixty four parts, and it is the unit the directory describes
16//! and the unit the file is laid out in: one page per column per stripe, holding that column's
17//! sixty four part payloads end to end.
18//!
19//! The two are separate because they are sized by different pressures. A part wants to be small
20//! because it is a vector and vectors live in cache. A stripe wants to be large because everything
21//! the directory holds is per stripe and the directory is one buffer that has to be read and
22//! decoded before a single row can be answered. A hundred million rows of the hundred and five
23//! column ClickBench table is ninety seven thousand parts, and a directory with a page entry and a
24//! pair of bounds per part per column is several hundred megabytes, which is what made that load
25//! fail before this split existed. Sixty four parts to a stripe divides that by sixty four.
26//!
27//! Where the parts of a page start is not in the directory either, for the same reason. Each
28//! stripe writes one index page holding a length and a checksum per part per column, and a reader
29//! preads the sixty four entries belonging to the column it wants. A scan reads the whole column
30//! page once and slices it; a sparse row fetch reads the index entries and then only the part it
31//! needs.
32
33#![forbid(unsafe_code)]
34
35use std::cmp::Ordering;
36use std::collections::{HashMap, VecDeque};
37use std::fs::{File, OpenOptions};
38use std::io::{Read, Seek, SeekFrom};
39use std::mem::{size_of, size_of_val};
40use std::path::Path;
41use std::slice;
42use std::sync::atomic::{AtomicUsize, Ordering as Atomic};
43use std::sync::{Arc, Mutex, OnceLock};
44
45use rudb_common::bounds::{self, Bound, Op, scaled_as};
46use rudb_common::{Clustering, Error, Field, LogicalType, PhysicalType, Result, Value, Width};
47use rudb_encoding::{bitpack, chooser, integer, string};
48use rudb_storage::sieve::Sieve;
49use rudb_storage::{Probe, Range, Zone};
50use rudb_vector::string::StringColumn;
51use rudb_vector::validity::Validity;
52use rudb_vector::{Buffer, Chunk, Data, Packed, TextSource, Vector, search_below};
53
54pub mod graph;
55pub mod section;
56pub mod stats;
57mod zones;
58
59pub use section::Section;
60pub use zones::{Common, Stripes, distincts};
61
62const MAGIC: &[u8; 8] = b"RUDBNV10";
63const DIRECTORY: &[u8; 8] = b"RUDBDI10";
64const CATALOG: &[u8; 8] = b"RUDBCA10";
65const FORMAT: u32 = 24;
66
67/// Formats this build can open.
68///
69/// More than one, for the first time, and the reason is spec/graph/10-milestones.md's G1 exit
70/// criterion: a build with the section table in it has to open a file written before the section
71/// table existed, unchanged and without a rewrite. Formats 22 and 23 are those files, and both read
72/// as a table with an empty section table, which is exactly what section 3.1 says a table with no
73/// graph sections is.
74///
75/// Both of the older two are readable for the same reason. What took the format from 22 to 23 was
76/// tags for fourteen more column types, and a file written before that has none of them in it, so
77/// nothing in an older file is a tag this build cannot read. What takes it from 23 to 24 is the
78/// section table, which a file written before it simply does not have.
79///
80/// This is not a general compatibility promise. Three formats are readable because there was a
81/// specific reason for each, and the list shrinks again the moment the older ones stop being worth
82/// carrying.
83const READABLE: &[u32] = &[22, 23, FORMAT];
84
85const HEADER: u64 = 80;
86const SLOT_BYTES: usize = 28;
87const MAX_PAGE: usize = 256 * 1024 * 1024;
88const MAX_DIRECTORY: usize = 128 * 1024 * 1024;
89const FREQUENCIES: &[u8; 8] = b"RUDBFQ2\0";
90/// The clustering declaration, written after the frequencies and only when there is one.
91///
92/// No format bump for this, which is the convention the frequency section set in #728: a new
93/// optional trailing section with its own magic leaves every file that does not use it byte for
94/// byte what it was, and the version is bumped for a change to a layout that already exists, as
95/// #1029 did. A file with no declaration is the same bytes this build wrote yesterday.
96const CLUSTERING: &[u8; 8] = b"RUDBCL1\0";
97/// The graph section table, written after the clustering declaration and written even when empty.
98///
99/// Same convention and the same reason as the block above it, with one difference: this one is
100/// always there, so a file written by this build says which sections it has rather than leaving a
101/// reader to infer it from where the bytes ran out. Section 3.1 of the graph spec is what makes
102/// that safe to add without a format bump, because a table with no sections answers every query
103/// the way it did before, only without the graph path.
104const SECTIONS: &[u8; 8] = b"RUDBSE1\0";
105
106/// The most sections one table's directory may name.
107///
108/// A relationship contributes at most three sections, so this bounds a table at a few thousand
109/// relationships, which is far past anything a schema has. The bound is here so that a torn
110/// directory naming four billion of them is refused at decode rather than turned into an
111/// allocation, the same reason the extent count has one.
112const MAX_SECTIONS: usize = 4096;
113const FREQUENCY_CANDIDATES: usize = 32_768;
114const FREQUENCY_ENTRIES: usize = 512;
115const FREQUENCY_BUILD_RANK: usize = 10;
116const FREQUENCY_ORDINALS: usize = 65_536;
117/// The most threads the two per column passes at the end of a commit are spread over.
118///
119/// A table like `hits` has ninety numeric columns, so on a machine with more cores than this the
120/// cap is what decides how long the frequencies take rather than the columns are. It is here at all
121/// because each worker holds a candidate table and a decoded part, and a hundred of those at once
122/// on a narrow machine would be worse than waiting.
123const MAX_FREQUENCY_WORKERS: usize = 32;
124
125/// The most threads one stripe's encode is spread over.
126///
127/// Higher than the frequency cap because this is the load itself rather than a pass at the end of
128/// it, and the work is one column of sixty four parts, which is large enough that a thread that
129/// takes one is not a thread that was started for nothing. A machine with more cores than this has
130/// the rest of them on the Parquet read, which is still one thread and is the other half of #808.
131const MAX_ENCODE_WORKERS: usize = 32;
132
133/// The most bytes one column of one part may spend on a membership sieve.
134///
135/// A part is a thousand rows, so a filter sized for every one of them being distinct is about
136/// thirteen hundred bytes and this never binds in practice. It is here so that a part that somehow
137/// arrives much wider than a vector cannot put an unbounded index in the file. What does bind is the
138/// rule in `encode_column` that a sieve may not be as large as the part it indexes, which is a cap
139/// per column rather than one number for the whole file.
140const SIEVE_BUDGET: usize = 8 * 1024;
141
142/// The most bytes one end of a per part range may spend on a string.
143///
144/// A bound is allowed to be wider than the truth and never narrower, so a long string is cut down to
145/// this many bytes for the low end and cut down and then stepped up for the high end. The reason for
146/// a cap at all is that there are nine hundred and seventy four parts of a hundred and five columns
147/// in a million rows of ClickBench and `URL` runs to hundreds of bytes, so keeping every end whole
148/// would put more in the directory than the skipping is worth. Twenty four bytes is past the point
149/// where two URLs of the same site still look alike.
150const PART_BOUND_BYTES: usize = 24;
151
152fn io(error: std::io::Error) -> Error {
153    Error::io(error.to_string())
154}
155
156fn invalid(message: &str) -> Error {
157    Error::invalid_input(format!("invalid rudb native file: {message}"))
158}
159
160/// Adds a sequence of byte counts without an overflow the caller has to think about.
161fn sum(counts: impl Iterator<Item = u64>) -> u64 {
162    counts.fold(0, u64::saturating_add)
163}
164
165/// One column's span out of a per column list, or zero when the list is shorter than the column.
166fn span_bytes(spans: &[Span], at: usize) -> u64 {
167    spans.get(at).map_or(0, |span| u64::from(span.length))
168}
169
170/// One column's page out of a per column list, or zero when that column has no page at all.
171fn page_bytes(pages: &[Option<Page>], at: usize) -> u64 {
172    pages.get(at).and_then(Option::as_ref).map_or(0, Page::bytes)
173}
174
175/// The xxHash64 of `bytes`, which is what every span this format stores is checked against.
176///
177/// It walks the input as chunks rather than as offsets into it, and that is the only thing about it
178/// worth a comment. The offset form reads `bytes[at..at + 8]`, and neither the slicing nor the
179/// `try_into` behind it can be proved in range by a compiler that does not know where `at` stopped,
180/// so each of the four lanes paid for a bounds check and a length check on every thirty two bytes.
181/// A chunk carries its own length, so both fold away and the loop is the multiplies and rotates it
182/// was meant to be. That loop runs over every byte of every span a query reads, which on ClickBench
183/// 8 is about five percent of the query.
184fn checksum(bytes: &[u8]) -> u64 {
185    const P1: u64 = 11_400_714_785_074_694_791;
186    const P2: u64 = 14_029_467_366_897_019_727;
187    const P3: u64 = 1_609_587_929_392_839_161;
188    const P4: u64 = 9_650_029_242_287_828_579;
189    const P5: u64 = 2_870_177_450_012_600_261;
190    let round = |state: u64, word: u64| {
191        state.wrapping_add(word.wrapping_mul(P2)).rotate_left(31).wrapping_mul(P1)
192    };
193    let merge = |state: u64, lane: u64| (state ^ round(0, lane)).wrapping_mul(P1).wrapping_add(P4);
194    let word = |chunk: &[u8]| u64::from_le_bytes(chunk.try_into().expect("eight checksum bytes"));
195
196    // Asked for before the loop rather than after it, because a `ChunksExact` settles what it
197    // cannot divide when it is built and hands back the same tail whether it has been walked or not.
198    let mut blocks = bytes.chunks_exact(32);
199    let mut rest = blocks.remainder();
200    let mut hash = if bytes.len() >= 32 {
201        let mut one = P1.wrapping_add(P2);
202        let mut two = P2;
203        let mut three = 0;
204        let mut four = 0_u64.wrapping_sub(P1);
205        for block in blocks.by_ref() {
206            one = round(one, word(&block[..8]));
207            two = round(two, word(&block[8..16]));
208            three = round(three, word(&block[16..24]));
209            four = round(four, word(&block[24..]));
210        }
211        let combined = one
212            .rotate_left(1)
213            .wrapping_add(two.rotate_left(7))
214            .wrapping_add(three.rotate_left(12))
215            .wrapping_add(four.rotate_left(18));
216        merge(merge(merge(merge(combined, one), two), three), four)
217    } else {
218        P5
219    };
220    hash = hash.wrapping_add(bytes.len() as u64);
221    let mut words = rest.chunks_exact(8);
222    for chunk in words.by_ref() {
223        hash ^= round(0, word(chunk));
224        hash = hash.rotate_left(27).wrapping_mul(P1).wrapping_add(P4);
225    }
226    rest = words.remainder();
227    if rest.len() >= 4 {
228        let (head, tail) = rest.split_at(4);
229        let quarter = u32::from_le_bytes(head.try_into().expect("four checksum bytes"));
230        hash ^= u64::from(quarter).wrapping_mul(P1);
231        hash = hash.rotate_left(23).wrapping_mul(P2).wrapping_add(P3);
232        rest = tail;
233    }
234    for &byte in rest {
235        hash ^= u64::from(byte).wrapping_mul(P5);
236        hash = hash.rotate_left(11).wrapping_mul(P1);
237    }
238    hash ^= hash >> 33;
239    hash = hash.wrapping_mul(P2);
240    hash ^= hash >> 29;
241    hash = hash.wrapping_mul(P3);
242    hash ^ (hash >> 32)
243}
244
245#[derive(Debug, Clone, Copy)]
246struct Slot {
247    offset: u64,
248    length: u32,
249    generation: u64,
250    hash: u64,
251}
252
253impl Slot {
254    fn bytes(self) -> [u8; SLOT_BYTES] {
255        let mut result = [0; SLOT_BYTES];
256        result[..8].copy_from_slice(&self.offset.to_le_bytes());
257        result[8..12].copy_from_slice(&self.length.to_le_bytes());
258        result[12..20].copy_from_slice(&self.generation.to_le_bytes());
259        result[20..28].copy_from_slice(&self.hash.to_le_bytes());
260        result
261    }
262
263    fn read(bytes: &[u8]) -> Self {
264        Self {
265            offset: u64::from_le_bytes(bytes[..8].try_into().expect("eight bytes")),
266            length: u32::from_le_bytes(bytes[8..12].try_into().expect("four bytes")),
267            generation: u64::from_le_bytes(bytes[12..20].try_into().expect("eight bytes")),
268            hash: u64::from_le_bytes(bytes[20..28].try_into().expect("eight bytes")),
269        }
270    }
271}
272
273#[derive(Debug, Clone, Copy)]
274struct Page {
275    offset: u64,
276    length: u32,
277    hash: u64,
278}
279
280impl Page {
281    /// How much of the file this page takes, for [`Reader::layout`].
282    fn bytes(&self) -> u64 {
283        u64::from(self.length)
284    }
285}
286
287#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
288enum FrequencyValue {
289    Null,
290    Integer(i128),
291    Code(u32),
292}
293
294#[derive(Debug, Clone)]
295struct FrequencyEntry {
296    value: FrequencyValue,
297    count: u64,
298}
299
300/// Exact leading frequencies for one column.
301///
302/// Values outside `entries` occur at most `omitted_max` times. This lets a count-descending TopN
303/// use the synopsis only when its last winner is strictly above every omitted value.
304#[derive(Debug, Clone)]
305struct FrequencySummary {
306    entries: Vec<FrequencyEntry>,
307    omitted_max: u64,
308    ordinals: Vec<u64>,
309}
310
311/// The values one column's frequency synopsis lists, with a bound on everything it left out.
312///
313/// What [`Reader::frequency_prefix`] answers. The counts are exact, and `omitted_max` is how many
314/// rows any value not in the list can hold, which is zero when nothing was left out at all.
315#[derive(Debug, Clone)]
316pub struct FrequencyPrefix {
317    /// Every value the synopsis lists, with the number of rows holding it, count descending.
318    pub entries: Vec<(Value, u64)>,
319    /// How many rows the most common value outside the list holds, and zero for a complete list.
320    pub omitted_max: u64,
321}
322
323/// Sparse row ordinals covered by a numeric frequency candidate set.
324#[derive(Debug, Clone, PartialEq, Eq)]
325pub struct FrequencyOccurrences {
326    /// Upper bound for the frequency of every value absent from the fetched rows.
327    pub omitted_max: u64,
328    /// Table-wide row ordinals in ascending order.
329    pub ordinals: Vec<u64>,
330}
331
332/// Where one column's page for one stripe sits in the file.
333///
334/// A column page has no checksum of its own because every part inside it carries one, and the
335/// stripe's index page holds those. Checking a part on the way out of the page covers exactly the
336/// bytes a reader is about to decode, and covers them once whether the reader took the whole page
337/// or pulled one part out of the middle of it.
338#[derive(Debug, Clone, Copy, Default)]
339struct Span {
340    offset: u64,
341    length: u32,
342}
343
344/// One independently readable stripe of a table.
345#[derive(Debug, Clone)]
346pub struct Stripe {
347    rows: usize,
348    /// Rows in each part, in source order. Kept in the directory so that mapping a row ordinal to a
349    /// part, which every sparse fetch does, never reads the file.
350    parts: Vec<u32>,
351    /// The index page: one section per column, holding a length and a checksum for every part and
352    /// then a checksum of the section itself, so that a reader can pread one column's section and
353    /// still know it is intact.
354    index: Span,
355    pages: Vec<Span>,
356    memberships: Vec<Option<Page>>,
357    /// One page per column holding the membership sieve of every part of the stripe, for the
358    /// columns that have one. A column whose parts all declined a sieve has no page at all.
359    sieves: Vec<Option<Page>>,
360    /// One page per column holding the two ends and the null count of every part of the stripe.
361    ///
362    /// The stripe's own `zone` below covers sixty four times as many rows, and on a column that is
363    /// not the one the rows are ordered by that is the difference between skipping half the file and
364    /// skipping all but three percent of it. On ClickBench 24 the cutoff the answer settles at
365    /// leaves eight stripes of sixteen alive and thirty parts of nine hundred and seventy four.
366    ///
367    /// A page per column rather than one page for the stripe, so that a query that compares one
368    /// column reads the ends of that column and not of the hundred and four beside it. Read lazily
369    /// for the same reason, like the sieves.
370    part_ranges: Vec<Option<Page>>,
371    zone: Zone,
372}
373
374impl Stripe {
375    /// Number of rows in this stripe.
376    #[must_use]
377    pub fn rows(&self) -> usize {
378        self.rows
379    }
380
381    /// Number of parts in this stripe.
382    #[must_use]
383    pub fn parts(&self) -> usize {
384        self.parts.len()
385    }
386
387    /// The two ends and the null count of every column over the whole stripe.
388    ///
389    /// In the directory and so in memory, which is what makes it the one a planner can ask. The
390    /// finer ones are a page per column per stripe in the file, read by [`Reader::skips`] when a
391    /// scan wants to know which parts to open.
392    #[must_use]
393    pub fn zone(&self) -> &Zone {
394        &self.zone
395    }
396}
397
398/// The committed table directory.
399#[derive(Debug, Clone)]
400pub struct Table {
401    name: String,
402    fields: Vec<Field>,
403    stripes: Vec<Stripe>,
404    rows: usize,
405    dictionaries: Vec<Option<Page>>,
406    frequencies: Vec<Option<FrequencySummary>>,
407    /// How many distinct values each column holds, for the columns that know.
408    ///
409    /// A dictionary entry is made the first time a value is seen and nothing ever removes one, so
410    /// the size of the dictionary is the number of distinct values in the column. That is the whole
411    /// story for a column with no null in it, and the wrong number by one for a column with a null
412    /// in it, because a null row is written as the code for the empty string and makes an entry the
413    /// dictionary would not otherwise have. The writer knows which case it is, since it counts the
414    /// non-null rows that use each code while it builds the frequency summary, and the reader cannot
415    /// work it out from the dictionary alone. So the writer settles it here.
416    distincts: Vec<Option<u64>>,
417    /// The order the rows of this table are meant to be stored in, if anybody declared one.
418    ///
419    /// A declaration and not a measurement. Nothing here checks that the stripes actually arrived
420    /// in this order, and the reason it is worth storing anyway is that the order is the only thing
421    /// about a table that a rewrite destroys without anybody noticing. The fragment ranges prune on
422    /// whatever order the rows came in, so a table loaded sorted prunes and the same table after a
423    /// checkpoint that did not know to keep the order quietly stops pruning and nothing says why.
424    clustering: Option<Clustering>,
425    /// The file generation of the commit that last wrote this table's column pages.
426    ///
427    /// This is what spec/graph/03-the-file-format.md section 3.2 calls the table generation, and
428    /// the definition is deliberately about the pages rather than about the directory. A graph
429    /// section is a restatement of a column in terms of row ids, so what invalidates one is the
430    /// rows being renumbered, and nothing else. Adding a second table to the file, or attaching a
431    /// section to this one, commits a new file generation without touching a single row of this
432    /// table, and a definition that moved with those would declare every section in the file stale
433    /// for no reason.
434    ///
435    /// Zero on a table written before format 23, where nothing recorded it. Real generations start
436    /// at one, so zero can never match a section's stamp, and a table from format 22 has no
437    /// sections for it to match anyway.
438    generation: u64,
439    /// The graph sections this table carries, per spec/graph/03-the-file-format.md section 3.2.
440    ///
441    /// Empty for every table written before the section table existed, and empty is not a
442    /// degraded state: section 3.1 says deleting every graph section from a file changes no answer,
443    /// only the time, so a table with none here answers every query the same way and slower. That
444    /// is what lets this field arrive without a migration.
445    sections: Vec<Section>,
446}
447
448impl Table {
449    /// The SQL table name held by this snapshot.
450    #[must_use]
451    pub fn name(&self) -> &str {
452        &self.name
453    }
454
455    /// Columns in their SQL order.
456    #[must_use]
457    pub fn fields(&self) -> &[Field] {
458        &self.fields
459    }
460
461    /// Committed row count.
462    #[must_use]
463    pub fn rows(&self) -> usize {
464        self.rows
465    }
466
467    /// Independently readable stripes.
468    #[must_use]
469    pub fn stripes(&self) -> &[Stripe] {
470        &self.stripes
471    }
472
473    /// The order the rows are meant to be stored in, if this table was declared with one.
474    #[must_use]
475    pub fn clustering(&self) -> Option<&Clustering> {
476        self.clustering.as_ref()
477    }
478
479    /// The generation every section of this table is judged against.
480    ///
481    /// See the field. A caller deciding whether to read a section asks [`Section::usable`] with
482    /// this.
483    #[must_use]
484    pub fn generation(&self) -> u64 {
485        self.generation
486    }
487
488    /// Every graph section this table names, including the kinds this build does not know.
489    ///
490    /// Including them is the point. A caller that wants only the ones it can use asks
491    /// [`Section::usable`], and a caller rewriting the directory carries the rest through, so a
492    /// file opened by an older build and written again does not silently lose a section that build
493    /// had no name for.
494    #[must_use]
495    pub fn sections(&self) -> &[Section] {
496        &self.sections
497    }
498}
499
500/// One table's line in the catalog directory.
501///
502/// The small level of the two. It holds what opening a database needs and nothing else: the name to
503/// bind, the shape to plan against, the row count, and where the table's own directory sits. A file
504/// of eight tables is eight of these, and reading them costs the same whether the tables hold a
505/// thousand rows or a billion.
506///
507/// The name, the fields and the row count are repeated here rather than pointed at inside the table
508/// directory, which is the entire point of having two levels. A catalog that pointed at them would
509/// have to read every table directory at open to answer what tables there are, which is the cost
510/// this level exists to avoid.
511#[derive(Debug, Clone)]
512struct Entry {
513    name: String,
514    fields: Vec<Field>,
515    rows: usize,
516    /// Where this table's own directory sits, with the checksum it was committed under.
517    directory: Page,
518}
519
520/// Where one column's bytes went, taken from the directory rather than by reading pages.
521#[derive(Debug, Clone)]
522pub struct ColumnLayout {
523    /// The column's name, so a report does not have to carry the field list beside this.
524    pub name: String,
525    /// The type, spelled the way the catalog spells it.
526    pub kind: String,
527    /// Every stripe's page of this column added up, which is the encoded data itself.
528    pub pages: u64,
529    /// Every stripe's exact code membership page for this column.
530    pub memberships: u64,
531    /// Every stripe's membership sieve page for this column.
532    pub sieves: u64,
533    /// Every stripe's per part range page for this column.
534    pub part_ranges: u64,
535    /// The table wide dictionary of this column, if it has one.
536    pub dictionary: u64,
537}
538
539impl ColumnLayout {
540    /// Everything this column costs, which is what the file would lose if the column went.
541    #[must_use]
542    pub fn total(&self) -> u64 {
543        self.pages
544            .saturating_add(self.memberships)
545            .saturating_add(self.sieves)
546            .saturating_add(self.part_ranges)
547            .saturating_add(self.dictionary)
548    }
549}
550
551/// Where a whole file's bytes went.
552///
553/// Every number here comes out of the committed directory, so taking it costs one directory read
554/// however large the file is. That is the point: a 45 GB table has to be able to say where it went
555/// without being read, or nobody will ask.
556///
557/// The parts that are not a column are kept apart rather than shared out over the columns. The
558/// stripe index page holds a section per column and could be split, and the directory and the
559/// header cannot be, so splitting one of the three and not the others would read as if the columns
560/// accounted for everything. They do not, and the gap is the thing worth looking at.
561#[derive(Debug, Clone)]
562pub struct Layout {
563    /// The size of the file on disk.
564    pub file: u64,
565    /// Committed rows.
566    pub rows: usize,
567    /// Committed stripes.
568    pub stripes: usize,
569    /// Committed parts, which is how many chunks a scan reads.
570    pub parts: usize,
571    /// One entry per column, in the table's column order.
572    pub columns: Vec<ColumnLayout>,
573    /// Every stripe's index page, which carries a length and a checksum for every part of every
574    /// column and is charged per stripe rather than per column.
575    pub indexes: u64,
576    /// The committed directory itself, the one that was read to build this.
577    pub directory: u64,
578    /// The fixed header, which holds the magic, the format and the two directory slots.
579    pub header: u64,
580}
581
582impl Layout {
583    /// Everything the columns cost together.
584    #[must_use]
585    pub fn columns_total(&self) -> u64 {
586        self.columns.iter().map(ColumnLayout::total).fold(0, u64::saturating_add)
587    }
588
589    /// What the file holds that this does not account for.
590    ///
591    /// A committed file is written once and never rewritten in place, so an earlier directory and
592    /// the pages of an earlier snapshot are still in it. That is the honest place for them: they
593    /// are bytes on disk that no column owns.
594    #[must_use]
595    pub fn unaccounted(&self) -> u64 {
596        self.file
597            .saturating_sub(self.columns_total())
598            .saturating_sub(self.indexes)
599            .saturating_sub(self.directory)
600            .saturating_sub(self.header)
601    }
602}
603
604/// How one part of one column is stored, which is one row of `pragma_storage_info`.
605///
606/// Everything here is read off the file rather than worked out from the schema, because the whole
607/// question this answers is what the encoder chose, and the encoder chooses per part. Two files
608/// holding the same rows in a different order give different answers and that difference is the
609/// reason to ask.
610///
611/// The encoding costs a read of the column's page, so this is not free the way [`Layout`] is. It is
612/// one read per column per stripe rather than one per part, because a part is a few kilobytes out
613/// of a page that is a quarter of a megabyte.
614#[derive(Debug, Clone)]
615pub struct StoredPart {
616    /// Which stripe the part belongs to.
617    pub stripe: usize,
618    /// Which part of that stripe it is, counting from zero inside the stripe.
619    pub part: usize,
620    /// The table wide row number the part starts at.
621    pub row: usize,
622    /// How many rows it holds.
623    pub rows: usize,
624    /// What the encoder made of it, as a line of text like `DICT(PACKED, PACKED)`.
625    pub encoding: String,
626    /// The stored bytes of the part, which is what it costs in the file.
627    pub bytes: u64,
628    /// Where in the file the column page holding this part starts.
629    pub page: u64,
630    /// Where in that page the part starts.
631    pub offset: u64,
632    /// The smallest value the part holds, when the stored ranges say.
633    pub low: Option<Value>,
634    /// The largest, same.
635    pub high: Option<Value>,
636    /// How many of its rows are null, when the stored ranges say.
637    pub nulls: Option<usize>,
638}
639
640/// Appends pages and commits a new directory for one table.
641#[derive(Debug)]
642struct GlobalDictionary {
643    primary: HashMap<u64, u32>,
644    collisions: HashMap<u64, Vec<u32>>,
645    offsets: Vec<u32>,
646    payload: Vec<u8>,
647    counts: Vec<u64>,
648    nulls: u64,
649}
650
651impl GlobalDictionary {
652    fn new() -> Self {
653        Self {
654            primary: HashMap::new(),
655            collisions: HashMap::new(),
656            offsets: vec![0],
657            payload: Vec::new(),
658            counts: Vec::new(),
659            nulls: 0,
660        }
661    }
662
663    fn bytes(&self, code: u32) -> Option<&[u8]> {
664        let start = *self.offsets.get(code as usize)? as usize;
665        let end = *self.offsets.get(code as usize + 1)? as usize;
666        self.payload.get(start..end)
667    }
668
669    fn code(&mut self, text: &str) -> Result<u32> {
670        let hash = checksum(text.as_bytes());
671        if let Some(&code) = self.primary.get(&hash) {
672            if self.bytes(code) == Some(text.as_bytes()) {
673                return Ok(code);
674            }
675            if let Some(codes) = self.collisions.get(&hash) {
676                if let Some(code) =
677                    codes.iter().copied().find(|&code| self.bytes(code) == Some(text.as_bytes()))
678                {
679                    return Ok(code);
680                }
681            }
682            let code = self.insert(text)?;
683            self.collisions.entry(hash).or_default().push(code);
684            return Ok(code);
685        }
686        let code = self.insert(text)?;
687        self.primary.insert(hash, code);
688        Ok(code)
689    }
690
691    fn insert(&mut self, text: &str) -> Result<u32> {
692        let code = u32::try_from(self.offsets.len() - 1)
693            .map_err(|_| invalid("global dictionary has too many values"))?;
694        self.payload.extend_from_slice(text.as_bytes());
695        self.offsets.push(
696            u32::try_from(self.payload.len())
697                .map_err(|_| invalid("global dictionary payload exceeds 4 GiB"))?,
698        );
699        self.counts.push(0);
700        Ok(code)
701    }
702
703    /// This dictionary's values in sorted order, each as the first eight bytes of the value and the
704    /// code that holds it, so entry `rank` describes the value that sits at `rank` when the values
705    /// are sorted by their bytes.
706    ///
707    /// Codes themselves stay in first appearance order, which is what lets the writer hand one out
708    /// the moment it sees a value rather than waiting for the last stripe, and which also keeps a
709    /// stripe's codes close together because the data is clustered. This is what puts the values
710    /// back in order for anything that needs it, and it is separate from the codes so that getting
711    /// it costs a sort of the distinct values at the end rather than a rewrite of every code page.
712    ///
713    /// The sort compares the first eight bytes as one integer before it compares the values, which
714    /// settles almost every pair without touching the payload. Padding with zero on the right is
715    /// order preserving for byte strings, because a shorter value differs from a longer one that
716    /// starts the same way at a position where the shorter one has run out, and zero is below every
717    /// byte that could be there. A pair the head cannot settle falls through to the bytes.
718    ///
719    /// The heads are kept rather than thrown away once the sort is over, because a reader searching
720    /// this order wants exactly the same comparison and for exactly the same reason. Eight bytes an
721    /// entry of file is what buys a binary search that reads no values at all in the ordinary case.
722    fn ranked(&self) -> Vec<(u64, u32)> {
723        let count = self.offsets.len() - 1;
724        let mut ranked = (0..count)
725            .map(|code| {
726                let code = code as u32;
727                (head(self.bytes(code).unwrap_or_default()), code)
728            })
729            .collect::<Vec<_>>();
730        ranked.sort_unstable_by(|left, right| {
731            left.0.cmp(&right.0).then_with(|| self.bytes(left.1).cmp(&self.bytes(right.1)))
732        });
733        ranked
734    }
735
736    fn observe(&mut self, code: u32, null: bool) -> Result<()> {
737        if null {
738            self.nulls = self.nulls.saturating_add(1);
739            return Ok(());
740        }
741        let count = self
742            .counts
743            .get_mut(code as usize)
744            .ok_or_else(|| invalid("global dictionary count code is out of range"))?;
745        *count = count.saturating_add(1);
746        Ok(())
747    }
748}
749
750/// Appends pages and commits a new directory.
751///
752/// One writer covers a whole file rather than one table. [`Writer::next`] closes the table it is on
753/// and opens another over the same file, and [`Writer::finish`] commits every table it has closed in
754/// one generation. That is what makes a checkpoint atomic across tables: there is one slot write at
755/// the end of it and a reader sees every table at the generation before it or every table at the
756/// generation after it.
757#[derive(Debug)]
758pub struct Writer {
759    file: File,
760    /// Where the next write goes, counted here rather than asked of the file.
761    ///
762    /// The file's own cursor is not ours. Building the numeric frequencies reads pages back through
763    /// [`read_at`], and a positional read is only positional about where it reads from: `pread`
764    /// leaves the cursor alone, and the call Windows has for it moves the cursor to the end of what
765    /// it read. A writer that asked the file where it was would then write the directory over a
766    /// page it had already written, which is what it did.
767    at: u64,
768    table: Table,
769    generation: u64,
770    /// The first and the last source position in every stripe, in the order the stripes were
771    /// written.
772    order: Vec<((u64, u64), (u64, u64))>,
773    next_order: u64,
774    dictionaries: Vec<Option<GlobalDictionary>>,
775    pending: Vec<PendingChunk>,
776    /// The tables already closed in this generation, in the order they were written.
777    closed: Vec<Entry>,
778}
779
780/// A chunk that has arrived and is waiting for the rest of its stripe.
781///
782/// The rows are kept rather than the pages they encode to, which is the whole of #808's first half.
783/// Encoding on arrival put every column of every part on the thread that called `append_at`, and
784/// that thread is the only one the load has. Encoding at the flush instead means a stripe's worth
785/// of work is on the table at once, and a stripe splits by column into a hundred and five pieces
786/// that share nothing.
787#[derive(Debug)]
788struct PendingChunk {
789    order: (u64, u64),
790    chunk: Chunk,
791}
792
793/// One column's share of a stripe, which is what one encode worker produces.
794///
795/// Indexed by part, so a stripe is a column of these and the write loop reads down one of them.
796/// That is also the order the loop wanted: `flush_pending` walks a column at a time and lays its
797/// parts next to each other, and it used to reach across a row of parts to do it.
798#[derive(Debug)]
799struct ColumnStripe {
800    pages: Vec<Vec<u8>>,
801    codes: Vec<Option<Vec<u32>>>,
802    sieves: Vec<Option<Sieve>>,
803    ranges: Vec<Range>,
804}
805
806/// Roughly what encoding a column of this type costs, for ordering the encode queue.
807///
808/// Only the order matters and only roughly. A string column hashes and copies every value into a
809/// dictionary and is in a different class from everything else, and among the fixed widths the wide
810/// ones carry more bytes through the cascade than the narrow ones. Anything finer than that would
811/// be a cost model, and the queue already absorbs a wrong guess: it only has to avoid finishing on
812/// a column nobody else can help with.
813fn weight(ty: &LogicalType) -> usize {
814    match ty {
815        LogicalType::Varchar | LogicalType::Blob | LogicalType::Bit => 64,
816        LogicalType::HugeInt
817        | LogicalType::UHugeInt
818        | LogicalType::Uuid
819        | LogicalType::Interval => 16,
820        LogicalType::BigInt
821        | LogicalType::UBigInt
822        | LogicalType::Timestamp
823        | LogicalType::Time
824        | LogicalType::TimeTz
825        | LogicalType::TimestampTz
826        | LogicalType::TimestampS
827        | LogicalType::TimestampMs
828        | LogicalType::TimestampNs
829        | LogicalType::Double
830        | LogicalType::Decimal { .. } => 8,
831        LogicalType::Integer | LogicalType::UInteger | LogicalType::Date | LogicalType::Float => 4,
832        LogicalType::SmallInt | LogicalType::USmallInt => 2,
833        _ => 1,
834    }
835}
836
837/// Parts in one stripe.
838///
839/// Sixty four thousand rows is the smallest stripe that keeps the ClickBench directory in single
840/// digit megabytes at a hundred million rows, and it puts a four byte column's page at a quarter of
841/// a megabyte, which is the size a sequential read wants. Larger stripes buy a smaller directory
842/// and cost a sparse fetch, which has to read a page index before it can reach one part.
843pub const STRIPE_PARTS: usize = 64;
844
845/// How many rows the writer wants to see before it decides whether a varchar column gets to keep
846/// its global dictionary.
847///
848/// See [`Writer::encode_column`]. A stripe is up to [`STRIPE_PARTS`] parts, so most tables give it
849/// far more than this and it binds only on a table that is smaller than one stripe. A handful of
850/// rows says nothing about whether a column repeats itself, and the answer that costs nothing when
851/// the sample is that small is the one the writer has always given, which is to keep the dictionary.
852const DICTIONARY_DECIDE_ROWS: usize = 4_096;
853
854/// Out of ten. A varchar column loses its dictionary when more than this many rows in ten of the
855/// first stripe held a value that stripe had not seen before.
856///
857/// See [`Writer::encode_column`]. Nine and not five, because the properties a dictionary buys are
858/// worth keeping everywhere they are real. On ClickBench the widest string column is `Referer` at
859/// 0.131 of its first stripe and every other one is below that, so nothing there is near this and
860/// every one of them keeps its dictionary, which is what a group by on codes wants. On TPC-H
861/// `o_comment` and `c_comment` are at 0.97 and are what this catches.
862///
863/// `l_comment` sits at 0.883 and so keeps its dictionary. Eight was built and measured rather than
864/// argued about, and it is not a clear win: it takes `select l_comment from lineitem` from 4.335 G
865/// instructions to 3.473 G and the file from 280.2 MB to 260.4 MB, and it takes a `like` over the
866/// same column from 3.29 G to 4.27 G, because a dictionary runs the predicate once a distinct value
867/// and there are 3.6 M of those to 6.0 M rows. The 22 query suite came out 6.91 s against 7.07 s in
868/// favour of nine. So nine stays until there is a reason to prefer one of those shapes. See #1137.
869const DICTIONARY_DISTINCT_IN_TEN: usize = 9;
870
871/// Bytes one part takes in a stripe's index page: four for the length, eight for the checksum.
872const INDEX_ENTRY: usize = size_of::<u32>() + size_of::<u64>();
873
874/// Bytes one column's section of a stripe's index page takes, including its own trailing checksum.
875fn index_section(parts: usize) -> Result<usize> {
876    parts
877        .checked_mul(INDEX_ENTRY)
878        .and_then(|bytes| bytes.checked_add(size_of::<u64>()))
879        .ok_or_else(|| invalid("index page length overflow"))
880}
881
882impl Writer {
883    /// Opens a committed file and starts a table in the generation after the one it holds.
884    ///
885    /// The tables already in the file are carried forward by name and by directory pointer, and
886    /// their pages are not read. Nothing in the file is overwritten: the new table's pages and the
887    /// new catalog go on the end, past the catalog the committed generation points at, and the one
888    /// write that is not an append is the slot in the header that [`Writer::finish`] does last.
889    ///
890    /// That slot is the other one. A file committed at generation 1 is named by the slot at 16 and
891    /// generation 2 writes the one at 44, so until the last four bytes of the commit land the file
892    /// still reads as the generation before it, and a slot torn across a write fails its checksum
893    /// and the reader falls back to the one beside it. This is what the second slot has always been
894    /// for.
895    ///
896    /// # Errors
897    ///
898    /// If the file has no valid committed directory, is not this build's format, repeats the name
899    /// of a table already in it that holds rows, has a field with no scalar encoding, or cannot be
900    /// written.
901    pub fn open(
902        path: impl AsRef<Path>,
903        name: impl Into<String>,
904        fields: Vec<Field>,
905    ) -> Result<Self> {
906        for field in &fields {
907            type_tag(&field.ty)?;
908        }
909        let name = name.into();
910        let path = path.as_ref();
911        let (_, size, slot, bytes, _) = slot_bytes(path)?;
912        let mut closed = decode_catalog(&bytes, size)?;
913        // A table already in the file under this name is only in the way if it holds rows. One that
914        // holds none has no pages for this generation to carry and no reader that could lose
915        // anything, so the table being started here takes its place in the catalog rather than
916        // colliding with it, and `finish` writes the new entry where the old one was.
917        //
918        // That is not a corner. It is the shape every loading script writes: the schema goes in one
919        // statement and the rows go in the next, and a checkpoint between them commits the empty
920        // table. Before this, the second statement had to build the whole table in memory because
921        // the first had already put the name in the file, which is how a load of a table larger
922        // than memory became a load that needed memory the size of the table.
923        if let Some(at) = closed.iter().position(|held| held.name == name) {
924            if closed[at].rows > 0 {
925                return Err(invalid("two tables in one native file have the same name"));
926            }
927            closed.remove(at);
928        }
929        // The generation of the slot whose bytes checksummed, and not the highest number in the
930        // header. A slot torn across a write can hold any number at all, and taking that one would
931        // be choosing which slot to overwrite from a value nothing has vouched for, which is how a
932        // half written commit gets to destroy the one good copy beside it.
933        let generation = slot
934            .generation
935            .checked_add(1)
936            .ok_or_else(|| invalid("native file generation overflow"))?;
937        let file = OpenOptions::new().write(true).read(true).open(path).map_err(io)?;
938        Ok(Self {
939            file,
940            // The end of the file, so that the committed generation's catalog stays where its slot
941            // says it is and keeps naming a file a reader can still open.
942            at: size,
943            dictionaries: fields
944                .iter()
945                .map(|field| (field.ty == LogicalType::Varchar).then(GlobalDictionary::new))
946                .collect(),
947            table: Table {
948                name,
949                dictionaries: vec![None; fields.len()],
950                distincts: vec![None; fields.len()],
951                fields,
952                stripes: Vec::new(),
953                rows: 0,
954                frequencies: Vec::new(),
955                clustering: None,
956                generation,
957                sections: Vec::new(),
958            },
959            generation,
960            order: Vec::new(),
961            next_order: 0,
962            pending: Vec::with_capacity(STRIPE_PARTS),
963            closed,
964        })
965    }
966
967    /// Creates a new v10 file and its first table.
968    ///
969    /// # Errors
970    ///
971    /// If the file exists, a field has no scalar encoding, or the path cannot be written.
972    pub fn create(
973        path: impl AsRef<Path>,
974        name: impl Into<String>,
975        fields: Vec<Field>,
976    ) -> Result<Self> {
977        for field in &fields {
978            type_tag(&field.ty)?;
979        }
980        let file =
981            OpenOptions::new().write(true).read(true).create_new(true).open(path).map_err(io)?;
982        let mut header = [0; HEADER as usize];
983        header[..8].copy_from_slice(MAGIC);
984        header[8..12].copy_from_slice(&FORMAT.to_le_bytes());
985        write_at(&file, 0, &header)?;
986        Ok(Self {
987            file,
988            at: HEADER,
989            dictionaries: fields
990                .iter()
991                .map(|field| (field.ty == LogicalType::Varchar).then(GlobalDictionary::new))
992                .collect(),
993            table: Table {
994                name: name.into(),
995                dictionaries: vec![None; fields.len()],
996                distincts: vec![None; fields.len()],
997                fields,
998                stripes: Vec::new(),
999                rows: 0,
1000                frequencies: Vec::new(),
1001                clustering: None,
1002                generation: 1,
1003                sections: Vec::new(),
1004            },
1005            generation: 1,
1006            order: Vec::new(),
1007            next_order: 0,
1008            pending: Vec::with_capacity(STRIPE_PARTS),
1009            closed: Vec::new(),
1010        })
1011    }
1012
1013    /// Creates a new file that holds no table at all, committed and ready to open.
1014    ///
1015    /// A database somebody dropped the last table out of is still a database, and until this there
1016    /// was no way to write one down. Every other way into this file goes through a table, because
1017    /// [`Writer::create`] takes the first one and [`Writer::finish`] commits the one it is on, so a
1018    /// catalog with nothing in it could be read and not written. The format already allowed it: the
1019    /// catalog is a count and that many entries, and a count of nought encodes and decodes the same
1020    /// way every other count does, which is why nothing here is a version change.
1021    ///
1022    /// It hands back nothing rather than a writer, because a writer with no table is a writer with
1023    /// nothing to append to. A file that is going to hold a table is [`Writer::create`], and one
1024    /// that is going to have a table added to it later is [`Writer::open`], which reads what this
1025    /// wrote the same way it reads any other generation.
1026    ///
1027    /// # Errors
1028    ///
1029    /// If the file exists or the path cannot be written.
1030    pub fn empty(path: impl AsRef<Path>) -> Result<()> {
1031        let file =
1032            OpenOptions::new().write(true).read(true).create_new(true).open(path).map_err(io)?;
1033        let mut header = [0; HEADER as usize];
1034        header[..8].copy_from_slice(MAGIC);
1035        header[8..12].copy_from_slice(&FORMAT.to_le_bytes());
1036        write_at(&file, 0, &header)?;
1037        let catalog = encode_catalog(&[])?;
1038        write_at(&file, HEADER, &catalog)?;
1039        // The same two syncs in the same order as [`Writer::finish`], and for the same reason. The
1040        // catalog is on the disk before the slot names it, so a file this is interrupted in the
1041        // middle of is a header with no valid slot rather than a slot pointing at nothing.
1042        file.sync_all().map_err(io)?;
1043        let slot = Slot {
1044            offset: HEADER,
1045            length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
1046            generation: 1,
1047            hash: checksum(&catalog),
1048        };
1049        write_at(&file, slot_offset(1), &slot.bytes())?;
1050        file.sync_all().map_err(io)?;
1051        Ok(())
1052    }
1053
1054    /// Closes the table this writer is on and starts another one in the same file.
1055    ///
1056    /// Nothing is published here. The closed table's directory is written so that the bytes are on
1057    /// disk and its span is known, and the catalog that names it is only written by
1058    /// [`Writer::finish`], so a crash between two tables leaves the previous generation intact.
1059    ///
1060    /// # Errors
1061    ///
1062    /// If the name repeats a table already closed, a field has no scalar encoding, or the table
1063    /// being closed cannot be written.
1064    pub fn next(mut self, name: impl Into<String>, fields: Vec<Field>) -> Result<Self> {
1065        for field in &fields {
1066            type_tag(&field.ty)?;
1067        }
1068        let name = name.into();
1069        let entry = self.close()?;
1070        if self.closed.iter().chain(std::iter::once(&entry)).any(|held| held.name == name) {
1071            return Err(invalid("two tables in one native file have the same name"));
1072        }
1073        let Self { file, at, generation, mut closed, .. } = self;
1074        closed.push(entry);
1075        Ok(Self {
1076            file,
1077            at,
1078            generation,
1079            closed,
1080            dictionaries: fields
1081                .iter()
1082                .map(|field| (field.ty == LogicalType::Varchar).then(GlobalDictionary::new))
1083                .collect(),
1084            table: Table {
1085                name,
1086                dictionaries: vec![None; fields.len()],
1087                distincts: vec![None; fields.len()],
1088                fields,
1089                stripes: Vec::new(),
1090                rows: 0,
1091                frequencies: Vec::new(),
1092                clustering: None,
1093                generation,
1094                sections: Vec::new(),
1095            },
1096            order: Vec::new(),
1097            next_order: 0,
1098            pending: Vec::with_capacity(STRIPE_PARTS),
1099        })
1100    }
1101
1102    /// Records the order this table's rows are meant to be stored in.
1103    ///
1104    /// The declaration goes in the table directory and comes back out of
1105    /// [`Table::clustering`]. Nothing here sorts anything, and nothing here checks that the rows
1106    /// handed to [`Writer::append`] arrive in the order this claims. That is deliberate for now:
1107    /// the thing that was missing was a place to write the order down, and a loader that honours
1108    /// the declaration is the next piece rather than this one.
1109    ///
1110    /// The declaration applies to the table the writer is currently on, so it is set after
1111    /// [`Writer::next`] rather than once for the file.
1112    ///
1113    /// # Errors
1114    ///
1115    /// If the declaration names a column this table does not have.
1116    pub fn declare(mut self, clustering: Clustering) -> Result<Self> {
1117        // Rebuilt against this table's own column count rather than trusted, because the caller
1118        // built it against a catalog entry and the two could have drifted.
1119        self.table.clustering = Some(Clustering::new(
1120            clustering.columns().to_vec(),
1121            clustering.width(),
1122            &self.table.fields,
1123        )?);
1124        Ok(self)
1125    }
1126
1127    /// Appends bytes at the end of the file and moves the writer's own offset past them.
1128    ///
1129    /// Every write in here goes through this, so that [`Writer::at`] is the only answer to where
1130    /// anything is and the file's cursor is never consulted for it.
1131    fn put(&mut self, bytes: &[u8]) -> Result<()> {
1132        write_at(&self.file, self.at, bytes)?;
1133        self.at = self
1134            .at
1135            .checked_add(bytes.len() as u64)
1136            .ok_or_else(|| invalid("native file length overflow"))?;
1137        Ok(())
1138    }
1139
1140    /// Writes one chunk as independently readable column pages.
1141    ///
1142    /// # Errors
1143    ///
1144    /// If its width or types differ from the declared table, or a page exceeds its bound.
1145    pub fn append(&mut self, chunk: &Chunk) -> Result<()> {
1146        let order = (self.next_order, 0);
1147        self.next_order = self.next_order.saturating_add(1);
1148        self.append_at(order, chunk)
1149    }
1150
1151    /// Writes one chunk and records its source position for directory ordering.
1152    ///
1153    /// Pages may be encoded by parallel pipeline instances and reach the file in completion order.
1154    /// The stripe they land in is sorted by this key at commit, and [`Self::finish`] rejects a
1155    /// sequence whose parts do not come out in source order once the stripes are sorted, because a
1156    /// stripe groups whatever arrived together and cannot put a late part back where it belongs.
1157    ///
1158    /// # Errors
1159    ///
1160    /// The same as [`Self::append`].
1161    pub fn append_at(&mut self, order: (u64, u64), chunk: &Chunk) -> Result<()> {
1162        if chunk.is_empty() {
1163            return Ok(());
1164        }
1165        self.admit(chunk)?;
1166        if self.pending.last().is_some_and(|last| last.order > order) {
1167            self.flush_pending()?;
1168        }
1169        // Cloned rather than encoded, and a clone of a chunk that owns its buffers is a copy of
1170        // them. Sixty four parts of a hundred and five columns is tens of megabytes held for the
1171        // length of a stripe and a few seconds of memory traffic over a whole ClickBench load,
1172        // against the hundreds of seconds of encode this is what lets off one thread.
1173        self.pending.push(PendingChunk { order, chunk: chunk.clone() });
1174        if self.pending.len() == STRIPE_PARTS {
1175            self.flush_pending()?;
1176        }
1177        Ok(())
1178    }
1179
1180    /// Writes a run of chunks as one stripe of its own.
1181    ///
1182    /// [`Self::append_at`] decides where a stripe ends by watching the orders go past, which works
1183    /// when one caller hands over every chunk in source order and does not when several do. A
1184    /// writer being fed by more than one pipeline instance sees the orders interleave, and a stripe
1185    /// that ends every time two of them cross is a stripe of one or two parts.
1186    ///
1187    /// So the grouping moves to the caller. Whoever is buffering hands over a run it already knows
1188    /// is contiguous and in order, and gets a stripe holding exactly that run. The orders still
1189    /// have to come out in source order once the stripes are sorted, which [`Self::finish`] checks,
1190    /// so the runs from different callers may interleave with each other but may not overlap.
1191    ///
1192    /// # Errors
1193    ///
1194    /// The same as [`Self::append`], and if the run is longer than [`STRIPE_PARTS`].
1195    pub fn append_stripe(&mut self, parts: Vec<((u64, u64), Chunk)>) -> Result<()> {
1196        if parts.len() > STRIPE_PARTS {
1197            return Err(invalid("a stripe was handed more parts than it holds"));
1198        }
1199        // Whatever an earlier caller left behind is its own stripe rather than the front of this
1200        // one, because the two runs are from different places in the source and a stripe is a run.
1201        self.flush_pending()?;
1202        for (order, chunk) in parts {
1203            if chunk.is_empty() {
1204                continue;
1205            }
1206            self.admit(&chunk)?;
1207            self.pending.push(PendingChunk { order, chunk });
1208        }
1209        self.flush_pending()
1210    }
1211
1212    /// Checks a chunk against the declared table and counts its rows in.
1213    fn admit(&mut self, chunk: &Chunk) -> Result<()> {
1214        if chunk.width() != self.table.fields.len() {
1215            return Err(invalid("chunk width differs from table schema"));
1216        }
1217        for (index, field) in self.table.fields.iter().enumerate() {
1218            if chunk.column(index)?.logical_type() != &field.ty {
1219                return Err(invalid("chunk type differs from table schema"));
1220            }
1221        }
1222        self.table.rows = self
1223            .table
1224            .rows
1225            .checked_add(chunk.len())
1226            .ok_or_else(|| invalid("row count overflow"))?;
1227        Ok(())
1228    }
1229
1230    /// Encodes one column's parts of a stripe, and on the first stripe decides whether the column
1231    /// should have a dictionary at all.
1232    ///
1233    /// Every varchar column starts with one, because the writer cannot know what is in a column
1234    /// before it has seen some of it. A global dictionary is the right shape for a column of a few
1235    /// dozen values repeated down the table: the pages become small integers, a filter against a
1236    /// literal is one search of the sorted order rather than a comparison a row, and a group by is
1237    /// on the codes. It is the wrong shape for a column whose values are nearly all different.
1238    /// There the codes are as wide as row numbers, nothing is saved on the pages, and the
1239    /// membership index of a stripe is a list of very nearly every code in the column. On TPC-H the
1240    /// orders table written on its own goes from 52.3 MB to 41.4 MB, the load from 6.9 s to 5.8 s,
1241    /// and `select o_comment from orders` from 1.810 G instructions to 1.213 G, which is what the
1242    /// rudb parquet reader takes over the same values.
1243    ///
1244    /// So the first stripe of a column is the sample and the decision is made once on it. Once,
1245    /// rather than per stripe, because the codes of one column have to mean the same thing in every
1246    /// page of it, and a column that changed its mind halfway would need its earlier stripes
1247    /// rewritten. The first stripe is re-encoded when the answer comes out against the dictionary,
1248    /// which is the one stripe that pays for the decision.
1249    ///
1250    /// The threshold is deliberately near the top. [`DICTIONARY_DISTINCT_IN_TEN`] of the sample has
1251    /// to be values never seen before, which is a column with essentially no repeats. Everything
1252    /// with real repetition keeps its dictionary and keeps every property that hangs off it, and
1253    /// nothing is claimed here about where between the two the crossover really sits.
1254    ///
1255    /// Nothing here is shared with another column. The dictionary belongs to this one, the sieve
1256    /// reads only this one, and the page bytes go in a vector of this one's own. That is why the
1257    /// fan out below can hand a whole column to a thread and take a plain `&mut` on the dictionary
1258    /// rather than making it something several threads can grow at once, which is the harder half
1259    /// of #808 and is still open.
1260    fn encode_column(
1261        index: usize,
1262        held: &[PendingChunk],
1263        dictionary: &mut Option<GlobalDictionary>,
1264    ) -> Result<ColumnStripe> {
1265        // Empty means nothing has been written through it yet, so this is the column's first stripe
1266        // and the only stripe the decision below is allowed to be made on.
1267        let deciding = dictionary.as_ref().is_some_and(|held| held.offsets.len() == 1);
1268        let stripe = Self::encode_pages(index, held, dictionary.as_mut())?;
1269        if !deciding {
1270            return Ok(stripe);
1271        }
1272        let rows: usize = held.iter().map(|pending| pending.chunk.len()).sum();
1273        let distinct = dictionary.as_ref().map_or(0, |held| held.offsets.len() - 1);
1274        if rows < DICTIONARY_DECIDE_ROWS
1275            || distinct.saturating_mul(10) <= rows.saturating_mul(DICTIONARY_DISTINCT_IN_TEN)
1276        {
1277            return Ok(stripe);
1278        }
1279        *dictionary = None;
1280        Self::encode_pages(index, held, None)
1281    }
1282
1283    /// One column's parts of a stripe, with whatever dictionary it was given.
1284    fn encode_pages(
1285        index: usize,
1286        held: &[PendingChunk],
1287        mut dictionary: Option<&mut GlobalDictionary>,
1288    ) -> Result<ColumnStripe> {
1289        let mut stripe = ColumnStripe {
1290            pages: Vec::with_capacity(held.len()),
1291            codes: Vec::with_capacity(held.len()),
1292            sieves: Vec::with_capacity(held.len()),
1293            ranges: Vec::with_capacity(held.len()),
1294        };
1295        for pending in held {
1296            let column = pending.chunk.column(index)?;
1297            let (bytes, unique) = encode(column, dictionary.as_deref_mut())?;
1298            if bytes.len() > MAX_PAGE {
1299                return Err(invalid("column page exceeds the configured bound"));
1300            }
1301            // The range is built first because the sieve reads it rather than walking the column a
1302            // second time to find out how wide it is.
1303            let range = Range::of(column);
1304            // A column with a global dictionary already has an exact membership index per stripe,
1305            // so an approximate one beside it would cost a hash of every string in the table to
1306            // answer a question that is already answered. What it would buy is the finer grain, a
1307            // part rather than a stripe, and that is worth coming back for on its own.
1308            //
1309            // A sieve at least as large as the part it indexes is not written. A reader reads the
1310            // sieve to decide whether to read the part, so when the sieve is the larger of the two
1311            // it has already spent more than the read it is trying to avoid, and that holds even if
1312            // it rejects every time. It is a necessary condition rather than the whole rule, which
1313            // is that a sieve pays when its bytes are under the rejection rate times the part's,
1314            // but the rejection rate depends on what a query probes for and the writer does not
1315            // know that. The necessary half needs two numbers that are both in hand here.
1316            let sieve = match dictionary {
1317                Some(_) => None,
1318                None => Sieve::of(column, &range, SIEVE_BUDGET)
1319                    .filter(|sieve| sieve.len() < bytes.len()),
1320            };
1321            stripe.pages.push(bytes);
1322            stripe.codes.push(unique);
1323            stripe.sieves.push(sieve);
1324            stripe.ranges.push(range);
1325        }
1326        Ok(stripe)
1327    }
1328
1329    /// Encodes a whole stripe, one column to a worker.
1330    ///
1331    /// The columns are handed out through a queue rather than dealt in equal piles, because they
1332    /// are nothing like equal: `URL` on ClickBench is a global dictionary of sixty one million
1333    /// strings and `IsMobile` is a byte. A pile that happened to hold the four large string columns
1334    /// would be the whole stripe and the other workers would be waiting on it. The queue is sorted
1335    /// so the expensive ones are taken first, which is the classic answer to a last job that runs
1336    /// longer than everything after it.
1337    fn encode_columns(&mut self, held: &[PendingChunk]) -> Result<Vec<ColumnStripe>> {
1338        let width = self.table.fields.len();
1339        let workers = std::thread::available_parallelism()
1340            .map_or(1, usize::from)
1341            .min(MAX_ENCODE_WORKERS)
1342            .min(width);
1343        if workers <= 1 || held.len() <= 1 {
1344            return self
1345                .dictionaries
1346                .iter_mut()
1347                .enumerate()
1348                .map(|(index, dictionary)| Self::encode_column(index, held, dictionary))
1349                .collect();
1350        }
1351        // The dictionaries are moved out and back rather than borrowed, because a worker that takes
1352        // the next column off a queue cannot be holding a borrow of the vector the queue came from.
1353        let mut jobs: Vec<(usize, Option<GlobalDictionary>)> =
1354            std::mem::take(&mut self.dictionaries).into_iter().enumerate().collect();
1355        // Popped from the back, so the expensive columns go last in the vector.
1356        jobs.sort_by_key(|(index, _)| weight(&self.table.fields[*index].ty));
1357        let queue = Mutex::new(jobs);
1358        let pieces = std::thread::scope(|scope| {
1359            (0..workers)
1360                .map(|_| {
1361                    scope.spawn(|| {
1362                        let mut mine = Vec::new();
1363                        loop {
1364                            let taken = queue
1365                                .lock()
1366                                .map_err(|_| Error::internal("a native encode worker panicked"))?
1367                                .pop();
1368                            let Some((index, mut dictionary)) = taken else { break };
1369                            let encoded = Self::encode_column(index, held, &mut dictionary)?;
1370                            mine.push((index, dictionary, encoded));
1371                        }
1372                        Ok(mine)
1373                    })
1374                })
1375                .collect::<Vec<_>>()
1376                .into_iter()
1377                .map(|handle| {
1378                    handle.join().map_err(|_| Error::internal("a native encode worker panicked"))?
1379                })
1380                .collect::<Result<Vec<_>>>()
1381        })?;
1382        let mut dictionaries: Vec<Option<GlobalDictionary>> = (0..width).map(|_| None).collect();
1383        let mut encoded: Vec<Option<ColumnStripe>> = (0..width).map(|_| None).collect();
1384        for piece in pieces {
1385            for (index, dictionary, stripe) in piece {
1386                dictionaries[index] = dictionary;
1387                encoded[index] = Some(stripe);
1388            }
1389        }
1390        self.dictionaries = dictionaries;
1391        encoded
1392            .into_iter()
1393            .map(|stripe| stripe.ok_or_else(|| Error::internal("a column was never encoded")))
1394            .collect()
1395    }
1396
1397    /// Writes the buffered parts as one stripe, each column's parts contiguous on disk.
1398    fn flush_pending(&mut self) -> Result<()> {
1399        if self.pending.is_empty() {
1400            return Ok(());
1401        }
1402        let width = self.table.fields.len();
1403        // Held here rather than read off the writer, because writing a page needs the writer and
1404        // the borrow checker is right that those are two different uses of it.
1405        let mut held = std::mem::take(&mut self.pending);
1406        let parts = held.len();
1407        let encoded = self.encode_columns(&held)?;
1408        let mut pages = Vec::with_capacity(width);
1409        let mut memberships = vec![None; width];
1410        let mut ranges = Vec::with_capacity(width);
1411        let mut index = Vec::with_capacity(width.saturating_mul(index_section(parts)?));
1412        for stripe in &encoded {
1413            let offset = self.at;
1414            let section = index.len();
1415            let mut length = 0_usize;
1416            for bytes in &stripe.pages {
1417                write_at(&self.file, self.at + length as u64, bytes)?;
1418                put_u32(
1419                    &mut index,
1420                    u32::try_from(bytes.len()).map_err(|_| invalid("part length overflow"))?,
1421                );
1422                put_u64(&mut index, checksum(bytes));
1423                length = length
1424                    .checked_add(bytes.len())
1425                    .ok_or_else(|| invalid("column page length overflow"))?;
1426            }
1427            let hash = checksum(&index[section..]);
1428            put_u64(&mut index, hash);
1429            if length > MAX_PAGE {
1430                return Err(invalid("column page exceeds the configured bound"));
1431            }
1432            self.at = self
1433                .at
1434                .checked_add(length as u64)
1435                .ok_or_else(|| invalid("native file length overflow"))?;
1436            pages.push(Span {
1437                offset,
1438                length: u32::try_from(length).map_err(|_| invalid("page length overflow"))?,
1439            });
1440            ranges.push(merged_range(stripe.ranges.iter().cloned()));
1441        }
1442        for (membership, stripe) in memberships.iter_mut().zip(&encoded) {
1443            if stripe.codes.iter().all(Option::is_none) {
1444                continue;
1445            }
1446            let lists = stripe
1447                .codes
1448                .iter()
1449                .map(|codes| codes.clone().unwrap_or_default())
1450                .collect::<Vec<_>>();
1451            let bytes = encode_membership(&merged_codes(lists));
1452            let offset = self.at;
1453            self.put(&bytes)?;
1454            *membership = Some(Page {
1455                offset,
1456                length: u32::try_from(bytes.len())
1457                    .map_err(|_| invalid("membership page length overflow"))?,
1458                hash: checksum(&bytes),
1459            });
1460        }
1461        let mut sieves = vec![None; width];
1462        for (page, stripe) in sieves.iter_mut().zip(&encoded) {
1463            if stripe.sieves.iter().all(Option::is_none) {
1464                continue;
1465            }
1466            let bytes = encode_sieves(stripe.sieves.iter())?;
1467            let offset = self.at;
1468            self.put(&bytes)?;
1469            *page = Some(Page {
1470                offset,
1471                length: u32::try_from(bytes.len())
1472                    .map_err(|_| invalid("sieve page length overflow"))?,
1473                hash: checksum(&bytes),
1474            });
1475        }
1476        // A stripe of one part has the same rows in it as that part, so its own bounds are already
1477        // the part's and a page here would say what the directory says. Everywhere else the page is
1478        // written unless it comes to more than the column it indexes, which is the rule the sieves
1479        // go by and for the same reason: a reader reads this to decide whether to read the column,
1480        // so a page larger than the column has spent more than the read it is avoiding.
1481        let mut part_ranges = vec![None; width];
1482        if parts > 1 {
1483            for ((page, stripe), span) in part_ranges.iter_mut().zip(&encoded).zip(&pages) {
1484                let bytes = encode_part_ranges(&stripe.ranges)?;
1485                if bytes.len() >= span.length as usize {
1486                    continue;
1487                }
1488                let offset = self.at;
1489                self.put(&bytes)?;
1490                *page = Some(Page {
1491                    offset,
1492                    length: u32::try_from(bytes.len())
1493                        .map_err(|_| invalid("part range page length overflow"))?,
1494                    hash: checksum(&bytes),
1495                });
1496            }
1497        }
1498        let offset = self.at;
1499        self.put(&index)?;
1500        let index = Span {
1501            offset,
1502            length: u32::try_from(index.len())
1503                .map_err(|_| invalid("index page length overflow"))?,
1504        };
1505        let mut rows = 0_usize;
1506        let mut lengths = Vec::with_capacity(parts);
1507        let mut span = None;
1508        for pending in held.drain(..) {
1509            let part = pending.chunk.len();
1510            rows = rows.checked_add(part).ok_or_else(|| invalid("row count overflow"))?;
1511            lengths.push(u32::try_from(part).map_err(|_| invalid("part row count overflow"))?);
1512            span = Some(
1513                span.map_or((pending.order, pending.order), |(first, _)| (first, pending.order)),
1514            );
1515        }
1516        self.order.push(span.ok_or_else(|| invalid("a stripe was flushed with no parts"))?);
1517        self.table.stripes.push(Stripe {
1518            rows,
1519            parts: lengths,
1520            index,
1521            pages,
1522            memberships,
1523            sieves,
1524            part_ranges,
1525            zone: Zone::from_ranges(ranges),
1526        });
1527        // Back where it came from, empty, so the next stripe buffers into the same allocation.
1528        self.pending = held;
1529        Ok(())
1530    }
1531
1532    /// Finds exact heavy hitters without keeping a hash table for every numeric column while the
1533    /// load is live. The pages are already in the target file, so one column at a time uses a
1534    /// bounded Misra-Gries candidate table and then recounts only those candidates.
1535    fn numeric_frequency(&self, column: usize) -> Result<Option<FrequencySummary>> {
1536        let ty = &self.table.fields[column].ty;
1537        if !matches!(
1538            ty,
1539            LogicalType::TinyInt
1540                | LogicalType::SmallInt
1541                | LogicalType::Integer
1542                | LogicalType::BigInt
1543                | LogicalType::UTinyInt
1544                | LogicalType::USmallInt
1545                | LogicalType::UInteger
1546                | LogicalType::UBigInt
1547                | LogicalType::Date
1548                | LogicalType::Timestamp
1549        ) {
1550            return Ok(None);
1551        }
1552        let mut candidates: HashMap<FrequencyValue, u32> = HashMap::new();
1553        let mut decrements = 0_u64;
1554        self.visit_numeric(column, |_, value| {
1555            if let Some(count) = candidates.get_mut(&value) {
1556                *count = count.saturating_add(1);
1557            } else if candidates.len() < FREQUENCY_CANDIDATES {
1558                candidates.insert(value, 1);
1559            } else {
1560                candidates.retain(|_, count| {
1561                    *count -= 1;
1562                    *count != 0
1563                });
1564                decrements = decrements.saturating_add(1);
1565            }
1566        })?;
1567        let (exact, ordinals) = if decrements == 0 {
1568            (
1569                candidates
1570                    .into_iter()
1571                    .map(|(value, count)| (value, u64::from(count)))
1572                    .collect::<HashMap<_, _>>(),
1573                Vec::new(),
1574            )
1575        } else {
1576            let mut lower = candidates.values().copied().collect::<Vec<_>>();
1577            lower.sort_unstable_by(|left, right| right.cmp(left));
1578            if lower.len() < FREQUENCY_BUILD_RANK
1579                || u64::from(lower[FREQUENCY_BUILD_RANK - 1]) <= decrements
1580            {
1581                return Ok(None);
1582            }
1583            let mut exact =
1584                candidates.into_keys().map(|value| (value, 0_u64)).collect::<HashMap<_, _>>();
1585            let mut ordinals = Vec::new();
1586            let mut exceeded = false;
1587            self.visit_numeric(column, |ordinal, value| {
1588                if let Some(count) = exact.get_mut(&value) {
1589                    *count = count.saturating_add(1);
1590                    if !exceeded {
1591                        if ordinals.len() < FREQUENCY_ORDINALS {
1592                            ordinals.push(ordinal);
1593                        } else {
1594                            ordinals.clear();
1595                            exceeded = true;
1596                        }
1597                    }
1598                }
1599            })?;
1600            (exact, ordinals)
1601        };
1602        let mut entries = exact
1603            .into_iter()
1604            .map(|(value, count)| FrequencyEntry { value, count })
1605            .collect::<Vec<_>>();
1606        entries.sort_unstable_by(|left, right| {
1607            right.count.cmp(&left.count).then_with(|| frequency_order(left.value, right.value))
1608        });
1609        let omitted_max =
1610            entries.get(FREQUENCY_ENTRIES).map_or(decrements, |entry| decrements.max(entry.count));
1611        entries.truncate(FREQUENCY_ENTRIES);
1612        Ok(Some(FrequencySummary { entries, omitted_max, ordinals }))
1613    }
1614
1615    fn visit_numeric(
1616        &self,
1617        column: usize,
1618        mut visit: impl FnMut(u64, FrequencyValue),
1619    ) -> Result<()> {
1620        let ty = &self.table.fields[column].ty;
1621        let mut start = 0_u64;
1622        for stripe in &self.table.stripes {
1623            let spans = read_index(&self.file, stripe, column)?;
1624            let page = stripe.pages[column];
1625            let mut bytes = vec![0; page.length as usize];
1626            read_at(&self.file, page.offset, &mut bytes)?;
1627            for (span, &rows) in spans.iter().zip(&stripe.parts) {
1628                let part = part_bytes(&bytes, *span)?;
1629                if checksum(part) != span.hash {
1630                    return Err(invalid("column page checksum differs while building frequencies"));
1631                }
1632                let rows = rows as usize;
1633                let vector = decode(ty, rows, part, None)?;
1634                // row at a time: frequency construction visits decoded values to update bounded candidates.
1635                for row in 0..rows {
1636                    let value = if vector.is_null_at(row) {
1637                        FrequencyValue::Null
1638                    } else {
1639                        // An unsigned column has no signed reading, and the documented fallback is
1640                        // the value itself. Every unsigned width the format stores fits in the
1641                        // `i128` a candidate is keyed by, so nothing is lost on the way through.
1642                        let widened = match vector.signed_at(row) {
1643                            Some(value) => Some(value),
1644                            None => match vector.value_at(row) {
1645                                Value::UTinyInt(value) => Some(i128::from(value)),
1646                                Value::USmallInt(value) => Some(i128::from(value)),
1647                                Value::UInteger(value) => Some(i128::from(value)),
1648                                Value::UBigInt(value) => Some(i128::from(value)),
1649                                _ => None,
1650                            },
1651                        };
1652                        FrequencyValue::Integer(widened.ok_or_else(|| {
1653                            invalid("numeric frequency page did not contain an integer value")
1654                        })?)
1655                    };
1656                    visit(start.saturating_add(row as u64), value);
1657                }
1658                start = start.saturating_add(rows as u64);
1659            }
1660        }
1661        Ok(())
1662    }
1663
1664    /// Builds independent numeric synopses concurrently after all column pages are committed.
1665    ///
1666    /// The columns go through a queue rather than being cut into equal runs, because they are not
1667    /// equally expensive and they are not shuffled. A `BIGINT` column carries eight times the bytes
1668    /// of a `TINYINT` through the decode, and a run of them sits together in a schema the way it
1669    /// sits together in `hits`, so a worker that was handed the wrong six columns finishes long
1670    /// after one that was handed the right six and the whole phase waits for it.
1671    fn numeric_frequencies(&self) -> Result<Vec<Option<FrequencySummary>>> {
1672        let mut columns = self
1673            .table
1674            .fields
1675            .iter()
1676            .enumerate()
1677            .filter_map(|(column, field)| {
1678                matches!(
1679                    field.ty,
1680                    LogicalType::TinyInt
1681                        | LogicalType::SmallInt
1682                        | LogicalType::Integer
1683                        | LogicalType::BigInt
1684                        | LogicalType::UTinyInt
1685                        | LogicalType::USmallInt
1686                        | LogicalType::UInteger
1687                        | LogicalType::UBigInt
1688                        | LogicalType::Date
1689                        | LogicalType::Timestamp
1690                )
1691                .then_some(column)
1692            })
1693            .collect::<Vec<_>>();
1694        let workers = std::thread::available_parallelism()
1695            .map_or(1, usize::from)
1696            .min(MAX_FREQUENCY_WORKERS)
1697            .min(columns.len());
1698        if workers <= 1 {
1699            let mut frequencies = vec![None; self.table.fields.len()];
1700            for column in columns {
1701                frequencies[column] = self.numeric_frequency(column)?;
1702            }
1703            return Ok(frequencies);
1704        }
1705        // Popped from the back, so the expensive columns are the ones taken first and the cheap ones
1706        // are what is left to fill in behind them.
1707        columns.sort_by_key(|&column| weight(&self.table.fields[column].ty));
1708        let queue = Mutex::new(columns);
1709        let pieces = std::thread::scope(|scope| {
1710            (0..workers)
1711                .map(|_| {
1712                    scope.spawn(|| {
1713                        let mut mine = Vec::new();
1714                        loop {
1715                            let taken = queue
1716                                .lock()
1717                                .map_err(|_| Error::internal("a native frequency worker panicked"))?
1718                                .pop();
1719                            let Some(column) = taken else { break };
1720                            mine.push((column, self.numeric_frequency(column)?));
1721                        }
1722                        Ok(mine)
1723                    })
1724                })
1725                .collect::<Vec<_>>()
1726                .into_iter()
1727                .map(|handle| {
1728                    handle
1729                        .join()
1730                        .map_err(|_| Error::internal("a native frequency worker panicked"))?
1731                })
1732                .collect::<Result<Vec<_>>>()
1733        })?;
1734        let mut frequencies = vec![None; self.table.fields.len()];
1735        for piece in pieces {
1736            for (column, summary) in piece {
1737                frequencies[column] = summary;
1738            }
1739        }
1740        Ok(frequencies)
1741    }
1742
1743    /// Writes the directory of the table this writer is on and says where it went.
1744    ///
1745    /// Everything [`Writer::finish`] used to do except the two writes that publish. Pulling it out
1746    /// is what lets a second table follow a first: the bytes of a closed table are complete and
1747    /// addressable while nothing yet points at them, and the pointer is the last write of the
1748    /// commit.
1749    ///
1750    /// # Errors
1751    ///
1752    /// If directory encoding or writing fails.
1753    fn close(&mut self) -> Result<Entry> {
1754        self.flush_pending()?;
1755        let mut stripes = std::mem::take(&mut self.order)
1756            .into_iter()
1757            .zip(std::mem::take(&mut self.table.stripes))
1758            .collect::<Vec<_>>();
1759        stripes.sort_by_key(|(order, _)| order.0);
1760        let mut previous: Option<(u64, u64)> = None;
1761        for ((first, last), _) in &stripes {
1762            if previous.is_some_and(|previous| previous >= *first) {
1763                return Err(invalid("chunks did not arrive in source order"));
1764            }
1765            previous = Some(*last);
1766        }
1767        self.table.stripes = stripes.into_iter().map(|(_, stripe)| stripe).collect();
1768        self.table.frequencies = self.numeric_frequencies()?;
1769        let dictionaries = std::mem::take(&mut self.dictionaries);
1770        let orders = rankings(&dictionaries)?;
1771        for (index, (dictionary, order)) in dictionaries.into_iter().zip(orders).enumerate() {
1772            let Some(dictionary) = dictionary else { continue };
1773            // A code nothing counted is a code no non-null row of this column holds, which is the
1774            // empty string a null was written as and nothing else, because a code is only ever made
1775            // by a row asking for one.
1776            self.table.distincts[index] =
1777                Some(dictionary.counts.iter().filter(|count| **count != 0).count() as u64);
1778            self.table.frequencies[index] = Some(code_frequency(&dictionary));
1779            let encoded = encode_global_dictionary(dictionary, &order)?;
1780            let offset = self.at;
1781            self.put(&encoded.index)?;
1782            self.put(&encoded.ranks)?;
1783            for block in &encoded.payload {
1784                self.put(block)?;
1785            }
1786            let payload_len =
1787                encoded.payload.iter().try_fold(0_usize, |len, block| len.checked_add(block.len()));
1788            let length = payload_len
1789                .and_then(|len| len.checked_add(encoded.index.len()))
1790                .and_then(|len| len.checked_add(encoded.ranks.len()))
1791                .ok_or_else(|| invalid("dictionary page length overflow"))?;
1792            self.table.dictionaries[index] = Some(Page {
1793                offset,
1794                length: u32::try_from(length)
1795                    .map_err(|_| invalid("dictionary page length overflow"))?,
1796                hash: checksum(&encoded.index),
1797            });
1798        }
1799        let directory = encode_directory(&self.table)?;
1800        if directory.len() > MAX_DIRECTORY {
1801            return Err(invalid("directory exceeds the configured bound"));
1802        }
1803        let offset = self.at;
1804        self.put(&directory)?;
1805        Ok(Entry {
1806            name: self.table.name.clone(),
1807            fields: self.table.fields.clone(),
1808            rows: self.table.rows,
1809            directory: Page {
1810                offset,
1811                length: u32::try_from(directory.len())
1812                    .map_err(|_| invalid("directory length overflow"))?,
1813                hash: checksum(&directory),
1814            },
1815        })
1816    }
1817
1818    /// Commits every table this writer has written and syncs the file before publishing its header
1819    /// slot.
1820    ///
1821    /// The table handed back is the one the writer was on, which is the last of them. Callers that
1822    /// wrote several already know the others, since they named them.
1823    ///
1824    /// # Errors
1825    ///
1826    /// If directory encoding, writing, or syncing fails.
1827    pub fn finish(mut self) -> Result<Table> {
1828        let entry = self.close()?;
1829        let mut tables = std::mem::take(&mut self.closed);
1830        tables.push(entry);
1831        let catalog = encode_catalog(&tables)?;
1832        if catalog.len() > MAX_DIRECTORY {
1833            return Err(invalid("catalog exceeds the configured bound"));
1834        }
1835        let offset = self.at;
1836        self.put(&catalog)?;
1837        // Every page and every table directory is on the disk before anything points at them. The
1838        // slot write below is what makes this generation the one a reader picks, so the order of
1839        // these two syncs is the whole of the commit.
1840        self.file.sync_all().map_err(io)?;
1841        let slot = Slot {
1842            offset,
1843            length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
1844            generation: self.generation,
1845            hash: checksum(&catalog),
1846        };
1847        // The one write that is not an append, and the last one. It goes back over the slot in the
1848        // header, so it names its offset rather than going through `put`, and `at` does not move.
1849        // Which of the two slots it is alternates with the generation, so the one naming the
1850        // generation before this is still intact and still valid until this write lands.
1851        write_at(&self.file, slot_offset(self.generation), &slot.bytes())?;
1852        self.file.sync_all().map_err(io)?;
1853        Ok(self.table)
1854    }
1855}
1856
1857/// Appends one run of bytes at `at` and moves it past them, answering where they went.
1858///
1859/// The append half of [`attach`], which cannot use [`Writer::put`] because it is not writing a
1860/// table. Every byte a section costs goes through here, so the offsets in an extent table come
1861/// from one place.
1862fn append(file: &File, at: &mut u64, bytes: &[u8]) -> Result<u64> {
1863    let offset = *at;
1864    write_at(file, offset, bytes)?;
1865    *at =
1866        at.checked_add(bytes.len() as u64).ok_or_else(|| invalid("native file length overflow"))?;
1867    Ok(offset)
1868}
1869
1870/// Writes one attachment's payload as extents and returns the entry that names it.
1871///
1872/// The split is by bytes, and the `first` of each extent is therefore a byte count. A section kind
1873/// whose extents should break on a row boundary instead will want to hand its extents over already
1874/// split; nothing needs that yet, and guessing at the shape of it now would be guessing.
1875fn write_section(
1876    file: &File,
1877    at: &mut u64,
1878    one: &section::Attachment<'_>,
1879    generation: u64,
1880) -> Result<Section> {
1881    if one.header_bytes as usize > one.bytes.len() {
1882        return Err(invalid("a section's header is longer than its payload"));
1883    }
1884    let mut extents = Vec::new();
1885    let mut first = 0_u64;
1886    for chunk in one.bytes.chunks(section::MAX_EXTENT as usize) {
1887        let offset = append(file, at, chunk)?;
1888        extents.push(section::Extent {
1889            offset,
1890            length: u32::try_from(chunk.len()).map_err(|_| invalid("extent length overflow"))?,
1891            hash: checksum(chunk),
1892            first,
1893        });
1894        first += chunk.len() as u64;
1895    }
1896    let mut table = Vec::with_capacity(extents.len() * section::EXTENT_BYTES);
1897    section::encode_extents(&extents, &mut table)?;
1898    // A payload of nothing is a section of no extents and no extent table, and its `extent_page`
1899    // is zero rather than the end of the file. Section 3.7 wants that entry to exist: it is how a
1900    // relationship that did not fit the budget is recorded as not built rather than forgotten.
1901    let extent_page = if table.is_empty() { 0 } else { append(file, at, &table)? };
1902    Ok(Section {
1903        kind: one.kind,
1904        id: one.id,
1905        generation,
1906        extents: u32::try_from(extents.len()).map_err(|_| invalid("too many extents"))?,
1907        extent_page,
1908        extent_bytes: u32::try_from(table.len()).map_err(|_| invalid("extent table overflow"))?,
1909        hash: checksum(&table),
1910        flags: one.flags,
1911        header_bytes: one.header_bytes,
1912    })
1913}
1914
1915/// Attaches graph sections to a table already committed in a file, without rewriting a page.
1916///
1917/// This is the second pass spec/graph/03-the-file-format.md section 3.8 asks for. A key map has to
1918/// exist before the link that uses it can be built, and it is built by reading the key column back,
1919/// so the structures of a table cannot be written during the load that wrote the table. They are
1920/// written afterwards, by this, and the file in between the two is a correct file that answers
1921/// every query more slowly.
1922///
1923/// Nothing is overwritten. The payloads, the extent tables, the new directory for this table and
1924/// the new catalog all go on the end of the file past the committed generation, and the last write
1925/// is the header slot, exactly as [`Writer::finish`] does it. So a crash anywhere in here leaves
1926/// the generation before it intact, and leaves unreferenced trailing bytes that the next commit
1927/// writes past.
1928///
1929/// An attachment replaces any section of the same kind and id, and every other section is carried
1930/// through untouched, including one whose kind this build does not know. The table's own generation
1931/// is carried through too, because attaching a section moves no row: see [`Table::generation`].
1932///
1933/// # Errors
1934///
1935/// If the file has no valid committed directory, is an older format than this build writes, holds
1936/// no table of that name, names a section whose payload cannot be written, or would end up naming
1937/// more sections than the format allows.
1938pub fn attach(
1939    path: impl AsRef<Path>,
1940    table: &str,
1941    attachments: &[section::Attachment<'_>],
1942) -> Result<Table> {
1943    let path = path.as_ref();
1944    let (_, size, slot, bytes, _) = slot_bytes(path)?;
1945    let mut entries = decode_catalog(&bytes, size)?;
1946    let at = entries
1947        .iter()
1948        .position(|entry| entry.name == table)
1949        .ok_or_else(|| invalid(&format!("the file holds no table called {table}")))?;
1950    let file = OpenOptions::new().write(true).read(true).open(path).map_err(io)?;
1951    let mut version = [0; 4];
1952    read_at(&file, 8, &mut version)?;
1953    let version = u32::from_le_bytes(version);
1954    // Readable is not the same as writable. A format 22 file has no section table, and giving its
1955    // directory one without moving the number in its header would leave a file that claims to be
1956    // format 22 and is not, which is worse than refusing. Rewriting it with this build is the
1957    // answer, and the format is at 0.3.x, so nobody has one of these that this project did not
1958    // just make.
1959    if version != FORMAT {
1960        return Err(invalid(&format!(
1961            "the file is format {version} and a graph section needs format {FORMAT}, so it has \
1962             to be written again"
1963        )));
1964    }
1965    let mut directory = vec![0; entries[at].directory.length as usize];
1966    read_at(&file, entries[at].directory.offset, &mut directory)?;
1967    if checksum(&directory) != entries[at].directory.hash {
1968        return Err(invalid(&format!("the directory of table {table} does not checksum")));
1969    }
1970    let mut held = decode_directory(&directory, size)?;
1971    let mut cursor = size;
1972    for one in attachments {
1973        let written = write_section(&file, &mut cursor, one, held.generation)?;
1974        held.sections.retain(|old| !(old.kind == one.kind && old.id == one.id));
1975        held.sections.push(written);
1976    }
1977    if held.sections.len() > MAX_SECTIONS {
1978        return Err(invalid("the table would name more sections than the bound allows"));
1979    }
1980    let encoded = encode_directory(&held)?;
1981    if encoded.len() > MAX_DIRECTORY {
1982        return Err(invalid("directory exceeds the configured bound"));
1983    }
1984    let offset = append(&file, &mut cursor, &encoded)?;
1985    entries[at].directory = Page {
1986        offset,
1987        length: u32::try_from(encoded.len()).map_err(|_| invalid("directory length overflow"))?,
1988        hash: checksum(&encoded),
1989    };
1990    let catalog = encode_catalog(&entries)?;
1991    if catalog.len() > MAX_DIRECTORY {
1992        return Err(invalid("catalog exceeds the configured bound"));
1993    }
1994    let offset = append(&file, &mut cursor, &catalog)?;
1995    file.sync_all().map_err(io)?;
1996    let generation =
1997        slot.generation.checked_add(1).ok_or_else(|| invalid("native file generation overflow"))?;
1998    let committed = Slot {
1999        offset,
2000        length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
2001        generation,
2002        hash: checksum(&catalog),
2003    };
2004    write_at(&file, slot_offset(generation), &committed.bytes())?;
2005    file.sync_all().map_err(io)?;
2006    Ok(held)
2007}
2008
2009/// Reads committed native column pages without holding the table in memory.
2010#[derive(Debug, Clone)]
2011pub struct Reader {
2012    file: Arc<File>,
2013    table: Arc<Table>,
2014    dictionaries: Arc<Vec<OnceLock<Arc<Vector>>>>,
2015    /// Held while a global dictionary is being opened, one per column.
2016    ///
2017    /// The [`OnceLock`] above says whether one has been opened, which is the question a reader that
2018    /// already has it needs answered and is free. It does not say whether one is being opened, and
2019    /// the difference matters because every worker of a scan wants the same dictionary at the same
2020    /// moment. Without this they all miss, all read the page, all verify it and all decode it, and
2021    /// all but one throw the answer away. ClickBench 38 reads the URL dictionary, which is 515,958
2022    /// entries, and was paying for it twice.
2023    loading: Arc<Vec<Mutex<()>>>,
2024    /// How many global dictionaries have been opened. A scan of a dictionary column should open its
2025    /// dictionary once however many workers it has, and the test that says so is the only thing
2026    /// keeping it that way.
2027    opened: Arc<AtomicUsize>,
2028    /// The membership sieves of one stripe of one column, by column and then by stripe, read the
2029    /// first time a probe asks about them. A query filters on one or two columns and never looks at
2030    /// the rest, so reading these at open would be the whole index for the sake of a fraction of it.
2031    sieves: Arc<Vec<Vec<SieveSlot>>>,
2032    /// The per part ranges of one stripe of one column, by column and then by stripe, read the
2033    /// first time something compares that column and kept after that.
2034    part_ranges: Arc<Vec<Vec<RangeSlot>>>,
2035    /// Which stripe and which part of it every part of the table is, by table wide part number.
2036    places: Arc<Vec<Place>>,
2037    cache: Arc<Vec<Mutex<Cached>>>,
2038    /// How many whole stripe pages have been read, which is what the sharing above is judged on. A
2039    /// scan of a column should read each of its stripes once however many workers it has.
2040    pages: Arc<AtomicUsize>,
2041    /// How many index sections have been read. A scan of a column should read each of its stripes
2042    /// once here too, and the test that says so is the only thing keeping it that way.
2043    indexes: Arc<AtomicUsize>,
2044    /// How many stripes of one column the page cache keeps. See [`CACHED_STRIPES_PER_COLUMN`] for
2045    /// what sets it and [`Reader::keep_stripes`] for who raises it.
2046    kept: Arc<AtomicUsize>,
2047    /// The file's size when it was opened, for [`Reader::layout`].
2048    size: u64,
2049    /// The committed directory's size, for [`Reader::layout`].
2050    directory: u64,
2051    /// What opening the file cost, which is a number rather than a claim.
2052    opening: Opening,
2053}
2054
2055/// What [`Reader::open`] read before it returned.
2056///
2057/// `spec/stats/04-in-memory.md` section 4.2 says opening a table reads the header and the directory
2058/// and nothing else, and once that document's statistics are in the file the tempting change is to
2059/// load a column summary or two on the way past, because they are small and the next query will
2060/// want them. A hundred milliseconds of that is a hundred milliseconds nobody asked for, and an
2061/// embedded database is opened by processes that are about to run one trivial query.
2062///
2063/// So the claim gets a number. Both of these are fixed by the schema and the stripe count and are
2064/// independent of how many rows the file holds, and the test that says so is what stops the
2065/// tempting change from landing quietly.
2066#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
2067pub struct Opening {
2068    /// How many times the file was read. The header, then each directory slot that looked valid
2069    /// enough to check, so three at the most.
2070    pub reads: u32,
2071    /// How many bytes those reads asked for.
2072    pub bytes: u64,
2073}
2074
2075/// What a reader has read, while it was being opened and since.
2076#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
2077pub struct Reads {
2078    /// What opening cost, before any query had been planned.
2079    pub opening: Opening,
2080    /// Whole stripe pages read since.
2081    pub pages: usize,
2082    /// Index sections read since.
2083    pub indexes: usize,
2084    /// Global dictionaries opened since. One per dictionary column that a query touched, however
2085    /// many workers touched it, which is a claim only a test can keep true.
2086    pub dictionaries: usize,
2087}
2088
2089/// Where one table wide part number lands.
2090#[derive(Debug, Clone, Copy)]
2091struct Place {
2092    stripe: u32,
2093    part: u32,
2094    rows: u32,
2095}
2096
2097/// One part's bytes inside one column page.
2098#[derive(Debug, Clone, Copy)]
2099struct PartSpan {
2100    start: usize,
2101    length: usize,
2102    hash: u64,
2103}
2104
2105/// What a reader holds for one stripe of one column.
2106///
2107/// The index is small and is loaded whether the caller wants the whole page or one part of it. The
2108/// page is loaded only by a scan, because a sparse fetch that wants a thousand rows out of sixty
2109/// four thousand would be reading sixty four times what it uses.
2110#[derive(Debug, Clone)]
2111struct CachedColumn {
2112    stripe: usize,
2113    index: Arc<Vec<PartSpan>>,
2114    page: Option<Arc<Vec<u8>>>,
2115}
2116
2117/// One column's stripes a reader holds, and which of them somebody is reading right now.
2118///
2119/// The pages are one slot per stripe of the table rather than a list of the ones being kept, so
2120/// finding a page is an index and not a walk. That matters because the walk happened under the
2121/// lock, once per part per column, and a scan that gives a whole stripe to each of thirty two
2122/// workers keeps enough pages that walking them was the longest thing the lock was held for. The
2123/// slots cost a pointer per stripe per column, which on the ClickBench file is eight kilobytes
2124/// against the forty megabytes of pages they point at. `order` is which of them are filled, oldest
2125/// first, because that is the one thing the slots cannot say by themselves.
2126///
2127/// `loading` is what keeps a scan from reading the same page once per worker. It is a list and not
2128/// a set because it holds at most one stripe per worker on the column and is walked far less often
2129/// than a hash of it would be built.
2130///
2131/// `index` is every index this reader has ever read for the column, one slot per stripe, and it is
2132/// never evicted. An index is a few hundred bytes and a page is a quarter of a megabyte, so the two
2133/// do not belong under the same budget. Riding in the page cache meant a worker that came back to a
2134/// stripe after its page had been evicted read the index again with it, which on the full
2135/// ClickBench file was about thirteen hundred reads out of a hundred and fourteen thousand.
2136#[derive(Debug, Default)]
2137struct Cached {
2138    pages: Vec<Option<Arc<Vec<u8>>>>,
2139    order: VecDeque<usize>,
2140    loading: Vec<usize>,
2141    index: Vec<Option<Arc<Vec<PartSpan>>>>,
2142}
2143
2144/// Stripes of one column a reader keeps the bytes of, when nobody has asked for more.
2145///
2146/// This has to hold at least as many stripes as a column has workers in it at once, or the workers
2147/// evict each other's pages and read them again. Four is what a scan that hands parts out in order
2148/// needs, because then every worker is within a few parts of every other and at most a couple of
2149/// stripes are open at a time. A scan that hands a whole stripe to each worker has one stripe open
2150/// per worker for the length of that stripe, and it says so with [`Reader::keep_stripes`] rather
2151/// than paying for sixteen slots on every table that is read one part at a time.
2152///
2153/// It multiplies by the page size, which is a quarter of a megabyte for a four byte column, and by
2154/// the number of columns a query touches.
2155const CACHED_STRIPES_PER_COLUMN: usize = 4;
2156
2157/// The sieves of one stripe of one column, once somebody has asked for them.
2158type SieveSlot = OnceLock<Arc<Vec<Option<Sieve>>>>;
2159
2160type RangeSlot = OnceLock<Arc<Vec<Range>>>;
2161
2162#[derive(Debug)]
2163struct NativeText {
2164    file: Arc<File>,
2165    /// How many values the dictionary holds.
2166    values: usize,
2167    /// Where each value ends inside its payload block, packed at `offset_bits` in runs of
2168    /// [`TEXT_OFFSET_RUN`].
2169    ///
2170    /// Ends rather than starts, because then a block of 1,024 values is 1,024 numbers rather than
2171    /// 1,025: the start of a value is the end of the one before it, and the first value of a block
2172    /// starts at zero by construction. Relative to the block rather than to the payload, because a
2173    /// reader decodes a whole block and slices it, so an offset into the payload is a number it
2174    /// would have to subtract a base from anyway.
2175    offsets: Vec<u8>,
2176    /// Bits one offset is packed at, which is what the largest block of this column spans and is the
2177    /// same for every block of it.
2178    offset_bits: usize,
2179    /// How many entries the sorted order has, which is the value count.
2180    ranks: usize,
2181    /// Where the sorted order starts in the file. It is read a block at a time and only when
2182    /// something searches it, so a query that never compares this column against a literal never
2183    /// touches it at all.
2184    rank_at: u64,
2185    /// Where each block of the sorted order ends, as a byte offset from `rank_at`. A block is packed
2186    /// at whatever width its own heads need, so unlike the entries it replaced its length is not
2187    /// arithmetic on the block number.
2188    rank_ends: Vec<u64>,
2189    rank_hashes: Vec<u64>,
2190    rank_blocks: Vec<OnceLock<Result<Vec<u8>>>>,
2191    /// Bits one code is packed at, which is what the value count needs and is the same for every
2192    /// block of the column.
2193    code_bits: usize,
2194    /// The sorted order turned round, built the first time a reader asks for it.
2195    ///
2196    /// Four bytes per value against the four the offsets already hold, so a column that has this is
2197    /// carrying half again what it carried before rather than something of a new order. It is built
2198    /// only when something asks, which is a grouped min or max over this column and nothing else,
2199    /// and that reader was going to read the payload of this column once per row otherwise.
2200    code_ranks: OnceLock<Option<Vec<u32>>>,
2201    payload: u64,
2202    /// Where each block of the payload ends in the file, as a byte offset from `payload`. The
2203    /// blocks are stored back to back, so a block starts where the one before it ended.
2204    ends: Vec<u64>,
2205    hashes: Vec<u64>,
2206    /// The payload, read and decoded a block at a time and kept after that.
2207    blocks: Vec<OnceLock<Result<Vec<u8>>>>,
2208    /// How many decoded payload bytes this column keeps before a sweep stops keeping what it reads.
2209    /// [`TEXT_KEEP_BUDGET`] everywhere but in the test of the ceiling.
2210    keep_budget: usize,
2211    /// Roughly how many decoded payload bytes are being kept, which is what [`TEXT_KEEP_BUDGET`]
2212    /// is measured against.
2213    ///
2214    /// Roughly, because two threads that keep the same block at the same time both add its length
2215    /// while [`OnceLock`] keeps one of the two. That makes the count read high and the budget bind
2216    /// a little early, which is the harmless direction, and it costs one relaxed add a block rather
2217    /// than a lock on the path every scan of a string column goes through.
2218    payload_kept: AtomicUsize,
2219    /// The boundaries this dictionary has already been searched for, by the value searched for.
2220    ///
2221    /// A search is the expensive thing this type does. It settles a probe on the stored head where
2222    /// it can and reads a value where it cannot, and reading a value decodes the payload block it
2223    /// sits in, so one search can cost several blocks. The thing that makes remembering worth it is
2224    /// that the same search comes back: a top N asks once a chunk whether anything left can beat its
2225    /// worst candidate, and the worst candidate settles long before the chunks run out.
2226    ///
2227    /// Shared across the instances of a scan rather than kept per instance, because each of them has
2228    /// its own worst candidate and all of them are searching the same dictionary. One lock per chunk
2229    /// is nothing next to a probe of a file.
2230    ///
2231    /// Bounded by [`TEXT_SEARCH_MEMO`] and emptied rather than evicted when it is full. What fills
2232    /// it is a top N improving its bound, which happens a few dozen times and then stops, so the
2233    /// bound is there for the filter that searches for a different literal every chunk rather than
2234    /// for anything this is meant to help.
2235    searched: Mutex<HashMap<Vec<u8>, (usize, bool)>>,
2236}
2237
2238/// How many searched for values a column's dictionary remembers the boundary of.
2239///
2240/// See [`NativeText::searched`]. Small because the case it is for repeats one value, not because a
2241/// larger one would be wrong.
2242const TEXT_SEARCH_MEMO: usize = 64;
2243
2244/// How many values of a dictionary go in one block of the payload.
2245///
2246/// The block is the unit the string cascade encodes, the unit a checksum covers, and the unit a
2247/// reader has to decode to get at a single value, so it is the one number the payload format turns
2248/// on. Blocking by values rather than by bytes is what keeps a value out of two blocks at once: the
2249/// block holding a code is `code / TEXT_PAYLOAD_VALUES` and nothing has to be stitched.
2250///
2251/// A probe on the five ClickBench columns that have a dictionary worth the name, written up on
2252/// #347, measured the ratio and the decode speed at 128, 256, 512, 1,024 and 4,096 values. Both get
2253/// better all the way up, because front coding and the LZ matcher have more to look back at and
2254/// because the per chunk setup is spread over more values. What stops it is the point read: a query
2255/// that wants ten values has to decode ten blocks, so the block is what a lookup costs. At 1,024
2256/// values a block is between 67 KB and 394 KB decoded across those five columns, and the ratios are
2257/// 2.3 to 4.5. Going up to 4,096 buys two to six percent more and makes a block as much as 1.5 MB.
2258/// Going down to 512 gives up five to nine percent.
2259const TEXT_PAYLOAD_VALUES: usize = 1024;
2260
2261/// How many decoded payload bytes one dictionary keeps before a sweep stops keeping what it reads.
2262///
2263/// A sweep of the whole dictionary decodes every block whatever it does, and the only question is
2264/// whether it hangs on to them. Keeping all of them is 4.2 GB on ClickBench `URL` at a hundred
2265/// million rows, which is what #997 was right to stop. Keeping none of them means the next query
2266/// asking the same thing decodes all of it again, and on the same column at a million rows that
2267/// took a `LIKE` from 2.7 ms to 16.2 ms, because the decode used to be paid once by a session and
2268/// is now paid by every statement in it. Neither end is the answer. A bound is.
2269///
2270/// So a sweep keeps what it decodes until the column is holding this much and decodes without
2271/// keeping after that. At a million rows the five ClickBench string columns decode to between 8 MB
2272/// and 85 MB, so they sit inside it and a repeated `LIKE` reads a decoded block rather than a
2273/// stored one. At a hundred million rows `URL` fills it and the rest of that column is read and
2274/// dropped, which is the old cost on the part that does not fit and none of the old footprint.
2275///
2276/// Two hundred and fifty six megabytes a column is a number and not a policy, and the policy is
2277/// what should replace it: this wants to be a buffer pool over the whole database, sized against
2278/// the memory limit the session was given, with the blocks of every column competing for it and the
2279/// least useful one evicted. That is F2 work. What is here is the part of it that can be written
2280/// without an eviction order, which is a ceiling.
2281const TEXT_KEEP_BUDGET: usize = 256 * 1024 * 1024;
2282
2283/// How many offsets go in one packed run.
2284///
2285/// A payload block holds 1,024 values and `bitpack::pack_tail` takes fewer than 1,024 at a time,
2286/// since a whole unit of that many belongs in the transposed layout instead. So the offsets of a
2287/// block go in two runs. Five hundred and twelve values at any width is a whole number of bytes, so
2288/// a run starts where a multiply says it does and nothing is padded.
2289const TEXT_OFFSET_RUN: usize = 512;
2290
2291/// Bytes at the front of a global dictionary index: the value count, the values a payload block
2292/// holds, the block count and the bits an offset is packed at.
2293const DICTIONARY_HEADER: usize = 16;
2294
2295/// How many entries of a dictionary's sorted order sit in one block that is read and checked as a
2296/// unit.
2297///
2298/// Five hundred and twelve entries is between two and three kilobytes on the ClickBench string
2299/// columns, which is well under a page. A binary search over half a million entries makes nineteen
2300/// probes, and the first ten land in ten different blocks while the last nine land in the one block
2301/// that holds the answer, so the whole search reads about thirty kilobytes of a megabyte of order. A
2302/// smaller block would save a little on the early probes, cost a checksum and an end list four times
2303/// as long, and give the heads less to share a base with. A larger one would read more than it uses
2304/// on every probe.
2305const TEXT_RANK_BLOCK: usize = 512;
2306
2307/// Bytes at the front of a rank block, which is the base of its heads and the width they are packed
2308/// at.
2309///
2310/// An entry used to be twelve bytes flat, eight for the head and four for the code, and on the five
2311/// ClickBench columns that have a dictionary worth the name that was 744 MB of a 12.2 GB file. Both
2312/// halves of it are nearly empty. The heads are the first eight bytes of the values in sorted order,
2313/// so a block of five hundred and twelve of them spans a tiny slice of the column, and on a column of
2314/// URLs they are all `http://w` and the block holds one distinct head. The codes are positions in a
2315/// dictionary of eighteen million, which is twenty five bits and not thirty two.
2316///
2317/// So a block now writes the smallest head in it, the bits the largest is above that, and the heads
2318/// and the codes packed at the width each needs. A block where every head agrees costs nine bytes
2319/// and the codes.
2320const RANK_BLOCK_HEADER: usize = size_of::<u64>() + 1;
2321
2322impl NativeText {
2323    /// One block of the payload, read and decoded the first time anything asks for a value in it.
2324    ///
2325    /// The bytes handed back are the values of the block laid end to end, which is what the offsets
2326    /// describe, so a caller slices it with the offsets it already has. Where the block sits in the
2327    /// file is the only thing the caller cannot work out for itself, because the stored form is
2328    /// shorter than the decoded one and by a different amount in every block.
2329    fn payload_block(&self, block: usize) -> Result<Option<&[u8]>> {
2330        let Some(slot) = self.blocks.get(block) else { return Ok(None) };
2331        let bytes = slot.get_or_init(|| self.decode_block(block)).as_ref().map_err(Clone::clone)?;
2332        Ok(Some(bytes.as_slice()))
2333    }
2334
2335    /// Reads and decodes one block of the payload, without deciding who keeps it.
2336    ///
2337    /// [`Self::payload_block`] keeps it forever, which is what a point read wants and what a walk
2338    /// of the whole dictionary must not do. Both call this and they differ in nothing else.
2339    fn decode_block(&self, block: usize) -> Result<Vec<u8>> {
2340        let start = if block == 0 { 0 } else { self.ends[block - 1] };
2341        let end = self.ends[block];
2342        let len = end
2343            .checked_sub(start)
2344            .ok_or_else(|| invalid("global dictionary block ends before it starts"))?;
2345        let mut stored = vec![
2346            0;
2347            usize::try_from(len).map_err(|_| invalid(
2348                "global dictionary block does not fit in memory"
2349            ))?
2350        ];
2351        read_at(&self.file, self.payload + start, &mut stored)?;
2352        if checksum(&stored) != self.hashes[block] {
2353            return Err(invalid("global dictionary payload checksum differs"));
2354        }
2355        let first = block * TEXT_PAYLOAD_VALUES;
2356        let last = (first + TEXT_PAYLOAD_VALUES).min(self.values);
2357        let want = self.end_within(last - 1)? as usize;
2358        let values = string::decode_flat(&stored)?;
2359        if values.len() != last - first {
2360            return Err(invalid("global dictionary block holds the wrong value count"));
2361        }
2362        let bytes = values.into_bytes();
2363        if bytes.len() != want {
2364            return Err(invalid("global dictionary block decodes to the wrong length"));
2365        }
2366        Ok(bytes)
2367    }
2368
2369    /// Where the value at `index` ends inside its payload block.
2370    fn end_within(&self, index: usize) -> Result<u32> {
2371        let run = index / TEXT_OFFSET_RUN;
2372        let bytes = self
2373            .offsets
2374            .get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)
2375            .ok_or_else(|| invalid("global dictionary offsets are short"))?;
2376        let end = bitpack::tail_at(bytes, self.offset_bits, index % TEXT_OFFSET_RUN)
2377            .map_err(|_| invalid("global dictionary offsets are short"))?;
2378        u32::try_from(end).map_err(|_| invalid("global dictionary offset is past the payload"))
2379    }
2380
2381    /// Where every value in `first..last` ends inside its payload block, in one pass over the runs.
2382    ///
2383    /// [`Self::end_within`] answers for one value and pays for it twice over: it shifts a window to
2384    /// the bit the value starts at, and the copy that fills that window is a length the compiler does
2385    /// not know, so it is a call to `memcpy` rather than a load. A sweep asked for two of those per
2386    /// value, one for the end and one for the start that is the end before it, and on the ClickBench
2387    /// `URL` dictionary of eighteen million that was most of the half second a `LIKE` over it took.
2388    ///
2389    /// [`bitpack::unpack_tail`] walks the run instead, which makes the window a fixed width and so
2390    /// an unaligned load, and reads the bit position off a counter. A run is five hundred and twelve
2391    /// values and a block is two of them, so a block of a thousand and twenty four values costs two
2392    /// calls here and nothing per value.
2393    fn ends_within(&self, first: usize, last: usize) -> Result<Vec<u64>> {
2394        let mut ends = Vec::with_capacity(last.saturating_sub(first));
2395        let mut at = first;
2396        while at < last {
2397            let run = at / TEXT_OFFSET_RUN;
2398            let stop = ((run + 1) * TEXT_OFFSET_RUN).min(last);
2399            let held = self.values.saturating_sub(run * TEXT_OFFSET_RUN).min(TEXT_OFFSET_RUN);
2400            let bytes = self
2401                .offsets
2402                .get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)
2403                .ok_or_else(|| invalid("global dictionary offsets are short"))?;
2404            let run_ends = bitpack::unpack_tail(bytes, self.offset_bits, held)
2405                .map_err(|_| invalid("global dictionary offsets are short"))?;
2406            let within = run_ends
2407                .get(at % TEXT_OFFSET_RUN..stop - run * TEXT_OFFSET_RUN)
2408                .ok_or_else(|| invalid("global dictionary offsets are short"))?;
2409            ends.extend_from_slice(within);
2410            at = stop;
2411        }
2412        Ok(ends)
2413    }
2414
2415    /// Where the value at `index` starts inside its payload block, which is where the value before
2416    /// it ended unless it is the first of the block.
2417    fn start_within(&self, index: usize) -> Result<u32> {
2418        if index % TEXT_PAYLOAD_VALUES == 0 { Ok(0) } else { self.end_within(index - 1) }
2419    }
2420
2421    /// Where the value at `index` starts and ends inside its payload block.
2422    ///
2423    /// The two offsets sit next to each other in the same run unless the value opens one, and a run
2424    /// of seventeen bit offsets, which is what a block of a thousand strings needs, puts a pair of
2425    /// them inside one eight byte load. So the common case reads the packed bytes once rather than
2426    /// twice and does the bounds arithmetic once. This is asked once per string a text column hands
2427    /// out, and on ClickBench 27 the two reads together were a quarter of the query.
2428    fn span_within(&self, index: usize) -> Result<(u32, u32)> {
2429        let within = index % TEXT_OFFSET_RUN;
2430        let (start, end) = if within == 0 {
2431            (self.start_within(index)?, self.end_within(index)?)
2432        } else {
2433            let run = index / TEXT_OFFSET_RUN;
2434            let bytes = self
2435                .offsets
2436                .get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)
2437                .ok_or_else(|| invalid("global dictionary offsets are short"))?;
2438            let (start, end) = bitpack::tail_pair(bytes, self.offset_bits, within)
2439                .map_err(|_| invalid("global dictionary offsets are short"))?;
2440            let ends = u32::try_from(end)
2441                .map_err(|_| invalid("global dictionary offset is past the payload"))?;
2442            let starts = u32::try_from(start)
2443                .map_err(|_| invalid("global dictionary offset is past the payload"))?;
2444            (starts, ends)
2445        };
2446        if start > end {
2447            return Err(invalid("global dictionary value ends before it starts"));
2448        }
2449        Ok((start, end))
2450    }
2451
2452    /// The block of the sorted order that holds `rank`, and where in it that rank sits.
2453    ///
2454    /// The block is read from the file and checked against the hash the index carries for it the
2455    /// first time anything asks, and kept after that, the same way a payload block is. A search
2456    /// makes about as many probes as the order has bits, so the whole search reads a handful of
2457    /// these and never the rest.
2458    fn rank_parts(&self, rank: usize) -> Result<(&[u8], usize)> {
2459        let slot = self
2460            .rank_blocks
2461            .get(rank / TEXT_RANK_BLOCK)
2462            .ok_or_else(|| invalid("global dictionary rank is past the order"))?;
2463        let block = slot
2464            .get_or_init(|| {
2465                let which = rank / TEXT_RANK_BLOCK;
2466                let start = if which == 0 { 0 } else { self.rank_ends[which - 1] };
2467                let end = self.rank_ends[which];
2468                let mut bytes = vec![0; (end - start) as usize];
2469                read_at(&self.file, self.rank_at + start, &mut bytes)?;
2470                if checksum(&bytes)
2471                    != *self
2472                        .rank_hashes
2473                        .get(rank / TEXT_RANK_BLOCK)
2474                        .ok_or_else(|| invalid("global dictionary rank block has no checksum"))?
2475                {
2476                    return Err(invalid("global dictionary rank checksum differs"));
2477                }
2478                Ok(bytes)
2479            })
2480            .as_ref()
2481            .map_err(Clone::clone)?;
2482        Ok((block.as_slice(), rank % TEXT_RANK_BLOCK))
2483    }
2484
2485    /// The first eight bytes of the value at `rank`, as the integer a comparison reads.
2486    fn head_at(&self, rank: usize) -> Result<u64> {
2487        let (block, within) = self.rank_parts(rank)?;
2488        let (base, width, packed) = rank_heads(block)?;
2489        let above = bitpack::tail_at(packed, width, within)
2490            .map_err(|_| invalid("global dictionary rank block is short of heads"))?;
2491        Ok(base.wrapping_add(above))
2492    }
2493
2494    /// The packed codes of one rank block, which follow the heads on the next byte boundary.
2495    fn rank_codes<'block>(&self, block: &'block [u8], count: usize) -> Result<&'block [u8]> {
2496        let (_, width, packed) = rank_heads(block)?;
2497        packed
2498            .get(bitpack::tail_len(count, width)..)
2499            .ok_or_else(|| invalid("global dictionary rank block is short of codes"))
2500    }
2501
2502    /// How many entries the block holding `rank` has, which is a full block except at the end.
2503    fn rank_block_len(&self, rank: usize) -> usize {
2504        let first = rank / TEXT_RANK_BLOCK * TEXT_RANK_BLOCK;
2505        TEXT_RANK_BLOCK.min(self.ranks - first)
2506    }
2507}
2508
2509/// The base, the width and the packed bytes of one rank block's heads.
2510fn rank_heads(block: &[u8]) -> Result<(u64, usize, &[u8])> {
2511    let header = block
2512        .get(..RANK_BLOCK_HEADER)
2513        .ok_or_else(|| invalid("global dictionary rank block is short"))?;
2514    let base = u64::from_le_bytes(header[..8].try_into().expect("eight bytes"));
2515    let width = header[8] as usize;
2516    if width > 64 {
2517        return Err(invalid("global dictionary rank block packs heads past a word"));
2518    }
2519    Ok((base, width, &block[RANK_BLOCK_HEADER..]))
2520}
2521
2522/// Bits one offset of a dictionary takes, which is what its widest payload block spans.
2523///
2524/// One width for the whole column rather than one a block. A block is 1,024 values of the same
2525/// column, so the blocks of a column are within a factor of two of each other on every ClickBench
2526/// string column, and a width a block would save a fraction of a bit and cost a byte a block plus
2527/// the arithmetic that finds where a block starts.
2528fn offset_width(offsets: &[u32]) -> usize {
2529    let values = offsets.len() - 1;
2530    let mut span = 0;
2531    for first in (0..values).step_by(TEXT_PAYLOAD_VALUES) {
2532        let last = (first + TEXT_PAYLOAD_VALUES).min(values);
2533        span = span.max(offsets[last] - offsets[first]);
2534    }
2535    (u32::BITS - span.leading_zeros()) as usize
2536}
2537
2538/// How many bytes `values` offsets take at `bits`, which is what the reader has to know before it
2539/// has read any of them.
2540fn offset_bytes(values: usize, bits: usize) -> usize {
2541    let full = values / TEXT_OFFSET_RUN;
2542    let rest = values % TEXT_OFFSET_RUN;
2543    full * TEXT_OFFSET_RUN / 8 * bits + bitpack::tail_len(rest, bits)
2544}
2545
2546/// The end of every value within its payload block, packed a run at a time.
2547fn encode_offsets(offsets: &[u32], bits: usize, out: &mut Vec<u8>) -> Result<()> {
2548    let values = offsets.len() - 1;
2549    let mut run = Vec::with_capacity(TEXT_OFFSET_RUN);
2550    for first in (0..values).step_by(TEXT_OFFSET_RUN) {
2551        let last = (first + TEXT_OFFSET_RUN).min(values);
2552        let base = offsets[first / TEXT_PAYLOAD_VALUES * TEXT_PAYLOAD_VALUES];
2553        run.clear();
2554        run.extend((first..last).map(|value| u64::from(offsets[value + 1] - base)));
2555        bitpack::pack_tail(&run, bits, out)
2556            .map_err(|_| invalid("global dictionary offsets do not pack"))?;
2557    }
2558    Ok(())
2559}
2560
2561/// How many bits a code of a dictionary of `values` entries takes.
2562fn code_width(values: usize) -> usize {
2563    match u64::try_from(values).unwrap_or(u64::MAX) {
2564        0 | 1 => 0,
2565        last => (u64::BITS - (last - 1).leading_zeros()) as usize,
2566    }
2567}
2568
2569impl TextSource for NativeText {
2570    fn len(&self) -> usize {
2571        self.values
2572    }
2573
2574    fn bytes_at(&self, index: usize) -> Result<Option<&[u8]>> {
2575        if index >= self.values {
2576            return Ok(None);
2577        }
2578        let (start, end) = self.span_within(index)?;
2579        if start == end {
2580            return Ok(Some(&[]));
2581        }
2582        // A block holds a fixed number of values rather than a fixed number of bytes, so the value
2583        // is in one block and the offsets already say where in it.
2584        let block = index / TEXT_PAYLOAD_VALUES;
2585        let Some(bytes) = self.payload_block(block)? else { return Ok(None) };
2586        Ok(bytes.get(start as usize..end as usize))
2587    }
2588
2589    fn bytes_len_at(&self, index: usize) -> Result<Option<usize>> {
2590        if index >= self.values {
2591            return Ok(None);
2592        }
2593        let (start, end) = self.span_within(index)?;
2594        Ok(Some((end - start) as usize))
2595    }
2596
2597    /// The rest of the block holding `first`, decoded into a buffer that may die with the call.
2598    ///
2599    /// A block is the unit this format decodes, so a walk that wants every value is going to decode
2600    /// every block whatever it does. The question is whether it keeps them, and both answers are
2601    /// wrong on their own. [`Self::payload_block`] keeps every block it is asked for, so a reader
2602    /// that walked the whole dictionary through `bytes_at` ended up holding the whole dictionary
2603    /// decoded, 4.2 GB on ClickBench `URL`. Keeping none of them makes the next statement asking
2604    /// the same question decode all of it again, which on the same column at a million rows is a
2605    /// `LIKE` going from 2.7 ms to 16.2 ms.
2606    ///
2607    /// So a sweep keeps what it decodes while the column is under [`TEXT_KEEP_BUDGET`] and drops it
2608    /// after that. A block already in hand is used where it is there and costs nothing either way.
2609    fn sweep(
2610        &self,
2611        first: usize,
2612        limit: usize,
2613        body: &mut dyn FnMut(usize, &[u8]) -> Result<()>,
2614    ) -> Result<usize> {
2615        let limit = limit.min(self.values);
2616        if first >= limit {
2617            return Ok(first);
2618        }
2619        let block = first / TEXT_PAYLOAD_VALUES;
2620        let last = ((block + 1) * TEXT_PAYLOAD_VALUES).min(limit);
2621        let decoded;
2622        let bytes: &[u8] = match self.blocks.get(block).and_then(OnceLock::get) {
2623            Some(Ok(kept)) => kept,
2624            _ if self.payload_kept.load(Atomic::Relaxed) < self.keep_budget => {
2625                let kept = self
2626                    .payload_block(block)?
2627                    .ok_or_else(|| invalid("global dictionary block is past the payload"))?;
2628                self.payload_kept.fetch_add(kept.len(), Atomic::Relaxed);
2629                kept
2630            }
2631            _ => {
2632                decoded = self.decode_block(block)?;
2633                &decoded
2634            }
2635        };
2636        let ends = self.ends_within(first, last)?;
2637        if ends.len() != last - first {
2638            return Err(invalid("global dictionary offsets are short"));
2639        }
2640        let mut start = u64::from(self.start_within(first)?);
2641        // row at a time: the caller is handed one value after another, and what it does with one is
2642        // its own business, so there is no shape here for anything but a walk.
2643        for (index, &end) in (first..last).zip(&ends) {
2644            let value = usize::try_from(start)
2645                .ok()
2646                .zip(usize::try_from(end).ok())
2647                .and_then(|(from, to)| bytes.get(from..to))
2648                .ok_or_else(|| invalid("global dictionary value is past its block"))?;
2649            body(index, value)?;
2650            start = end;
2651        }
2652        Ok(last)
2653    }
2654
2655    fn ranks(&self) -> Option<usize> {
2656        (self.ranks > 0).then_some(self.ranks)
2657    }
2658
2659    /// The boundary for `wanted`, out of [`Self::searched`] where it is there and put there where
2660    /// it is not.
2661    ///
2662    /// The lock is held over the search rather than dropped and taken again, so that two threads
2663    /// asking for the same value at the same time do the work once between them. That is the shape
2664    /// the scan actually arrives in: sixteen instances of a top N, all reading the same column, all
2665    /// improving their bound over the same early chunks.
2666    fn below(&self, ranks: usize, wanted: &[u8]) -> Result<(usize, bool)> {
2667        let mut memo = self.searched.lock().map_err(|_| invalid("a poisoned dictionary search"))?;
2668        if let Some(&answer) = memo.get(wanted) {
2669            return Ok(answer);
2670        }
2671        let answer = search_below(self, ranks, wanted)?;
2672        if memo.len() >= TEXT_SEARCH_MEMO {
2673            memo.clear();
2674        }
2675        memo.insert(wanted.to_vec(), answer);
2676        Ok(answer)
2677    }
2678
2679    fn compare_rank(&self, rank: usize, wanted: &[u8]) -> Result<Ordering> {
2680        // The head settles the probe unless the two values start with the same eight bytes, and
2681        // only then is a value read. On a column of URLs that is the difference between a search
2682        // that touches one block of the payload and a search that touches nineteen of them.
2683        let settled = self.head_at(rank)?.cmp(&head(wanted));
2684        if settled != Ordering::Equal {
2685            return Ok(settled);
2686        }
2687        let code = self.code_at_rank(rank)?;
2688        let bytes = self
2689            .bytes_at(code as usize)?
2690            .ok_or_else(|| invalid("global dictionary order names a code it does not have"))?;
2691        Ok(bytes.cmp(wanted))
2692    }
2693
2694    fn code_at_rank(&self, rank: usize) -> Result<u32> {
2695        let (block, within) = self.rank_parts(rank)?;
2696        let codes = self.rank_codes(block, self.rank_block_len(rank))?;
2697        let code = bitpack::tail_at(codes, self.code_bits, within)
2698            .map_err(|_| invalid("global dictionary rank block is short of codes"))?;
2699        let code = u32::try_from(code)
2700            .map_err(|_| invalid("global dictionary order names a code it does not have"))?;
2701        if code as usize >= self.len() {
2702            return Err(invalid("global dictionary order names a code it does not have"));
2703        }
2704        Ok(code)
2705    }
2706
2707    fn code_ranks(&self) -> Option<&[u32]> {
2708        // The order is a permutation of the positions, so inverting it needs every position to be
2709        // named exactly once. Anything else and the slice would have holes, and a caller indexing
2710        // it by a code would read a rank that belongs to nothing.
2711        if self.ranks == 0 || self.ranks != self.len() {
2712            return None;
2713        }
2714        self.code_ranks
2715            .get_or_init(|| {
2716                let mut ranks = vec![u32::MAX; self.ranks];
2717                // A block at a time rather than a rank at a time, because reading it per rank pays
2718                // for the bounds check, the division and the lock on every one of them.
2719                for first in (0..self.ranks).step_by(TEXT_RANK_BLOCK) {
2720                    let (block, _) = self.rank_parts(first).ok()?;
2721                    let count = self.rank_block_len(first);
2722                    let codes = self.rank_codes(block, count).ok()?;
2723                    for (within, code) in bitpack::unpack_tail(codes, self.code_bits, count)
2724                        .ok()?
2725                        .into_iter()
2726                        .enumerate()
2727                    {
2728                        let code = usize::try_from(code).ok()?;
2729                        *ranks.get_mut(code)? = u32::try_from(first + within).ok()?;
2730                    }
2731                }
2732                if ranks.contains(&u32::MAX) {
2733                    return None;
2734                }
2735                Some(ranks)
2736            })
2737            .as_deref()
2738    }
2739
2740    fn footprint(&self) -> usize {
2741        self.offsets.capacity()
2742            + self
2743                .code_ranks
2744                .get()
2745                .and_then(Option::as_ref)
2746                .map_or(0, |ranks| ranks.capacity() * size_of::<u32>())
2747            + self.rank_hashes.capacity() * size_of::<u64>()
2748            + self.rank_ends.capacity() * size_of::<u64>()
2749            + self.rank_blocks.capacity() * size_of::<OnceLock<Result<Vec<u8>>>>()
2750            + self
2751                .rank_blocks
2752                .iter()
2753                .filter_map(OnceLock::get)
2754                .filter_map(|result| result.as_ref().ok())
2755                .map(Vec::capacity)
2756                .sum::<usize>()
2757            + self.blocks.capacity() * size_of::<OnceLock<Result<Vec<u8>>>>()
2758            + self.hashes.capacity() * size_of::<u64>()
2759            + self.ends.capacity() * size_of::<u64>()
2760            + self
2761                .blocks
2762                .iter()
2763                .filter_map(OnceLock::get)
2764                .filter_map(|result| result.as_ref().ok())
2765                .map(Vec::capacity)
2766                .sum::<usize>()
2767    }
2768}
2769
2770/// Every table wide part number in order, with the stripe it belongs to.
2771fn places(table: &Table) -> Result<Vec<Place>> {
2772    let mut places = Vec::with_capacity(table.stripes.len().saturating_mul(STRIPE_PARTS));
2773    for (at, stripe) in table.stripes.iter().enumerate() {
2774        let index = u32::try_from(at).map_err(|_| invalid("too many stripes"))?;
2775        for (part, &rows) in stripe.parts.iter().enumerate() {
2776            places.push(Place {
2777                stripe: index,
2778                part: u32::try_from(part).map_err(|_| invalid("too many parts in a stripe"))?,
2779                rows,
2780            });
2781        }
2782    }
2783    Ok(places)
2784}
2785
2786/// Reads one column's section of a stripe's index page.
2787///
2788/// The section carries its own checksum, so a reader that wants one column out of a hundred and
2789/// five preads a few hundred bytes and still knows that what it got is what was written.
2790fn read_index(file: &File, stripe: &Stripe, column: usize) -> Result<Vec<PartSpan>> {
2791    let parts = stripe.parts.len();
2792    let section = index_section(parts)?;
2793    let at = column.checked_mul(section).ok_or_else(|| invalid("index page offset overflow"))?;
2794    let end = at.checked_add(section).ok_or_else(|| invalid("index page offset overflow"))?;
2795    if end > stripe.index.length as usize {
2796        return Err(invalid("index page is shorter than its columns"));
2797    }
2798    let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
2799    let mut bytes = vec![0; section];
2800    let offset = stripe
2801        .index
2802        .offset
2803        .checked_add(at as u64)
2804        .ok_or_else(|| invalid("index page offset overflow"))?;
2805    read_at(file, offset, &mut bytes)?;
2806    let entries = section - size_of::<u64>();
2807    let stored = u64::from_le_bytes(bytes[entries..].try_into().expect("eight bytes"));
2808    if checksum(&bytes[..entries]) != stored {
2809        // With where it was read from, because the two ways this fires look identical from the
2810        // message alone: a file somebody damaged, and a file we wrote to the wrong offset.
2811        return Err(invalid(&format!(
2812            "index page section checksum differs, column {column} of {parts} parts at {offset}, \
2813             wanted {stored:016x} and got {:016x}",
2814            checksum(&bytes[..entries]),
2815        )));
2816    }
2817    let mut spans = Vec::with_capacity(parts);
2818    let mut start = 0_usize;
2819    for part in 0..parts {
2820        let at = part * INDEX_ENTRY;
2821        let length = u32::from_le_bytes(bytes[at..at + 4].try_into().expect("four bytes")) as usize;
2822        let hash = u64::from_le_bytes(bytes[at + 4..at + 12].try_into().expect("eight bytes"));
2823        spans.push(PartSpan { start, length, hash });
2824        start = start.checked_add(length).ok_or_else(|| invalid("column page length overflow"))?;
2825    }
2826    if start != page.length as usize {
2827        return Err(invalid("column page length differs from its index"));
2828    }
2829    Ok(spans)
2830}
2831
2832/// One part's bytes out of a whole column page.
2833fn part_bytes(page: &[u8], span: PartSpan) -> Result<&[u8]> {
2834    let end = span.start.checked_add(span.length).ok_or_else(|| invalid("part range overflow"))?;
2835    page.get(span.start..end).ok_or_else(|| invalid("part exceeds its column page"))
2836}
2837
2838/// Puts one stripe of one column in the cache, dropping the stripe that has been there longest.
2839///
2840/// The index goes in its own slot and stays. Only the page is under the budget, and `kept` is how
2841/// many pages that budget is.
2842fn remember(cached: &mut Cached, held: &CachedColumn, kept: usize) {
2843    if let Some(slot) = cached.index.get_mut(held.stripe) {
2844        if slot.is_none() {
2845            *slot = Some(Arc::clone(&held.index));
2846        }
2847    }
2848    let Some(page) = held.page.clone() else { return };
2849    let Some(slot) = cached.pages.get_mut(held.stripe) else { return };
2850    if slot.is_none() {
2851        cached.order.push_back(held.stripe);
2852    }
2853    *slot = Some(page);
2854    while cached.order.len() > kept.max(1) {
2855        let Some(oldest) = cached.order.pop_front() else { break };
2856        if let Some(slot) = cached.pages.get_mut(oldest) {
2857            *slot = None;
2858        }
2859    }
2860}
2861
2862/// Every table a native file holds, without the directory of any of them.
2863///
2864/// This is what opening a database reads. It is the small level of the directory, so the cost is
2865/// proportional to how many tables there are rather than to how much data they hold, and a session
2866/// that touches two tables of eight decodes two table directories.
2867///
2868/// The file handle is shared with every reader this hands out. Eight tables in one file is one open
2869/// file descriptor, not eight, which is the other thing one file buys over a file per table.
2870#[derive(Debug, Clone)]
2871pub struct Catalog {
2872    file: Arc<File>,
2873    size: u64,
2874    entries: Arc<Vec<Entry>>,
2875    opening: Opening,
2876}
2877
2878impl Catalog {
2879    /// Reads the highest valid catalog slot and nothing under it.
2880    ///
2881    /// # Errors
2882    ///
2883    /// If the file has no valid committed catalog or a catalog pointer is out of bounds.
2884    pub fn open(path: impl AsRef<Path>) -> Result<Self> {
2885        let (file, size, _, bytes, opening) = slot_bytes(path)?;
2886        let entries = decode_catalog(&bytes, size)?;
2887        Ok(Self { file: Arc::new(file), size, entries: Arc::new(entries), opening })
2888    }
2889
2890    /// The tables in the file, in the order they were written.
2891    pub fn names(&self) -> impl ExactSizeIterator<Item = &str> {
2892        self.entries.iter().map(|entry| entry.name.as_str())
2893    }
2894
2895    /// The same tables with how many rows each of them holds.
2896    ///
2897    /// The names alone answer which tables the file has, which is what a checkpoint needs to know.
2898    /// A load asks a second question: whether a table already in the file is really in the way of
2899    /// the one it wants to write. A table with no rows is not, because it has no pages the next
2900    /// generation would have to carry, so the count has to come out of the catalog beside the name.
2901    pub fn rows(&self) -> impl ExactSizeIterator<Item = (&str, usize)> {
2902        self.entries.iter().map(|entry| (entry.name.as_str(), entry.rows))
2903    }
2904
2905    /// How many tables the file holds.
2906    #[must_use]
2907    pub fn len(&self) -> usize {
2908        self.entries.len()
2909    }
2910
2911    /// Whether the file holds no table at all, which is what [`Writer::empty`] writes and what a
2912    /// database somebody dropped the last table out of comes back as.
2913    #[must_use]
2914    pub fn is_empty(&self) -> bool {
2915        self.entries.is_empty()
2916    }
2917
2918    /// Opens one table by name, decoding its directory now.
2919    ///
2920    /// # Errors
2921    ///
2922    /// If there is no table by that name, or its directory is torn or points outside the file.
2923    pub fn table(&self, name: &str) -> Result<Reader> {
2924        let entry = self
2925            .entries
2926            .iter()
2927            .find(|entry| entry.name == name)
2928            .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
2929        let mut bytes = vec![0; entry.directory.length as usize];
2930        read_at(&self.file, entry.directory.offset, &mut bytes)?;
2931        if checksum(&bytes) != entry.directory.hash {
2932            return Err(invalid(&format!("the directory of table {name} does not checksum")));
2933        }
2934        let mut opening = self.opening;
2935        opening.reads += 1;
2936        opening.bytes += u64::from(entry.directory.length);
2937        Reader::build(
2938            Arc::clone(&self.file),
2939            self.size,
2940            decode_directory(&bytes, self.size)?,
2941            u64::from(entry.directory.length),
2942            opening,
2943        )
2944    }
2945}
2946
2947/// Where the slot naming `generation` goes, which is the one the generation before it did not use.
2948///
2949/// Generation 1 takes the slot at 16, so a file written once is byte for byte the file this wrote
2950/// before there was a second generation to write.
2951fn slot_offset(generation: u64) -> u64 {
2952    16 + (generation - 1) % 2 * SLOT_BYTES as u64
2953}
2954
2955/// The header and the bytes the highest valid slot points at.
2956///
2957/// Both levels of the directory are reached this way, so the magic check, the version check and the
2958/// choice between the two slots live here rather than being written out twice.
2959fn slot_bytes(path: impl AsRef<Path>) -> Result<(File, u64, Slot, Vec<u8>, Opening)> {
2960    let mut file = File::open(path).map_err(io)?;
2961    let size = file.metadata().map_err(io)?.len();
2962    if size < HEADER {
2963        return Err(invalid("file is shorter than its header"));
2964    }
2965    let mut header = [0; HEADER as usize];
2966    file.read_exact(&mut header).map_err(io)?;
2967    let mut opening = Opening { reads: 1, bytes: HEADER };
2968    let version = u32::from_le_bytes([header[8], header[9], header[10], header[11]]);
2969    // The two halves are worth telling apart. A wrong magic is a file that was never ours and
2970    // the answer is to look at the path. A wrong version is our own file from another build,
2971    // and the number this build wants is the only thing that tells the reader whether to
2972    // rebuild the file or to go back to the binary that wrote it.
2973    if &header[..8] != MAGIC {
2974        return Err(invalid("the header does not begin with a rudb native magic"));
2975    }
2976    if !READABLE.contains(&version) {
2977        return Err(invalid(&format!(
2978            "the file is format {version} and this build reads format {FORMAT}, so it has to \
2979                 be written again"
2980        )));
2981    }
2982    let mut selected = None;
2983    for start in [16, 16 + SLOT_BYTES] {
2984        let slot = Slot::read(&header[start..start + SLOT_BYTES]);
2985        if slot.generation == 0 || slot.length == 0 || slot.length as usize > MAX_DIRECTORY {
2986            continue;
2987        }
2988        let Some(end) = slot.offset.checked_add(u64::from(slot.length)) else { continue };
2989        if slot.offset < HEADER || end > size {
2990            continue;
2991        }
2992        let mut bytes = vec![0; slot.length as usize];
2993        file.seek(SeekFrom::Start(slot.offset)).map_err(io)?;
2994        file.read_exact(&mut bytes).map_err(io)?;
2995        opening.reads += 1;
2996        opening.bytes += u64::from(slot.length);
2997        if checksum(&bytes) == slot.hash
2998            && selected
2999                .as_ref()
3000                .is_none_or(|(old, _): &(Slot, Vec<u8>)| old.generation < slot.generation)
3001        {
3002            selected = Some((slot, bytes));
3003        }
3004    }
3005    let (slot, bytes) = selected.ok_or_else(|| invalid("no committed directory slot is valid"))?;
3006    Ok((file, size, slot, bytes, opening))
3007}
3008
3009impl Reader {
3010    /// Opens a file that holds exactly one table.
3011    ///
3012    /// # Errors
3013    ///
3014    /// If the file has no valid committed directory, a directory pointer is out of bounds, or the
3015    /// file holds more than one table, which is a file that has to be opened by name.
3016    pub fn open(path: impl AsRef<Path>) -> Result<Self> {
3017        let catalog = Catalog::open(path)?;
3018        let mut names = catalog.names();
3019        let name = names.next().ok_or_else(|| invalid("the file holds no table"))?.to_string();
3020        if names.next().is_some() {
3021            return Err(invalid(
3022                "the file holds more than one table, so it has to be opened by name",
3023            ));
3024        }
3025        catalog.table(&name)
3026    }
3027
3028    /// Builds a reader over one decoded table directory.
3029    fn build(
3030        file: Arc<File>,
3031        size: u64,
3032        table: Table,
3033        directory: u64,
3034        opening: Opening,
3035    ) -> Result<Self> {
3036        let places = places(&table)?;
3037        let dictionaries = (0..table.fields.len()).map(|_| OnceLock::new()).collect();
3038        let table_fields = table.fields.len();
3039        let stripes = table.stripes.len();
3040        let cache = (0..table.fields.len())
3041            .map(|_| {
3042                Mutex::new(Cached {
3043                    pages: (0..stripes).map(|_| None).collect(),
3044                    index: (0..stripes).map(|_| None).collect(),
3045                    ..Cached::default()
3046                })
3047            })
3048            .collect::<Vec<_>>();
3049        let sieves: Vec<Vec<SieveSlot>> = (0..table.fields.len())
3050            .map(|_| table.stripes.iter().map(|_| OnceLock::new()).collect())
3051            .collect();
3052        let part_ranges: Vec<Vec<RangeSlot>> = (0..table.fields.len())
3053            .map(|_| table.stripes.iter().map(|_| OnceLock::new()).collect())
3054            .collect();
3055        Ok(Self {
3056            file,
3057            table: Arc::new(table),
3058            dictionaries: Arc::new(dictionaries),
3059            loading: Arc::new((0..table_fields).map(|_| Mutex::new(())).collect()),
3060            opened: Arc::new(AtomicUsize::new(0)),
3061            sieves: Arc::new(sieves),
3062            part_ranges: Arc::new(part_ranges),
3063            places: Arc::new(places),
3064            cache: Arc::new(cache),
3065            pages: Arc::new(AtomicUsize::new(0)),
3066            indexes: Arc::new(AtomicUsize::new(0)),
3067            kept: Arc::new(AtomicUsize::new(CACHED_STRIPES_PER_COLUMN)),
3068            size,
3069            directory,
3070            opening,
3071        })
3072    }
3073
3074    /// What this reader has read so far, and what opening it cost.
3075    ///
3076    /// Public because the claim of `spec/stats/04-in-memory.md` section 4.2 is about this number
3077    /// and a claim nobody can check is a comment. A caller that wants to know whether opening a
3078    /// file touched the data asks here, and gets an answer that does not depend on what the page
3079    /// cache happened to hold.
3080    #[must_use]
3081    pub fn reads(&self) -> Reads {
3082        Reads {
3083            opening: self.opening,
3084            pages: self.pages.load(Atomic::Relaxed),
3085            indexes: self.indexes.load(Atomic::Relaxed),
3086            dictionaries: self.opened.load(Atomic::Relaxed),
3087        }
3088    }
3089
3090    /// Where the file's bytes went, from the directory alone.
3091    ///
3092    /// No page is read, so this costs the same on a 45 GB table as on an empty one. See [`Layout`]
3093    /// for what is charged where and for why the three things that are not columns stay separate.
3094    #[must_use]
3095    pub fn layout(&self) -> Layout {
3096        let table = &self.table;
3097        let stripes = table.stripes.as_slice();
3098        let columns = table
3099            .fields
3100            .iter()
3101            .enumerate()
3102            .map(|(at, field)| ColumnLayout {
3103                name: field.name.clone(),
3104                kind: field.ty.to_string(),
3105                pages: sum(stripes.iter().map(|stripe| span_bytes(&stripe.pages, at))),
3106                memberships: sum(stripes.iter().map(|stripe| page_bytes(&stripe.memberships, at))),
3107                sieves: sum(stripes.iter().map(|stripe| page_bytes(&stripe.sieves, at))),
3108                part_ranges: sum(stripes.iter().map(|stripe| page_bytes(&stripe.part_ranges, at))),
3109                dictionary: page_bytes(&table.dictionaries, at),
3110            })
3111            .collect();
3112        Layout {
3113            file: self.size,
3114            rows: table.rows,
3115            stripes: stripes.len(),
3116            parts: self.places.len(),
3117            columns,
3118            indexes: sum(stripes.iter().map(|stripe| u64::from(stripe.index.length))),
3119            directory: self.directory,
3120            header: HEADER,
3121        }
3122    }
3123
3124    /// What every part of one column is stored as, which is what `pragma_storage_info` reports.
3125    ///
3126    /// Unlike [`Self::layout`] this reads the data, because the encoder's choice is in the page and
3127    /// nowhere else. The directory says how many bytes a column took and says nothing about what
3128    /// shape they are in, and the shape is the question worth asking: the same rows in a different
3129    /// order come back bit packed on one file and plain on another, and that is the difference a
3130    /// clustered load makes to a scan.
3131    ///
3132    /// One read per stripe rather than one per part. A part is a few kilobytes out of a page that
3133    /// is a quarter of a megabyte, so asking part by part would read the same page sixty four
3134    /// times. Nothing is put in the page cache, because a caller asking what a file looks like is
3135    /// not about to scan it and evicting the pages a real query wants would be a poor trade.
3136    ///
3137    /// # Errors
3138    ///
3139    /// If the column is outside the schema, or a page, index section or checksum is invalid.
3140    pub fn stored(&self, column: usize) -> Result<Vec<StoredPart>> {
3141        let field = self
3142            .table
3143            .fields
3144            .get(column)
3145            .ok_or_else(|| invalid("stored column index out of range"))?;
3146        let mut stored = Vec::with_capacity(self.places.len());
3147        let mut row = 0;
3148        for (at, stripe) in self.table.stripes.iter().enumerate() {
3149            let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
3150            let index = read_index(&self.file, stripe, column)?;
3151            let mut bytes = vec![0; page.length as usize];
3152            read_at(&self.file, page.offset, &mut bytes)?;
3153            let ranges = self.stripe_part_ranges(at, column);
3154            for (part, &rows) in stripe.parts.iter().enumerate() {
3155                let span = *index.get(part).ok_or_else(|| invalid("part index out of range"))?;
3156                let held = part_bytes(&bytes, span)?;
3157                let range = ranges.and_then(|held| held.get(part));
3158                stored.push(StoredPart {
3159                    stripe: at,
3160                    part,
3161                    row,
3162                    rows: rows as usize,
3163                    encoding: page_encoding(&field.ty, rows as usize, held),
3164                    bytes: span.length as u64,
3165                    page: page.offset,
3166                    offset: span.start as u64,
3167                    low: range
3168                        .and_then(|range| range.low.clone())
3169                        .and_then(|bound| bound.into_value(&field.ty)),
3170                    high: range
3171                        .and_then(|range| range.high.clone())
3172                        .and_then(|bound| bound.into_value(&field.ty)),
3173                    nulls: range.map(|range| range.nulls),
3174                });
3175                row += rows as usize;
3176            }
3177        }
3178        Ok(stored)
3179    }
3180
3181    /// How many parts the table has, which is how many chunks a scan of it reads.
3182    #[must_use]
3183    pub fn parts(&self) -> usize {
3184        self.places.len()
3185    }
3186
3187    /// The parts of each stripe, in table wide part numbers.
3188    ///
3189    /// A scan that wants one worker to own the page it reads hands work out in these runs. The
3190    /// stripes are contiguous in part numbering and all but the last hold sixty four parts, but a
3191    /// stripe can be flushed early when rows arrive out of order, so the runs are read off the
3192    /// directory rather than worked out from a constant.
3193    #[must_use]
3194    pub fn stripe_parts(&self) -> Vec<std::ops::Range<usize>> {
3195        let mut runs = Vec::with_capacity(self.table.stripes.len());
3196        let mut start = 0;
3197        for stripe in &self.table.stripes {
3198            let end = start + stripe.parts.len();
3199            runs.push(start..end);
3200            start = end;
3201        }
3202        runs
3203    }
3204
3205    /// How many rows one stripe holds, in the numbering [`Self::stripe_parts`] hands back.
3206    ///
3207    /// Off the directory, which is already in memory, rather than by the caller asking for each
3208    /// part in turn through the catalog. Nothing past the end holds any rows.
3209    #[must_use]
3210    pub fn stripe_rows(&self, stripe: usize) -> usize {
3211        self.table.stripes.get(stripe).map_or(0, |held| held.rows)
3212    }
3213
3214    /// Asks the page cache to keep `stripes` stripes of every column instead of the default.
3215    ///
3216    /// This only ever raises the number. A scan that gives each worker a whole stripe has one page
3217    /// per column per worker open at once, and a cache smaller than that is worse than no cache at
3218    /// all: every worker's page is evicted by the others before it has finished its stripe, so it
3219    /// reads a quarter of a megabyte for every part it takes out of it.
3220    pub fn keep_stripes(&self, stripes: usize) {
3221        self.kept.fetch_max(stripes, Atomic::Relaxed);
3222    }
3223
3224    /// Rows in one part, or zero when the part number is past the table.
3225    #[must_use]
3226    pub fn part_rows(&self, at: usize) -> usize {
3227        self.places.get(at).map_or(0, |place| place.rows as usize)
3228    }
3229
3230    /// The committed table directory.
3231    #[must_use]
3232    pub fn table(&self) -> &Table {
3233        &self.table
3234    }
3235
3236    /// Exact leading frequencies when the stored synopsis proves a count-descending prefix.
3237    ///
3238    /// The returned list can be longer than `top`. Keeping the stored tail lets a later TopN apply
3239    /// additional ordering keys without losing a value tied with the requested boundary.
3240    ///
3241    /// # Errors
3242    ///
3243    /// If the column is outside the schema or a stored value does not fit its declared type.
3244    pub fn top_frequencies(&self, column: usize, top: usize) -> Result<Option<Vec<(Value, u64)>>> {
3245        let field = self
3246            .table
3247            .fields
3248            .get(column)
3249            .ok_or_else(|| invalid("frequency column index out of range"))?;
3250        let Some(summary) = self.table.frequencies.get(column).and_then(Option::as_ref) else {
3251            return Ok(None);
3252        };
3253        if top == 0 || summary.entries.len() < top {
3254            return Ok(None);
3255        }
3256        let boundary = summary.entries[top - 1].count;
3257        if boundary <= summary.omitted_max {
3258            return Ok(None);
3259        }
3260        self.decode_frequencies(column, &field.ty, &summary.entries).map(Some)
3261    }
3262
3263    /// Every value of one column with the number of rows holding it, when the synopsis is complete.
3264    ///
3265    /// The heavy hitter pass keeps a bounded set of candidates and decrements them all when it runs
3266    /// out of room, so what it usually ends with is the leading values and a bound on everything it
3267    /// dropped. `omitted_max` of zero says that never happened: no candidate was ever decremented and
3268    /// the entries did not overflow the stored budget, so the list is every distinct value of the
3269    /// column with an exact count, and a null counts as a value of its own rather than being skipped.
3270    ///
3271    /// That makes a whole class of question answerable without reading a row. How many rows hold a
3272    /// value, how many do not, and what a `GROUP BY` of that column with a count over it produces are
3273    /// all in here. It is only ever true of a column with few enough distinct values, which is the
3274    /// case worth having, because that is exactly the column a grouping or an equality filter would
3275    /// otherwise walk every row to answer.
3276    ///
3277    /// `None` when the column has no synopsis, or has one that dropped anything.
3278    ///
3279    /// # Errors
3280    ///
3281    /// If the column is outside the schema or a stored value does not fit its declared type.
3282    pub fn exact_frequencies(&self, column: usize) -> Result<Option<Vec<(Value, u64)>>> {
3283        let Some(prefix) = self.frequency_prefix(column)? else {
3284            return Ok(None);
3285        };
3286        Ok((prefix.omitted_max == 0).then_some(prefix.entries))
3287    }
3288
3289    /// Every value the synopsis lists with the number of rows holding it, and a bound on the rest.
3290    ///
3291    /// The counts are exact whether or not the list is complete. The heavy hitter pass keeps a
3292    /// bounded candidate set and then recounts only the candidates that survived it, so a value that
3293    /// made it into the list carries the number of rows that really hold it rather than whatever the
3294    /// pass had left over. What the pass loses is values, not counts.
3295    ///
3296    /// `omitted_max` is how many rows the most common value left out can hold, and zero says nothing
3297    /// was left out at all, which is what [`exact_frequencies`] asks for. Above zero the list is the
3298    /// leading values of the column and everything else is somewhere between no rows and that bound.
3299    ///
3300    /// That prefix is worth reading on its own. A column with a value in half its rows and a long
3301    /// tail behind it has no complete synopsis and never will, and it is the column where dividing
3302    /// the rows by the distinct count is furthest from the truth.
3303    ///
3304    /// `None` when the column has no synopsis.
3305    ///
3306    /// # Errors
3307    ///
3308    /// If the column is outside the schema or a stored value does not fit its declared type.
3309    ///
3310    /// [`exact_frequencies`]: Self::exact_frequencies
3311    pub fn frequency_prefix(&self, column: usize) -> Result<Option<FrequencyPrefix>> {
3312        let field = self
3313            .table
3314            .fields
3315            .get(column)
3316            .ok_or_else(|| invalid("frequency column index out of range"))?;
3317        let Some(summary) = self.table.frequencies.get(column).and_then(Option::as_ref) else {
3318            return Ok(None);
3319        };
3320        let entries = self.decode_frequencies(column, &field.ty, &summary.entries)?;
3321        Ok(Some(FrequencyPrefix { entries, omitted_max: summary.omitted_max }))
3322    }
3323
3324    /// Turns stored frequency entries into values of the column's own type.
3325    fn decode_frequencies(
3326        &self,
3327        column: usize,
3328        ty: &LogicalType,
3329        entries: &[FrequencyEntry],
3330    ) -> Result<Vec<(Value, u64)>> {
3331        let dictionary = if *ty == LogicalType::Varchar { self.dictionary(column)? } else { None };
3332        let mut out = Vec::with_capacity(entries.len());
3333        for entry in entries {
3334            let value = match entry.value {
3335                FrequencyValue::Null => Value::Null,
3336                FrequencyValue::Integer(value) => match *ty {
3337                    LogicalType::TinyInt => Value::TinyInt(
3338                        i8::try_from(value)
3339                            .map_err(|_| invalid("frequency TINYINT is out of range"))?,
3340                    ),
3341                    LogicalType::UTinyInt => Value::UTinyInt(
3342                        u8::try_from(value)
3343                            .map_err(|_| invalid("frequency UTINYINT is out of range"))?,
3344                    ),
3345                    LogicalType::USmallInt => Value::USmallInt(
3346                        u16::try_from(value)
3347                            .map_err(|_| invalid("frequency USMALLINT is out of range"))?,
3348                    ),
3349                    LogicalType::UInteger => Value::UInteger(
3350                        u32::try_from(value)
3351                            .map_err(|_| invalid("frequency UINTEGER is out of range"))?,
3352                    ),
3353                    LogicalType::UBigInt => Value::UBigInt(
3354                        u64::try_from(value)
3355                            .map_err(|_| invalid("frequency UBIGINT is out of range"))?,
3356                    ),
3357                    LogicalType::SmallInt => Value::SmallInt(
3358                        i16::try_from(value)
3359                            .map_err(|_| invalid("frequency SMALLINT is out of range"))?,
3360                    ),
3361                    LogicalType::Integer => Value::Integer(
3362                        i32::try_from(value)
3363                            .map_err(|_| invalid("frequency INTEGER is out of range"))?,
3364                    ),
3365                    LogicalType::BigInt => Value::BigInt(
3366                        i64::try_from(value)
3367                            .map_err(|_| invalid("frequency BIGINT is out of range"))?,
3368                    ),
3369                    LogicalType::Date => Value::Date(
3370                        i32::try_from(value)
3371                            .map_err(|_| invalid("frequency DATE is out of range"))?,
3372                    ),
3373                    LogicalType::Timestamp => Value::Timestamp(
3374                        i64::try_from(value)
3375                            .map_err(|_| invalid("frequency TIMESTAMP is out of range"))?,
3376                    ),
3377                    _ => return Err(invalid("integer frequency belongs to another type")),
3378                },
3379                FrequencyValue::Code(code) => dictionary
3380                    .as_ref()
3381                    .ok_or_else(|| invalid("frequency code has no dictionary"))?
3382                    .try_value_at(code as usize)?,
3383            };
3384            out.push((value, entry.count));
3385        }
3386        Ok(out)
3387    }
3388
3389    /// Sparse rows belonging to the bounded numeric frequency candidate set.
3390    ///
3391    /// The list is omitted when collecting it would exceed the fixed storage budget. A composite
3392    /// aggregate may accept a result over these rows only when its requested boundary is strictly
3393    /// greater than `omitted_max`.
3394    ///
3395    /// # Errors
3396    ///
3397    /// If the column is outside the schema.
3398    pub fn frequency_occurrences(&self, column: usize) -> Result<Option<FrequencyOccurrences>> {
3399        self.table
3400            .fields
3401            .get(column)
3402            .ok_or_else(|| invalid("frequency column index out of range"))?;
3403        let Some(summary) = self.table.frequencies.get(column).and_then(Option::as_ref) else {
3404            return Ok(None);
3405        };
3406        if summary.ordinals.is_empty() {
3407            return Ok(None);
3408        }
3409        Ok(Some(FrequencyOccurrences {
3410            omitted_max: summary.omitted_max,
3411            ordinals: summary.ordinals.clone(),
3412        }))
3413    }
3414
3415    /// How many distinct values one column holds, counting a null as no value.
3416    ///
3417    /// A string column of this format is written against one dictionary that covers the whole table.
3418    /// A code is handed out the first time a value is seen and nothing ever removes one, so the
3419    /// number of codes is the number of distinct values exactly rather than an estimate. That makes
3420    /// `COUNT(DISTINCT column)` over a whole table a question the directory already knows the answer
3421    /// to, and the alternative is a hash table with a row per distinct value built from a pass over
3422    /// every row.
3423    ///
3424    /// A null in the column used to make this `None` and no longer does. A null row is written as
3425    /// the code for the empty string, so a nullable column's dictionary can hold an empty string
3426    /// that no row of it actually has, and the dictionary on its own does not say which case it is.
3427    /// The writer does know, because it counts the non-null rows that use each code on its way to
3428    /// the frequency summary, so it records how many codes any row holds and the directory carries
3429    /// that number. This reads it rather than the size of the dictionary, which also means the
3430    /// dictionary page is not opened to answer.
3431    ///
3432    /// `None` for a column the file has no dictionary for, which is every column that is not a
3433    /// string. A sketch would answer that approximately and SQL asked for the exact number.
3434    ///
3435    /// # Errors
3436    ///
3437    /// If the column is outside the schema.
3438    pub fn distinct_values(&self, column: usize) -> Result<Option<u64>> {
3439        self.table
3440            .distincts
3441            .get(column)
3442            .copied()
3443            .ok_or_else(|| invalid("distinct column index out of range"))
3444    }
3445
3446    /// How many rows of one column are null, added up over the stripes.
3447    ///
3448    /// Every stripe records this exactly when it is written, because a null count is not a bound
3449    /// that is allowed to be wide the way a minimum and a maximum are: a filter that reads one too
3450    /// many is slow and a `COUNT` that reads one too many is wrong. Adding up a few hundred numbers
3451    /// already in memory is what makes `COUNT(column)` over a whole table free.
3452    ///
3453    /// # Errors
3454    ///
3455    /// If the column is outside the schema.
3456    pub fn null_count(&self, column: usize) -> Result<u64> {
3457        if column >= self.table.fields.len() {
3458            return Err(invalid("null count column index out of range"));
3459        }
3460        let mut nulls = 0_u64;
3461        for stripe in &self.table.stripes {
3462            let range = stripe
3463                .zone
3464                .column(column)
3465                .ok_or_else(|| invalid("stripe zone is narrower than the schema"))?;
3466            nulls = nulls
3467                .checked_add(range.nulls as u64)
3468                .ok_or_else(|| invalid("null count overflow"))?;
3469        }
3470        Ok(nulls)
3471    }
3472
3473    /// The smallest and the largest value of one string column, from the order beside its values.
3474    ///
3475    /// The dictionary holds exactly the values the column holds, so the first and the last of them
3476    /// in sorted order are the column's minimum and maximum. Two reads of a rank block settle what
3477    /// otherwise walks a million rows.
3478    ///
3479    /// `None` when the column is not a string, when the file was written before version 9 and so has
3480    /// no order, when the column has no values at all, or when it has a null in it, which is the
3481    /// placeholder again: the empty string a null is written as would sort ahead of every real
3482    /// value and be reported as the minimum.
3483    ///
3484    /// # Errors
3485    ///
3486    /// If the column is outside the schema, or a rank names a code the dictionary does not have.
3487    pub fn text_extremes(&self, column: usize) -> Result<Option<(Value, Value)>> {
3488        if self.null_count(column)? > 0 {
3489            return Ok(None);
3490        }
3491        let Some(dictionary) = self.dictionary(column)? else { return Ok(None) };
3492        let Some(ranks) = dictionary.ranks() else { return Ok(None) };
3493        if ranks == 0 {
3494            return Ok(None);
3495        }
3496        let low = text_at_rank(&dictionary, 0)?;
3497        let high = text_at_rank(&dictionary, ranks - 1)?;
3498        Ok(Some((low, high)))
3499    }
3500
3501    /// The smallest and the largest value of one column, when every stripe wrote exact ends.
3502    ///
3503    /// A stripe's ends are allowed to be wider than the truth, because a bound that rules out a
3504    /// chunk that could not match is still correct when it rules out nothing. That is what makes
3505    /// them cheap to write for a bit packed or a dictionary column, and it is also what stops them
3506    /// answering a `MIN`. So each stripe says which of the two it wrote, and this answers only when
3507    /// all of them walked their rows.
3508    ///
3509    /// `None` for a column with no ends, for an empty table, and for a column any stripe of which
3510    /// guessed. Nulls need no special case, because the ends skip them the same way `MIN` does.
3511    ///
3512    /// One case is given up on that did not have to be. A stripe merges the ends of its sixty four
3513    /// parts, and a part with no ends at all erases the merged ones, because a part whose rows are
3514    /// not covered by the stripe's ends is a stripe that would skip rows it should keep. A part of
3515    /// nothing but nulls has no rows to cover and so did not need to erase anything, but the merge
3516    /// cannot tell that part from a part whose layout it could not read. So a column with a chunk
3517    /// of nothing but nulls in the middle of it goes and reads the rows. That is slow and right,
3518    /// and the fix is a row count per part rather than anything here.
3519    ///
3520    /// # Errors
3521    ///
3522    /// If the column is outside the schema.
3523    pub fn exact_extremes(&self, column: usize) -> Result<Option<(Bound, Bound)>> {
3524        if column >= self.table.fields.len() {
3525            return Err(invalid("extremes column index out of range"));
3526        }
3527        let mut low: Option<Bound> = None;
3528        let mut high: Option<Bound> = None;
3529        for stripe in &self.table.stripes {
3530            let range = stripe
3531                .zone
3532                .column(column)
3533                .ok_or_else(|| invalid("stripe zone is narrower than the schema"))?;
3534            if !range.exact {
3535                return Ok(None);
3536            }
3537            // A stripe of nothing but nulls has no ends and says nothing about the column's, which
3538            // is why this skips it rather than giving up on the whole column. A stripe that has
3539            // rows and still has no end is a layout whose values this cannot see, and skipping that
3540            // one would answer with an end taken from the other stripes, so it gives up instead.
3541            let (Some(small), Some(large)) = (range.low.as_ref(), range.high.as_ref()) else {
3542                if stripe.rows > range.nulls {
3543                    return Ok(None);
3544                }
3545                continue;
3546            };
3547            low = Some(low.map_or_else(|| small.clone(), |held| held.smaller(small.clone())));
3548            high = Some(high.map_or_else(|| large.clone(), |held| held.larger(large.clone())));
3549        }
3550        Ok(low.zip(high))
3551    }
3552
3553    /// The sum of one integer column and how many rows went into it, when every stripe wrote one.
3554    ///
3555    /// The count beside the sum is the non-null rows, because that is what a `SUM` adds up and what
3556    /// an `AVG` divides by, and a caller that had to work it out from the row count and the null
3557    /// count would be doing the same walk twice.
3558    ///
3559    /// `None` for anything that is not an integer column, for a file written by something that did
3560    /// not record it, and when adding the stripes together would overflow.
3561    ///
3562    /// # Errors
3563    ///
3564    /// If the column is outside the schema.
3565    pub fn exact_sum(&self, column: usize) -> Result<Option<(i128, u64)>> {
3566        if column >= self.table.fields.len() {
3567            return Err(invalid("sum column index out of range"));
3568        }
3569        let mut total = 0_i128;
3570        let mut rows = 0_u64;
3571        for stripe in &self.table.stripes {
3572            let range = stripe
3573                .zone
3574                .column(column)
3575                .ok_or_else(|| invalid("stripe zone is narrower than the schema"))?;
3576            let Some(part) = range.sum else { return Ok(None) };
3577            let Some(sum) = total.checked_add(part) else { return Ok(None) };
3578            total = sum;
3579            rows = rows.saturating_add(stripe.rows as u64 - range.nulls as u64);
3580        }
3581        Ok(Some((total, rows)))
3582    }
3583
3584    /// The global dictionary of a column, opened once however many workers ask for it at once.
3585    ///
3586    /// The unlocked look is first because it is the answer every time after the first and it costs a
3587    /// load. Everybody who misses it queues on [`Self::loading`] and looks again on the way in, so
3588    /// the one who arrived first does the reading and the rest take what it left. Waiting is the
3589    /// cheaper thing to do: the work behind the lock is a page read, a checksum and the decode of a
3590    /// dictionary that can hold half a million entries, and the alternative is every worker of the
3591    /// scan doing all of it and all but one dropping the result on the floor.
3592    fn dictionary(&self, column: usize) -> Result<Option<Arc<Vector>>> {
3593        let Some(page) = self.table.dictionaries[column] else { return Ok(None) };
3594        if let Some(dictionary) = self.dictionaries[column].get() {
3595            return Ok(Some(Arc::clone(dictionary)));
3596        }
3597        let _queued = self.loading[column].lock().map_err(|_| invalid("a poisoned dictionary"))?;
3598        if let Some(dictionary) = self.dictionaries[column].get() {
3599            return Ok(Some(Arc::clone(dictionary)));
3600        }
3601        self.opened.fetch_add(1, Atomic::Relaxed);
3602        let dictionary = Arc::new(open_global_dictionary(
3603            Arc::clone(&self.file),
3604            page,
3605            &self.table.fields[column].ty,
3606            TEXT_KEEP_BUDGET,
3607        )?);
3608        let _ = self.dictionaries[column].set(Arc::clone(&dictionary));
3609        Ok(Some(dictionary))
3610    }
3611
3612    /// Reads one section's extent table and checks it against the entry that names it.
3613    ///
3614    /// # Errors
3615    ///
3616    /// If the entry points outside the file, the table does not checksum, or it does not decode as
3617    /// a run of extents in element order.
3618    pub fn extents(&self, of: &Section) -> Result<Vec<section::Extent>> {
3619        if of.extent_bytes == 0 {
3620            return Ok(Vec::new());
3621        }
3622        let mut bytes = vec![0; of.extent_bytes as usize];
3623        read_at(&self.file, of.extent_page, &mut bytes)?;
3624        if checksum(&bytes) != of.hash {
3625            return Err(invalid("a section's extent table does not checksum"));
3626        }
3627        let extents = section::decode_extents(&bytes)?;
3628        if extents.len() != of.extents as usize {
3629            return Err(invalid("a section's extent table is not the length the entry says"));
3630        }
3631        Ok(extents)
3632    }
3633
3634    /// Reads and verifies one extent of a section.
3635    ///
3636    /// This is what section 3.2's second rule is for. A reduction that needs one extent of a two
3637    /// gigabyte forward link reads and checksums that extent and nothing else, which is the whole
3638    /// difference between a structure that works at SF100 and issue #745.
3639    ///
3640    /// # Errors
3641    ///
3642    /// If the extent points outside the file, or its bytes do not checksum.
3643    pub fn extent(&self, of: &section::Extent) -> Result<Vec<u8>> {
3644        let end = of
3645            .offset
3646            .checked_add(u64::from(of.length))
3647            .ok_or_else(|| invalid("an extent overflows the file"))?;
3648        if of.offset < HEADER || end > self.size {
3649            return Err(invalid("an extent is outside the file"));
3650        }
3651        let mut bytes = vec![0; of.length as usize];
3652        read_at(&self.file, of.offset, &mut bytes)?;
3653        if checksum(&bytes) != of.hash {
3654            return Err(invalid("an extent does not checksum"));
3655        }
3656        Ok(bytes)
3657    }
3658
3659    /// Reads a whole section's payload, every extent of it, in order.
3660    ///
3661    /// For a structure that is resident anyway, which a key map is. Anything large enough that the
3662    /// split matters should be walking [`Reader::extents`] and taking the one it needs.
3663    ///
3664    /// # Errors
3665    ///
3666    /// If the extent table or any extent fails its check.
3667    pub fn payload(&self, of: &Section) -> Result<Vec<u8>> {
3668        let extents = self.extents(of)?;
3669        let mut bytes =
3670            Vec::with_capacity(sum(extents.iter().map(|one| u64::from(one.length))) as usize);
3671        for one in &extents {
3672            if one.first != bytes.len() as u64 {
3673                return Err(invalid("a section's extents do not join up"));
3674            }
3675            bytes.extend_from_slice(&self.extent(one)?);
3676        }
3677        if of.header_bytes as usize > bytes.len() {
3678            return Err(invalid("a section's header is longer than its payload"));
3679        }
3680        Ok(bytes)
3681    }
3682
3683    /// Reads only the named columns from one part.
3684    ///
3685    /// The whole stripe page each column lives in is read and kept, because a scan asks for the
3686    /// parts of a stripe one after another and this is what turns sixty four reads into one.
3687    ///
3688    /// # Errors
3689    ///
3690    /// If a part, column, page, or checksum is invalid.
3691    pub fn read(&self, part: usize, columns: &[usize]) -> Result<Chunk> {
3692        self.read_impl(part, columns, true)
3693    }
3694
3695    /// Reads named columns from one part without keeping the stripe page it came out of.
3696    ///
3697    /// This is for sparse row fetches after a selective TopN or filter, which reach a few parts of
3698    /// a stripe rather than all of them. A caller that will read most of a stripe should use
3699    /// [`Self::read`] instead, because this reads and discards the page index every time.
3700    ///
3701    /// # Errors
3702    ///
3703    /// If a part, column, page, or checksum is invalid.
3704    pub fn read_sparse(&self, part: usize, columns: &[usize]) -> Result<Chunk> {
3705        self.read_impl(part, columns, false)
3706    }
3707
3708    /// Whether an exact global-code membership index proves that the stripe holding a part cannot
3709    /// contain any of the sorted candidate codes.
3710    ///
3711    /// # Errors
3712    ///
3713    /// If the part, column, index page, checksum, or delta stream is invalid.
3714    pub fn skips_codes(&self, part: usize, column: usize, candidates: &[u32]) -> Result<bool> {
3715        if candidates.is_empty() {
3716            return Ok(true);
3717        }
3718        if candidates.windows(2).any(|pair| pair[0] >= pair[1]) {
3719            return Err(Error::internal("native code candidates are not sorted and unique"));
3720        }
3721        let stripe = self.stripe_of(part)?;
3722        let Some(page) = stripe.memberships.get(column).copied().flatten() else {
3723            return Ok(false);
3724        };
3725        let mut bytes = vec![0; page.length as usize];
3726        read_at(&self.file, page.offset, &mut bytes)?;
3727        if checksum(&bytes) != page.hash {
3728            return Err(invalid("membership page checksum differs"));
3729        }
3730        let codes = decode_membership(&bytes)?;
3731        let mut left = 0;
3732        let mut right = 0;
3733        while left < codes.len() && right < candidates.len() {
3734            match codes[left].cmp(&candidates[right]) {
3735                Ordering::Less => left += 1,
3736                Ordering::Greater => right += 1,
3737                Ordering::Equal => return Ok(false),
3738            }
3739        }
3740        Ok(true)
3741    }
3742
3743    fn stripe_of(&self, part: usize) -> Result<&Stripe> {
3744        let place = self.places.get(part).ok_or_else(|| invalid("part index out of range"))?;
3745        self.table
3746            .stripes
3747            .get(place.stripe as usize)
3748            .ok_or_else(|| invalid("stripe index out of range"))
3749    }
3750
3751    /// The page index of one column of one stripe, and its page when the caller wants all of it.
3752    ///
3753    /// A scan hands parts out in order, so every worker on a column crosses into a new stripe within
3754    /// a few parts of the others and they all want the same page at the same moment. This used to
3755    /// let all of them read it, which cost the scan as many copies of every page as it had workers.
3756    /// On the full ClickBench file a `MIN(EventDate), MAX(EventDate)` moved 3.2 GB off the disk to
3757    /// look at 400 MB of column.
3758    ///
3759    /// A worker that finds the page it wants already being read neither waits for it nor reads it
3760    /// again. It comes back with the index alone, which sends [`Reader::read_impl`] down the path
3761    /// that reads the one part it came for, a few kilobytes against a quarter of a megabyte, and it
3762    /// picks the page up from the cache on its next part. Waiting would be the other way to avoid
3763    /// the duplicate read and it is worse: the pages that matter are the wide string ones, they take
3764    /// milliseconds to copy even warm, and every other worker would be stopped for all of it.
3765    ///
3766    /// The file is never read under the lock.
3767    fn held(&self, at: usize, stripe: &Stripe, column: usize, whole: bool) -> Result<CachedColumn> {
3768        let cache = self.cache.get(column).ok_or_else(|| invalid("column index out of range"))?;
3769        let mut cached = cache.lock().map_err(|_| invalid("column page cache is poisoned"))?;
3770        let known = cached.index.get(at).and_then(Clone::clone);
3771        let page = cached.pages.get(at).and_then(Clone::clone);
3772        if let Some(index) = known.clone() {
3773            if !whole || page.is_some() {
3774                return Ok(CachedColumn { stripe: at, index, page });
3775            }
3776        }
3777        if cached.loading.contains(&at) {
3778            drop(cached);
3779            // The index is almost always already here, because somebody read this stripe to get
3780            // into the loading list in the first place, so this branch usually costs no read at
3781            // all and the one part read in `read_impl` is all the losing worker pays for.
3782            if let Some(index) = known {
3783                return Ok(CachedColumn { stripe: at, index, page: None });
3784            }
3785            let held = self.page_of(stripe, column, at, false, None)?;
3786            let mut cached = cache.lock().map_err(|_| invalid("column page cache is poisoned"))?;
3787            remember(&mut cached, &held, self.kept.load(Atomic::Relaxed));
3788            return Ok(held);
3789        }
3790        cached.loading.push(at);
3791        drop(cached);
3792
3793        let read = self.page_of(stripe, column, at, whole, known);
3794
3795        // The stripe leaves the loading list and its page enters the cache under one lock. Doing
3796        // them separately would leave a moment where another worker sees neither and reads the
3797        // page a second time, which is the whole thing this is here to stop.
3798        let mut cached = cache.lock().map_err(|_| invalid("column page cache is poisoned"))?;
3799        if let Some(position) = cached.loading.iter().position(|loading| *loading == at) {
3800            cached.loading.remove(position);
3801        }
3802        let held = read?;
3803        remember(&mut cached, &held, self.kept.load(Atomic::Relaxed));
3804        Ok(held)
3805    }
3806
3807    /// Reads one stripe's index for a column, and its page when the caller wants all of it.
3808    ///
3809    /// `known` is the index when the reader has already read it, which after the first worker
3810    /// through a stripe it always has, because [`remember`] keeps every index for the life of the
3811    /// reader. Without that a scan reads the index again on every part that misses the page cache.
3812    fn page_of(
3813        &self,
3814        stripe: &Stripe,
3815        column: usize,
3816        at: usize,
3817        whole: bool,
3818        known: Option<Arc<Vec<PartSpan>>>,
3819    ) -> Result<CachedColumn> {
3820        let index = match known {
3821            Some(index) => index,
3822            None => {
3823                self.indexes.fetch_add(1, Atomic::Relaxed);
3824                Arc::new(read_index(&self.file, stripe, column)?)
3825            }
3826        };
3827        let page = if whole {
3828            self.pages.fetch_add(1, Atomic::Relaxed);
3829            let span = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
3830            let mut bytes = vec![0; span.length as usize];
3831            read_at(&self.file, span.offset, &mut bytes)?;
3832            Some(Arc::new(bytes))
3833        } else {
3834            None
3835        };
3836        Ok(CachedColumn { stripe: at, index, page })
3837    }
3838
3839    fn read_impl(&self, at: usize, columns: &[usize], whole: bool) -> Result<Chunk> {
3840        let place = *self.places.get(at).ok_or_else(|| invalid("part index out of range"))?;
3841        let index = place.stripe as usize;
3842        let stripe =
3843            self.table.stripes.get(index).ok_or_else(|| invalid("stripe index out of range"))?;
3844        let rows = place.rows as usize;
3845        let mut picked = Vec::with_capacity(columns.len());
3846        for &column in columns {
3847            let field = self
3848                .table
3849                .fields
3850                .get(column)
3851                .ok_or_else(|| invalid("column index out of range"))?;
3852            let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
3853            let held = self.held(index, stripe, column, whole)?;
3854            let span = *held
3855                .index
3856                .get(place.part as usize)
3857                .ok_or_else(|| invalid("part index out of range"))?;
3858            let owned;
3859            let bytes = match &held.page {
3860                Some(held) => part_bytes(held, span)?,
3861                None => {
3862                    let offset = page
3863                        .offset
3864                        .checked_add(span.start as u64)
3865                        .ok_or_else(|| invalid("part range overflow"))?;
3866                    let mut bytes = vec![0; span.length];
3867                    read_at(&self.file, offset, &mut bytes)?;
3868                    owned = bytes;
3869                    &owned
3870                }
3871            };
3872            if checksum(bytes) != span.hash {
3873                return Err(invalid(&format!(
3874                    "column page checksum differs, column {column} part {} at {}+{} of {} bytes, \
3875                     wanted {:016x} and got {:016x}",
3876                    place.part,
3877                    page.offset,
3878                    span.start,
3879                    span.length,
3880                    span.hash,
3881                    checksum(bytes),
3882                )));
3883            }
3884            let dictionary = self.dictionary(column)?;
3885            // Held as a page, because a column that came out of a file is handed out more than
3886            // once. A group by clones its key columns out of the chunk so the keys outlive it, a
3887            // projection of a bare column name does the same, and a cut of a flat run copies unless
3888            // the run is a page. One `Arc` per column per part buys all of those, and it moves the
3889            // run into the `Arc` without touching a value.
3890            picked.push(decode(&field.ty, rows, bytes, dictionary)?.into_pages());
3891        }
3892        Chunk::with_rows(picked, rows)
3893    }
3894
3895    /// Whether persisted statistics prove that a part cannot match the predicates.
3896    ///
3897    /// Three of them, asked cheapest first.
3898    ///
3899    /// The stripe's bounds are in memory already, so they are free, and they are also the coarsest:
3900    /// every part of a stripe gets the same answer and a scan that skips one part that way skips all
3901    /// sixty four. Then the part's own bounds, which are a read of one page per column per stripe
3902    /// and are sixty four times finer. Then the sieves, which are per part and answer equality, the
3903    /// test bounds are worst at: a column of identifiers has every stripe and nearly every part
3904    /// covering the whole of its type, so bounds keep them all and the sieve keeps the ones that
3905    /// really hold the value.
3906    ///
3907    /// The middle one is what an ordered comparison on a column the rows are not sorted by needs. On
3908    /// ClickBench 24 the stripe bounds leave eight stripes of sixteen alive, which is half the file,
3909    /// and the part bounds leave thirty parts of nine hundred and seventy four.
3910    #[must_use]
3911    pub fn skips(&self, part: usize, probes: &[Probe]) -> bool {
3912        let Some(place) = self.places.get(part).copied() else { return false };
3913        let Some(stripe) = self.table.stripes.get(place.stripe as usize) else { return false };
3914        if stripe.zone.skips(probes) {
3915            return true;
3916        }
3917        probes.iter().any(|probe| self.outside(place, probe) || self.sifted(place, probe))
3918    }
3919
3920    /// Whether the bounds of one part rule out one probe.
3921    ///
3922    /// The part's own two ends, which are narrower than the stripe's and cost a page read the first
3923    /// time this is asked about a column. A column with no page here answers `false`, which is the
3924    /// answer a caller got before there were any.
3925    fn outside(&self, place: Place, probe: &Probe) -> bool {
3926        match self.stripe_part_ranges(place.stripe as usize, probe.column) {
3927            Some(ranges) => ranges
3928                .get(place.part as usize)
3929                .is_some_and(|range| range.excludes(probe.op, &probe.value)),
3930            None => false,
3931        }
3932    }
3933
3934    /// The per part ranges of one stripe of one column, read once and kept.
3935    ///
3936    /// `None` when the column has no page in that stripe and when the page is damaged, on the same
3937    /// reasoning as the sieves: this is an index over data that is still there, so a caller that
3938    /// cannot read one reads the rows and gets the right answer slowly.
3939    fn stripe_part_ranges(&self, stripe: usize, column: usize) -> Option<&[Range]> {
3940        let slot = self.part_ranges.get(column)?.get(stripe)?;
3941        if let Some(held) = slot.get() {
3942            return Some(held);
3943        }
3944        let page = self.table.stripes.get(stripe)?.part_ranges.get(column).copied().flatten()?;
3945        let mut bytes = vec![0; page.length as usize];
3946        read_at(&self.file, page.offset, &mut bytes).ok()?;
3947        if checksum(&bytes) != page.hash {
3948            return None;
3949        }
3950        let ranges = Arc::new(decode_part_ranges(&bytes).ok()?);
3951        let _ = slot.set(ranges);
3952        slot.get().map(|held| held.as_slice())
3953    }
3954
3955    /// Whether persisted statistics prove that every row of a part matches the predicates.
3956    ///
3957    /// Only the bounds. The sieves say nothing here, because a sieve that holds a value is a sieve
3958    /// that may be holding somebody else's hash, so it can rule a part out and can never wave one
3959    /// through.
3960    ///
3961    /// The stripe first and the part after it, the same two steps and in the same order as
3962    /// [`Self::skips`]. The stripe's bounds are in memory already and its null count covers sixty
3963    /// four parts rather than one, so a stripe that answers is an answer for nothing, and the part's
3964    /// own bounds are only read for the probes it could not settle. Both directions are safe: a
3965    /// stretch where everything passes contains no narrower stretch where something fails, and a
3966    /// stripe with no nulls has no nulls in any of its parts.
3967    ///
3968    /// A string end a part recorded is cut down to its first few bytes, so a part's stretch can be
3969    /// wider than its rows really are as well. That is the same safe direction for the same reason,
3970    /// and it is why this asks the two ends rather than anything `exact` says.
3971    #[must_use]
3972    pub fn certain(&self, part: usize, probes: &[Probe]) -> bool {
3973        let Some(place) = self.places.get(part).copied() else { return false };
3974        let Some(stripe) = self.table.stripes.get(place.stripe as usize) else { return false };
3975        if stripe.zone.certain(probes) {
3976            return true;
3977        }
3978        probes
3979            .iter()
3980            .all(|probe| stripe.zone.certain(slice::from_ref(probe)) || self.inside(place, probe))
3981    }
3982
3983    /// Whether one part's own two ends prove that every row of it passes `probe`.
3984    ///
3985    /// The mirror of [`Self::outside`], reading the same page. `false` for a part whose stripe wrote
3986    /// no range page, which is a stripe of one part, because there the stripe's own bounds are the
3987    /// part's and the caller has already asked them.
3988    fn inside(&self, place: Place, probe: &Probe) -> bool {
3989        match self.stripe_part_ranges(place.stripe as usize, probe.column) {
3990            Some(ranges) => ranges
3991                .get(place.part as usize)
3992                .is_some_and(|range| range.certain(probe.op, &probe.value)),
3993            None => false,
3994        }
3995    }
3996
3997    /// Whether the bounds of one stripe prove that none of its parts can match the predicates.
3998    ///
3999    /// The cheap half of [`Self::skips`], asked about a whole stripe at once. The bounds live in the
4000    /// directory and are already in memory, so this answers without touching the file, and that is
4001    /// the reason it is worth having on its own: a caller that wants to know roughly where the work
4002    /// is before it starts any workers can ask this about sixteen stripes for nothing, where asking
4003    /// [`Self::skips`] about nine hundred parts would read and decode a sieve page per stripe first.
4004    ///
4005    /// It keeps stripes that [`Self::skips`] would rule out part by part, which is the right way for
4006    /// it to be wrong: the parts are still checked when they are read.
4007    #[must_use]
4008    pub fn stripe_skips(&self, stripe: usize, probes: &[Probe]) -> bool {
4009        self.table.stripes.get(stripe).is_some_and(|held| held.zone.skips(probes))
4010    }
4011
4012    /// Whether the sieve of one part rules out one probe.
4013    ///
4014    /// Only equality. An ordered comparison is what the bounds are for and a sieve says nothing
4015    /// about it, and a read that cannot answer keeps the part, which is the answer a caller with no
4016    /// sieve gets anyway.
4017    fn sifted(&self, place: Place, probe: &Probe) -> bool {
4018        if probe.op != Op::Equal {
4019            return false;
4020        }
4021        match self.stripe_sieves(place.stripe as usize, probe.column) {
4022            Some(sieves) => sieves
4023                .get(place.part as usize)
4024                .and_then(Option::as_ref)
4025                .is_some_and(|sieve| sieve.excludes(&probe.value)),
4026            None => false,
4027        }
4028    }
4029
4030    /// The sieves of one stripe of one column, read once and kept.
4031    ///
4032    /// `None` when the column has no sieves in that stripe, when the page is damaged, and when the
4033    /// bytes are not a page this version can read. A sieve is an index over data that is still there
4034    /// and a caller that cannot read one reads the rows, so this is the one place in the file where
4035    /// a bad checksum is a slow query rather than an error.
4036    fn stripe_sieves(&self, stripe: usize, column: usize) -> Option<&[Option<Sieve>]> {
4037        let slot = self.sieves.get(column)?.get(stripe)?;
4038        if let Some(held) = slot.get() {
4039            return Some(held);
4040        }
4041        let page = self.table.stripes.get(stripe)?.sieves.get(column).copied().flatten()?;
4042        let mut bytes = vec![0; page.length as usize];
4043        read_at(&self.file, page.offset, &mut bytes).ok()?;
4044        if checksum(&bytes) != page.hash {
4045            return None;
4046        }
4047        let sieves = Arc::new(decode_sieves(&bytes).ok()?);
4048        let _ = slot.set(sieves);
4049        slot.get().map(|held| held.as_slice())
4050    }
4051}
4052
4053/// The value sitting at one position of a dictionary's sorted order.
4054fn text_at_rank(dictionary: &Vector, rank: usize) -> Result<Value> {
4055    let code = dictionary.code_at_rank(rank)? as usize;
4056    let text = dictionary
4057        .try_text_at(code)?
4058        .ok_or_else(|| invalid("global dictionary order names a code it does not have"))?;
4059    Ok(Value::Varchar(text.into()))
4060}
4061
4062/// Writes one span of a file at an offset, without depending on where the cursor is.
4063///
4064/// The writer owns an offset of its own and passes it in here, so that nothing it writes depends on
4065/// a cursor that a read is entitled to move. Both of these can come back short and both loop.
4066#[cfg(unix)]
4067fn write_at(file: &File, mut offset: u64, mut bytes: &[u8]) -> Result<()> {
4068    use std::os::unix::fs::FileExt;
4069    while !bytes.is_empty() {
4070        let written = file.write_at(bytes, offset).map_err(io)?;
4071        if written == 0 {
4072            return Err(invalid("a write to the native file wrote nothing"));
4073        }
4074        offset += written as u64;
4075        bytes = &bytes[written..];
4076    }
4077    Ok(())
4078}
4079
4080/// The same write, on the call Windows spells differently.
4081#[cfg(windows)]
4082fn write_at(file: &File, mut offset: u64, mut bytes: &[u8]) -> Result<()> {
4083    use std::os::windows::fs::FileExt;
4084    while !bytes.is_empty() {
4085        let written = file.seek_write(bytes, offset).map_err(io)?;
4086        if written == 0 {
4087            return Err(invalid("a write to the native file wrote nothing"));
4088        }
4089        offset += written as u64;
4090        bytes = &bytes[written..];
4091    }
4092    Ok(())
4093}
4094
4095/// Somewhere that is neither, where the cursor is all there is.
4096#[cfg(not(any(unix, windows)))]
4097fn write_at(file: &File, offset: u64, bytes: &[u8]) -> Result<()> {
4098    use std::io::Write;
4099    let mut file = file.try_clone().map_err(io)?;
4100    file.seek(SeekFrom::Start(offset)).map_err(io)?;
4101    file.write_all(bytes).map_err(io)
4102}
4103
4104/// Reads one span of a file at an offset, without moving a cursor anybody else can see.
4105///
4106/// Every reader of a table shares one [`File`] behind an [`Arc`], and a grouped aggregate reads its
4107/// pages from several threads at once, so this has to be positional. Seeking and then reading is
4108/// two calls with a gap in the middle, and in that gap another thread's seek lands and the read
4109/// comes back with somebody else's bytes.
4110///
4111/// Both of these can come back short, so both loop. A read of zero bytes before the span is filled
4112/// means the file stops earlier than the directory said it does.
4113#[cfg(unix)]
4114fn read_at(file: &File, mut offset: u64, mut bytes: &mut [u8]) -> Result<()> {
4115    use std::os::unix::fs::FileExt;
4116    while !bytes.is_empty() {
4117        let read = file.read_at(bytes, offset).map_err(io)?;
4118        if read == 0 {
4119            return Err(invalid("column page ends before its declared length"));
4120        }
4121        offset += read as u64;
4122        bytes = &mut bytes[read..];
4123    }
4124    Ok(())
4125}
4126
4127/// The same read, on the call Windows spells differently.
4128///
4129/// `seek_read` is one `ReadFile` carrying the offset with it, so two of them cannot interleave the
4130/// way a seek and a read can. It does leave the shared cursor somewhere afterwards, which is why
4131/// nothing in this file may read that cursor.
4132#[cfg(windows)]
4133fn read_at(file: &File, mut offset: u64, mut bytes: &mut [u8]) -> Result<()> {
4134    use std::os::windows::fs::FileExt;
4135    while !bytes.is_empty() {
4136        let read = file.seek_read(bytes, offset).map_err(io)?;
4137        if read == 0 {
4138            return Err(invalid("column page ends before its declared length"));
4139        }
4140        offset += read as u64;
4141        bytes = &mut bytes[read..];
4142    }
4143    Ok(())
4144}
4145
4146/// Somewhere that is neither, where the cursor is all there is.
4147///
4148/// This one does race, and there is no way to write it so it does not. Nothing we build for runs
4149/// here, so it exists to keep the crate compiling rather than to be correct under threads.
4150#[cfg(not(any(unix, windows)))]
4151fn read_at(file: &File, offset: u64, bytes: &mut [u8]) -> Result<()> {
4152    let mut file = file.try_clone().map_err(io)?;
4153    file.seek(SeekFrom::Start(offset)).map_err(io)?;
4154    file.read_exact(bytes).map_err(io)
4155}
4156
4157/// What a column type is called in the directory.
4158///
4159/// A tag is a number in a file somebody else wrote, so a tag that has been used is used forever and
4160/// the only thing that may happen to this list is that it grows. 1 to 13 are the tags the format
4161/// had when it could store thirteen types, and 14 to 27 are the rest, in the order they were added
4162/// rather than in an order that means anything.
4163fn type_tag(ty: &LogicalType) -> Result<u8> {
4164    match ty {
4165        LogicalType::SmallInt => Ok(1),
4166        LogicalType::Integer => Ok(2),
4167        LogicalType::BigInt => Ok(3),
4168        LogicalType::Varchar => Ok(4),
4169        LogicalType::Date => Ok(5),
4170        LogicalType::Timestamp => Ok(6),
4171        LogicalType::Boolean => Ok(7),
4172        LogicalType::TinyInt => Ok(8),
4173        LogicalType::UTinyInt => Ok(9),
4174        LogicalType::USmallInt => Ok(10),
4175        LogicalType::UInteger => Ok(11),
4176        LogicalType::UBigInt => Ok(12),
4177        LogicalType::Decimal { .. } => Ok(13),
4178        LogicalType::Float => Ok(14),
4179        LogicalType::Double => Ok(15),
4180        LogicalType::HugeInt => Ok(16),
4181        LogicalType::UHugeInt => Ok(17),
4182        LogicalType::Time => Ok(18),
4183        LogicalType::TimeTz => Ok(19),
4184        LogicalType::TimestampTz => Ok(20),
4185        LogicalType::Interval => Ok(21),
4186        LogicalType::Uuid => Ok(22),
4187        LogicalType::Blob => Ok(23),
4188        LogicalType::Bit => Ok(24),
4189        LogicalType::TimestampS => Ok(25),
4190        LogicalType::TimestampMs => Ok(26),
4191        LogicalType::TimestampNs => Ok(27),
4192        _ => Err(Error::not_implemented(format!("native storage for {ty}"))),
4193    }
4194}
4195
4196/// The tag of a column type, and the parameters of the ones that have any.
4197///
4198/// Only `DECIMAL` has parameters today. Width and scale go after the tag rather than into it
4199/// because they are what says how wide a value is on disk, and a reader that guessed would read the
4200/// wrong number of bytes per row rather than the wrong number of digits.
4201fn put_type(out: &mut Vec<u8>, ty: &LogicalType) -> Result<()> {
4202    out.push(type_tag(ty)?);
4203    if let LogicalType::Decimal { width, scale } = ty {
4204        out.push(*width);
4205        out.push(*scale);
4206    }
4207    Ok(())
4208}
4209
4210/// The other half of [`put_type`], reading the parameters the tag says are there.
4211fn read_type(cur: &mut Cursor<'_>) -> Result<LogicalType> {
4212    let tag = cur.u8()?;
4213    if tag == 13 {
4214        let width = cur.u8()?;
4215        let scale = cur.u8()?;
4216        return LogicalType::decimal(width, scale)
4217            .map_err(|_| invalid("decimal column width and scale are not a decimal"));
4218    }
4219    tag_type(tag)
4220}
4221
4222fn tag_type(tag: u8) -> Result<LogicalType> {
4223    match tag {
4224        1 => Ok(LogicalType::SmallInt),
4225        2 => Ok(LogicalType::Integer),
4226        3 => Ok(LogicalType::BigInt),
4227        4 => Ok(LogicalType::Varchar),
4228        5 => Ok(LogicalType::Date),
4229        6 => Ok(LogicalType::Timestamp),
4230        7 => Ok(LogicalType::Boolean),
4231        8 => Ok(LogicalType::TinyInt),
4232        9 => Ok(LogicalType::UTinyInt),
4233        10 => Ok(LogicalType::USmallInt),
4234        11 => Ok(LogicalType::UInteger),
4235        12 => Ok(LogicalType::UBigInt),
4236        14 => Ok(LogicalType::Float),
4237        15 => Ok(LogicalType::Double),
4238        16 => Ok(LogicalType::HugeInt),
4239        17 => Ok(LogicalType::UHugeInt),
4240        18 => Ok(LogicalType::Time),
4241        19 => Ok(LogicalType::TimeTz),
4242        20 => Ok(LogicalType::TimestampTz),
4243        21 => Ok(LogicalType::Interval),
4244        22 => Ok(LogicalType::Uuid),
4245        23 => Ok(LogicalType::Blob),
4246        24 => Ok(LogicalType::Bit),
4247        25 => Ok(LogicalType::TimestampS),
4248        26 => Ok(LogicalType::TimestampMs),
4249        27 => Ok(LogicalType::TimestampNs),
4250        _ => Err(invalid("column type tag is unknown")),
4251    }
4252}
4253
4254fn put_u16(out: &mut Vec<u8>, value: u16) {
4255    out.extend_from_slice(&value.to_le_bytes());
4256}
4257fn put_u32(out: &mut Vec<u8>, value: u32) {
4258    out.extend_from_slice(&value.to_le_bytes());
4259}
4260fn put_u64(out: &mut Vec<u8>, value: u64) {
4261    out.extend_from_slice(&value.to_le_bytes());
4262}
4263fn put_var_u64(out: &mut Vec<u8>, mut value: u64) {
4264    while value >= 0x80 {
4265        out.push((value as u8 & 0x7f) | 0x80);
4266        value >>= 7;
4267    }
4268    out.push(value as u8);
4269}
4270
4271fn frequency_order(left: FrequencyValue, right: FrequencyValue) -> Ordering {
4272    match (left, right) {
4273        (FrequencyValue::Null, FrequencyValue::Null) => Ordering::Equal,
4274        (FrequencyValue::Null, _) => Ordering::Less,
4275        (_, FrequencyValue::Null) => Ordering::Greater,
4276        (FrequencyValue::Integer(left), FrequencyValue::Integer(right)) => left.cmp(&right),
4277        (FrequencyValue::Code(left), FrequencyValue::Code(right)) => left.cmp(&right),
4278        (FrequencyValue::Integer(_), FrequencyValue::Code(_)) => Ordering::Less,
4279        (FrequencyValue::Code(_), FrequencyValue::Integer(_)) => Ordering::Greater,
4280    }
4281}
4282
4283fn code_frequency(dictionary: &GlobalDictionary) -> FrequencySummary {
4284    let mut entries = dictionary
4285        .counts
4286        .iter()
4287        .enumerate()
4288        .filter(|(_, count)| **count != 0)
4289        .map(|(code, &count)| FrequencyEntry { value: FrequencyValue::Code(code as u32), count })
4290        .collect::<Vec<_>>();
4291    if dictionary.nulls != 0 {
4292        entries.push(FrequencyEntry { value: FrequencyValue::Null, count: dictionary.nulls });
4293    }
4294    entries.sort_unstable_by(|left, right| {
4295        right.count.cmp(&left.count).then_with(|| frequency_order(left.value, right.value))
4296    });
4297    let omitted_max = entries.get(FREQUENCY_ENTRIES).map_or(0, |entry| entry.count);
4298    entries.truncate(FREQUENCY_ENTRIES);
4299    FrequencySummary { entries, omitted_max, ordinals: Vec::new() }
4300}
4301
4302fn encode_directory(table: &Table) -> Result<Vec<u8>> {
4303    let mut out = DIRECTORY.to_vec();
4304    let name = table.name.as_bytes();
4305    put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("table name too long"))?);
4306    out.extend_from_slice(name);
4307    put_u16(&mut out, u16::try_from(table.fields.len()).map_err(|_| invalid("too many columns"))?);
4308    for field in &table.fields {
4309        let name = field.name.as_bytes();
4310        put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("column name too long"))?);
4311        out.extend_from_slice(name);
4312        put_type(&mut out, &field.ty)?;
4313        out.push(u8::from(field.not_null));
4314    }
4315    for dictionary in &table.dictionaries {
4316        match dictionary {
4317            None => out.push(0),
4318            Some(page) => {
4319                out.push(1);
4320                put_u64(&mut out, page.offset);
4321                put_u32(&mut out, page.length);
4322                put_u64(&mut out, page.hash);
4323            }
4324        }
4325    }
4326    for distinct in &table.distincts {
4327        match distinct {
4328            None => out.push(0),
4329            Some(count) => {
4330                out.push(1);
4331                put_u64(&mut out, *count);
4332            }
4333        }
4334    }
4335    put_u64(&mut out, u64::try_from(table.rows).map_err(|_| invalid("row count overflow"))?);
4336    put_u32(&mut out, u32::try_from(table.stripes.len()).map_err(|_| invalid("too many stripes"))?);
4337    for stripe in &table.stripes {
4338        put_u32(
4339            &mut out,
4340            u32::try_from(stripe.parts.len()).map_err(|_| invalid("too many parts in a stripe"))?,
4341        );
4342        for &rows in &stripe.parts {
4343            put_u32(&mut out, rows);
4344        }
4345        put_u64(&mut out, stripe.index.offset);
4346        put_u32(&mut out, stripe.index.length);
4347        for page in &stripe.pages {
4348            put_u64(&mut out, page.offset);
4349            put_u32(&mut out, page.length);
4350        }
4351        // A membership index says which of a dictionary's codes a part holds, so a column the writer
4352        // decided against giving a dictionary has nothing for it to be about and writes none. Every
4353        // file written before that decision existed has a dictionary on every varchar column, so
4354        // this reads those files byte for byte the way it always did.
4355        for ((field, dictionary), membership) in
4356            table.fields.iter().zip(&table.dictionaries).zip(&stripe.memberships)
4357        {
4358            if field.ty != LogicalType::Varchar || dictionary.is_none() {
4359                continue;
4360            }
4361            let page =
4362                membership.ok_or_else(|| invalid("string page has no code membership index"))?;
4363            put_u64(&mut out, page.offset);
4364            put_u32(&mut out, page.length);
4365            put_u64(&mut out, page.hash);
4366        }
4367        for sieve in &stripe.sieves {
4368            match sieve {
4369                None => out.push(0),
4370                Some(page) => {
4371                    out.push(1);
4372                    put_u64(&mut out, page.offset);
4373                    put_u32(&mut out, page.length);
4374                    put_u64(&mut out, page.hash);
4375                }
4376            }
4377        }
4378        for held in &stripe.part_ranges {
4379            match held {
4380                None => out.push(0),
4381                Some(page) => {
4382                    out.push(1);
4383                    put_u64(&mut out, page.offset);
4384                    put_u32(&mut out, page.length);
4385                    put_u64(&mut out, page.hash);
4386                }
4387            }
4388        }
4389        for range in stripe.zone.columns() {
4390            put_bound(&mut out, range.low.as_ref())?;
4391            put_bound(&mut out, range.high.as_ref())?;
4392            put_u32(
4393                &mut out,
4394                u32::try_from(range.nulls).map_err(|_| invalid("null count overflow"))?,
4395            );
4396            out.push(u8::from(range.exact));
4397            match range.sum {
4398                None => out.push(0),
4399                Some(total) => {
4400                    out.push(1);
4401                    out.extend_from_slice(&total.to_le_bytes());
4402                }
4403            }
4404        }
4405    }
4406    out.extend_from_slice(FREQUENCIES);
4407    put_u16(
4408        &mut out,
4409        u16::try_from(table.frequencies.len())
4410            .map_err(|_| invalid("too many frequency columns"))?,
4411    );
4412    for summary in &table.frequencies {
4413        let Some(summary) = summary else {
4414            out.push(0);
4415            continue;
4416        };
4417        out.push(1);
4418        put_u64(&mut out, summary.omitted_max);
4419        put_u32(
4420            &mut out,
4421            u32::try_from(summary.entries.len())
4422                .map_err(|_| invalid("too many frequency entries"))?,
4423        );
4424        for entry in &summary.entries {
4425            match entry.value {
4426                FrequencyValue::Null => out.push(0),
4427                FrequencyValue::Integer(value) => {
4428                    out.push(1);
4429                    out.extend_from_slice(&value.to_le_bytes());
4430                }
4431                FrequencyValue::Code(value) => {
4432                    out.push(2);
4433                    put_u32(&mut out, value);
4434                }
4435            }
4436            put_u64(&mut out, entry.count);
4437        }
4438        put_u32(
4439            &mut out,
4440            u32::try_from(summary.ordinals.len())
4441                .map_err(|_| invalid("too many frequency ordinals"))?,
4442        );
4443        let mut previous = 0_u64;
4444        for (at, &ordinal) in summary.ordinals.iter().enumerate() {
4445            let delta = if at == 0 {
4446                ordinal
4447            } else {
4448                ordinal
4449                    .checked_sub(previous)
4450                    .ok_or_else(|| invalid("frequency ordinals are not ordered"))?
4451            };
4452            if at != 0 && delta == 0 {
4453                return Err(invalid("frequency ordinals are not unique"));
4454            }
4455            put_var_u64(&mut out, delta);
4456            previous = ordinal;
4457        }
4458    }
4459    // Written only when there is a declaration, so that the common file is the same bytes it was
4460    // and the section is not a byte of zero on every table in the world that never asked for one.
4461    if let Some(clustering) = &table.clustering {
4462        out.extend_from_slice(CLUSTERING);
4463        out.push(clustering.width().tag());
4464        put_u16(
4465            &mut out,
4466            u16::try_from(clustering.columns().len())
4467                .map_err(|_| invalid("too many clustering columns"))?,
4468        );
4469        for &column in clustering.columns() {
4470            put_u16(
4471                &mut out,
4472                u16::try_from(column).map_err(|_| invalid("clustering column index overflow"))?,
4473            );
4474        }
4475    }
4476    // The section table, last, behind its own magic, for the same reason the frequency block is
4477    // behind its own: a reader that stops before it gets a table with no sections, and a table with
4478    // no sections is a correct table. The one difference from the blocks before it is that this one
4479    // is written even when it is empty, so that a file written by this build always says which
4480    // sections it has rather than leaving a reader to infer it from where the bytes ran out.
4481    out.extend_from_slice(SECTIONS);
4482    put_u64(&mut out, table.generation);
4483    put_u16(
4484        &mut out,
4485        u16::try_from(table.sections.len()).map_err(|_| invalid("too many sections"))?,
4486    );
4487    for held in &table.sections {
4488        held.encode(&mut out)?;
4489    }
4490    Ok(out)
4491}
4492
4493/// The small level of the directory, naming every table in the file.
4494///
4495/// This is what a footer slot points at. Each entry carries its own checksum over its table
4496/// directory, so a table whose directory is torn is found when that table is first touched rather
4497/// than being trusted because the catalog around it checksummed.
4498fn encode_catalog(entries: &[Entry]) -> Result<Vec<u8>> {
4499    let mut out = CATALOG.to_vec();
4500    put_u32(&mut out, u32::try_from(entries.len()).map_err(|_| invalid("too many tables"))?);
4501    for entry in entries {
4502        let name = entry.name.as_bytes();
4503        put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("table name too long"))?);
4504        out.extend_from_slice(name);
4505        put_u64(&mut out, u64::try_from(entry.rows).map_err(|_| invalid("row count overflow"))?);
4506        put_u16(
4507            &mut out,
4508            u16::try_from(entry.fields.len()).map_err(|_| invalid("too many columns"))?,
4509        );
4510        for field in &entry.fields {
4511            let name = field.name.as_bytes();
4512            put_u16(
4513                &mut out,
4514                u16::try_from(name.len()).map_err(|_| invalid("column name too long"))?,
4515            );
4516            out.extend_from_slice(name);
4517            put_type(&mut out, &field.ty)?;
4518            out.push(u8::from(field.not_null));
4519        }
4520        put_u64(&mut out, entry.directory.offset);
4521        put_u32(&mut out, entry.directory.length);
4522        put_u64(&mut out, entry.directory.hash);
4523    }
4524    Ok(out)
4525}
4526
4527/// Reads the catalog directory back, checking every span against the file before anything is
4528/// allocated for it.
4529fn decode_catalog(bytes: &[u8], size: u64) -> Result<Vec<Entry>> {
4530    let mut cur = Cursor { bytes, at: 0 };
4531    if cur.take(8)? != CATALOG {
4532        return Err(invalid("catalog magic differs"));
4533    }
4534    let count = cur.u32()? as usize;
4535    let mut entries: Vec<Entry> = Vec::with_capacity(count.min(1024));
4536    for _ in 0..count {
4537        let name = cur.text()?;
4538        let rows = usize::try_from(cur.u64()?).map_err(|_| invalid("row count does not fit"))?;
4539        let width = cur.u16()? as usize;
4540        let mut fields = Vec::with_capacity(width);
4541        for _ in 0..width {
4542            let name = cur.text()?;
4543            let ty = read_type(&mut cur)?;
4544            let not_null = match cur.u8()? {
4545                0 => false,
4546                1 => true,
4547                _ => return Err(invalid("nullability flag differs")),
4548            };
4549            fields.push(Field { name, ty, not_null });
4550        }
4551        let directory = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
4552        let end = directory
4553            .offset
4554            .checked_add(u64::from(directory.length))
4555            .ok_or_else(|| invalid("table directory offset overflow"))?;
4556        if directory.offset < HEADER
4557            || end > size
4558            || directory.length as usize > MAX_DIRECTORY
4559            || directory.length == 0
4560        {
4561            return Err(invalid("table directory range is outside the file"));
4562        }
4563        if entries.iter().any(|held| held.name == name) {
4564            return Err(invalid("two tables in the catalog have the same name"));
4565        }
4566        entries.push(Entry { name, fields, rows, directory });
4567    }
4568    Ok(entries)
4569}
4570
4571struct Cursor<'a> {
4572    bytes: &'a [u8],
4573    at: usize,
4574}
4575impl<'a> Cursor<'a> {
4576    fn take(&mut self, len: usize) -> Result<&'a [u8]> {
4577        let end = self.at.checked_add(len).ok_or_else(|| invalid("directory offset overflow"))?;
4578        let bytes =
4579            self.bytes.get(self.at..end).ok_or_else(|| invalid("directory is truncated"))?;
4580        self.at = end;
4581        Ok(bytes)
4582    }
4583    fn u8(&mut self) -> Result<u8> {
4584        Ok(self.take(1)?[0])
4585    }
4586    fn u16(&mut self) -> Result<u16> {
4587        Ok(u16::from_le_bytes(self.take(2)?.try_into().expect("two bytes")))
4588    }
4589    fn u32(&mut self) -> Result<u32> {
4590        Ok(u32::from_le_bytes(self.take(4)?.try_into().expect("four bytes")))
4591    }
4592    fn u64(&mut self) -> Result<u64> {
4593        Ok(u64::from_le_bytes(self.take(8)?.try_into().expect("eight bytes")))
4594    }
4595    fn var_u64(&mut self) -> Result<u64> {
4596        let mut value = 0_u64;
4597        for shift in (0..=63).step_by(7) {
4598            let byte = self.u8()?;
4599            let part = u64::from(byte & 0x7f);
4600            if shift == 63 && part > 1 {
4601                return Err(invalid("frequency ordinal varint overflows"));
4602            }
4603            value |= part << shift;
4604            if byte & 0x80 == 0 {
4605                return Ok(value);
4606            }
4607        }
4608        Err(invalid("frequency ordinal varint is too long"))
4609    }
4610    /// A zone map's end, in the layout `rudb_common::bounds` defines.
4611    ///
4612    /// The bytes are the ones this directory has written since format 10 and the codec moved to
4613    /// rank zero rather than being copied, because a column summary now writes the same two ends
4614    /// and two encodings of one type is how the two quietly stop agreeing.
4615    fn bound(&mut self) -> Result<Option<Bound>> {
4616        bounds::get(self.bytes, &mut self.at)
4617    }
4618    fn text(&mut self) -> Result<String> {
4619        let len = self.u16()? as usize;
4620        String::from_utf8(self.take(len)?.to_vec()).map_err(|_| invalid("name is not UTF-8"))
4621    }
4622}
4623
4624fn decode_directory(bytes: &[u8], size: u64) -> Result<Table> {
4625    let mut cur = Cursor { bytes, at: 0 };
4626    if cur.take(8)? != DIRECTORY {
4627        return Err(invalid("directory magic differs"));
4628    }
4629    let name = cur.text()?;
4630    let width = cur.u16()? as usize;
4631    let mut fields = Vec::with_capacity(width);
4632    for _ in 0..width {
4633        let name = cur.text()?;
4634        let ty = read_type(&mut cur)?;
4635        let not_null = match cur.u8()? {
4636            0 => false,
4637            1 => true,
4638            _ => return Err(invalid("nullability flag differs")),
4639        };
4640        fields.push(Field { name, ty, not_null });
4641    }
4642    let mut dictionaries = Vec::with_capacity(width);
4643    for _ in 0..width {
4644        dictionaries.push(match cur.u8()? {
4645            0 => None,
4646            1 => {
4647                let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
4648                let end = page
4649                    .offset
4650                    .checked_add(u64::from(page.length))
4651                    .ok_or_else(|| invalid("dictionary page offset overflow"))?;
4652                // A global dictionary covers a whole column, not one bounded stripe. Its lazy
4653                // payload is intentionally allowed to grow past `MAX_PAGE`; only ordinary column
4654                // pages are capped there. `Writer::finish` has already bounded this length by the
4655                // on-disk `u32`, and the range check below keeps it inside the file.
4656                if page.offset < HEADER || end > size {
4657                    return Err(invalid("dictionary page range is outside the file"));
4658                }
4659                Some(page)
4660            }
4661            _ => return Err(invalid("dictionary page tag differs")),
4662        });
4663    }
4664    let mut distincts = Vec::with_capacity(width);
4665    for _ in 0..width {
4666        distincts.push(match cur.u8()? {
4667            0 => None,
4668            1 => Some(cur.u64()?),
4669            _ => return Err(invalid("distinct count tag differs")),
4670        });
4671    }
4672    let rows = usize::try_from(cur.u64()?).map_err(|_| invalid("row count does not fit"))?;
4673    let count = cur.u32()? as usize;
4674    let mut stripes = Vec::with_capacity(count);
4675    let mut total = 0_usize;
4676    for _ in 0..count {
4677        let count = cur.u32()? as usize;
4678        if count == 0 || count > STRIPE_PARTS {
4679            return Err(invalid("stripe part count is outside its bound"));
4680        }
4681        let mut parts = Vec::with_capacity(count);
4682        let mut stripe_rows = 0_usize;
4683        for _ in 0..count {
4684            let rows = cur.u32()?;
4685            if rows == 0 {
4686                return Err(invalid("empty part"));
4687            }
4688            parts.push(rows);
4689            stripe_rows = stripe_rows
4690                .checked_add(rows as usize)
4691                .ok_or_else(|| invalid("stripe row count overflow"))?;
4692        }
4693        total =
4694            total.checked_add(stripe_rows).ok_or_else(|| invalid("stripe row count overflow"))?;
4695        let index = Span { offset: cur.u64()?, length: cur.u32()? };
4696        let section = index_section(count)?;
4697        let wanted = section
4698            .checked_mul(width)
4699            .and_then(|bytes| u32::try_from(bytes).ok())
4700            .ok_or_else(|| invalid("index page length overflow"))?;
4701        let end = index
4702            .offset
4703            .checked_add(u64::from(index.length))
4704            .ok_or_else(|| invalid("index page offset overflow"))?;
4705        if index.offset < HEADER || end > size || index.length != wanted {
4706            return Err(invalid("index page range is outside the file"));
4707        }
4708        let mut pages = Vec::with_capacity(width);
4709        for _ in 0..width {
4710            let offset = cur.u64()?;
4711            let length = cur.u32()?;
4712            let end = offset
4713                .checked_add(u64::from(length))
4714                .ok_or_else(|| invalid("page offset overflow"))?;
4715            if offset < HEADER || end > size || length as usize > MAX_PAGE {
4716                return Err(invalid("page range is outside the file"));
4717            }
4718            pages.push(Span { offset, length });
4719        }
4720        let mut memberships = vec![None; width];
4721        for (column, field) in fields.iter().enumerate() {
4722            if field.ty != LogicalType::Varchar || dictionaries[column].is_none() {
4723                continue;
4724            }
4725            let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
4726            let end = page
4727                .offset
4728                .checked_add(u64::from(page.length))
4729                .ok_or_else(|| invalid("membership page offset overflow"))?;
4730            if page.offset < HEADER || end > size || page.length as usize > MAX_PAGE {
4731                return Err(invalid("membership page range is outside the file"));
4732            }
4733            memberships[column] = Some(page);
4734        }
4735        let mut sieves = vec![None; width];
4736        for sieve in sieves.iter_mut().take(width) {
4737            match cur.u8()? {
4738                0 => continue,
4739                1 => {}
4740                _ => return Err(invalid("a sieve page has an unknown tag")),
4741            }
4742            let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
4743            let end = page
4744                .offset
4745                .checked_add(u64::from(page.length))
4746                .ok_or_else(|| invalid("sieve page offset overflow"))?;
4747            if page.offset < HEADER || end > size || page.length as usize > MAX_PAGE {
4748                return Err(invalid("sieve page range is outside the file"));
4749            }
4750            *sieve = Some(page);
4751        }
4752        let mut part_ranges = vec![None; width];
4753        for held in part_ranges.iter_mut().take(width) {
4754            match cur.u8()? {
4755                0 => continue,
4756                1 => {}
4757                _ => return Err(invalid("a part range page has an unknown tag")),
4758            }
4759            let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
4760            let end = page
4761                .offset
4762                .checked_add(u64::from(page.length))
4763                .ok_or_else(|| invalid("part range page offset overflow"))?;
4764            if page.offset < HEADER || end > size || page.length as usize > MAX_PAGE {
4765                return Err(invalid("part range page range is outside the file"));
4766            }
4767            *held = Some(page);
4768        }
4769        let mut ranges = Vec::with_capacity(width);
4770        for column in 0..width {
4771            let low = cur.bound()?;
4772            let high = cur.bound()?;
4773            let nulls = cur.u32()? as usize;
4774            if nulls > stripe_rows {
4775                return Err(invalid("null count exceeds stripe rows"));
4776            }
4777            let exact = cur.u8()? != 0;
4778            let sum = match cur.u8()? {
4779                0 => None,
4780                1 => Some(i128::from_le_bytes(
4781                    cur.take(16)?.try_into().map_err(|_| invalid("a stripe sum is truncated"))?,
4782                )),
4783                _ => return Err(invalid("a stripe sum has an unknown tag")),
4784            };
4785            // Files written before the ends of a decimal or a timestamp column carried their power
4786            // of ten hold a bare integer here, and that integer is the one the column holds, which
4787            // is what the power is over. So the type puts it back on the way in and an old file
4788            // prunes as well as a new one. A file that already wrote the power keeps it, because
4789            // this leaves anything that is not an integer alone.
4790            let ty = &fields.get(column).ok_or_else(|| invalid("a stripe range has no column"))?.ty;
4791            let low = low.map(|bound| scaled_as(bound, ty));
4792            let high = high.map(|bound| scaled_as(bound, ty));
4793            ranges.push(Range { low, high, nulls, exact, sum });
4794        }
4795        stripes.push(Stripe {
4796            rows: stripe_rows,
4797            parts,
4798            index,
4799            pages,
4800            memberships,
4801            sieves,
4802            part_ranges,
4803            zone: Zone::from_ranges(ranges),
4804        });
4805    }
4806    if total != rows {
4807        return Err(invalid("table row count differs from stripes"));
4808    }
4809    let frequencies = if cur.at == bytes.len() {
4810        vec![None; width]
4811    } else {
4812        if cur.take(8)? != FREQUENCIES {
4813            return Err(invalid("directory extension magic differs"));
4814        }
4815        if cur.u16()? as usize != width {
4816            return Err(invalid("frequency column count differs"));
4817        }
4818        let mut frequencies = Vec::with_capacity(width);
4819        for field in &fields {
4820            let summary = match cur.u8()? {
4821                0 => None,
4822                1 => {
4823                    let omitted_max = cur.u64()?;
4824                    let count = cur.u32()? as usize;
4825                    if count > FREQUENCY_ENTRIES {
4826                        return Err(invalid("frequency entry count exceeds its bound"));
4827                    }
4828                    let mut entries = Vec::with_capacity(count);
4829                    // row at a time: directory decoding validates each persisted bounded frequency entry.
4830                    for _ in 0..count {
4831                        let value = match cur.u8()? {
4832                            0 => FrequencyValue::Null,
4833                            1 => FrequencyValue::Integer(i128::from_le_bytes(
4834                                cur.take(16)?.try_into().expect("sixteen bytes"),
4835                            )),
4836                            2 => FrequencyValue::Code(cur.u32()?),
4837                            _ => return Err(invalid("frequency value tag differs")),
4838                        };
4839                        let valid = matches!(
4840                            (&field.ty, value),
4841                            (_, FrequencyValue::Null)
4842                                | (LogicalType::Varchar, FrequencyValue::Code(_))
4843                                | (
4844                                    LogicalType::TinyInt
4845                                        | LogicalType::SmallInt
4846                                        | LogicalType::Integer
4847                                        | LogicalType::BigInt
4848                                        | LogicalType::UTinyInt
4849                                        | LogicalType::USmallInt
4850                                        | LogicalType::UInteger
4851                                        | LogicalType::UBigInt
4852                                        | LogicalType::Date
4853                                        | LogicalType::Timestamp,
4854                                    FrequencyValue::Integer(_),
4855                                )
4856                        );
4857                        if !valid {
4858                            return Err(invalid("frequency value does not match its column"));
4859                        }
4860                        let count = cur.u64()?;
4861                        if count == 0 || count > rows as u64 {
4862                            return Err(invalid("frequency count is outside the table"));
4863                        }
4864                        entries.push(FrequencyEntry { value, count });
4865                    }
4866                    if entries.windows(2).any(|pair| pair[0].count < pair[1].count) {
4867                        return Err(invalid("frequency entries are not descending"));
4868                    }
4869                    let ordinals = {
4870                        let ordinal_count = cur.u32()? as usize;
4871                        if ordinal_count > FREQUENCY_ORDINALS || ordinal_count > rows {
4872                            return Err(invalid("frequency ordinal count exceeds its bound"));
4873                        }
4874                        let mut ordinals = Vec::with_capacity(ordinal_count);
4875                        let mut previous = 0_u64;
4876                        for at in 0..ordinal_count {
4877                            let delta = cur.var_u64()?;
4878                            if at != 0 && delta == 0 {
4879                                return Err(invalid("frequency ordinals are not increasing"));
4880                            }
4881                            let ordinal = if at == 0 {
4882                                delta
4883                            } else {
4884                                previous
4885                                    .checked_add(delta)
4886                                    .ok_or_else(|| invalid("frequency ordinal overflows"))?
4887                            };
4888                            if ordinal >= rows as u64 {
4889                                return Err(invalid("frequency ordinal is outside the table"));
4890                            }
4891                            ordinals.push(ordinal);
4892                            previous = ordinal;
4893                        }
4894                        ordinals
4895                    };
4896                    Some(FrequencySummary { entries, omitted_max, ordinals })
4897                }
4898                _ => return Err(invalid("frequency summary tag differs")),
4899            };
4900            frequencies.push(summary);
4901        }
4902        frequencies
4903    };
4904    // There are two optional trailing blocks now rather than one, so the reader dispatches on the
4905    // magic it finds rather than on where the bytes ran out. That is what lets the two arrive
4906    // independently: a format 22 directory ends here and has neither, a directory written before
4907    // the section table has only the clustering declaration, and each one still opens without a
4908    // rewrite. It is the G1 exit criterion, which is that a reader that knows about sections opens
4909    // a file that predates them and answers every query, only without the graph path.
4910    //
4911    // A repeated block is refused rather than allowed to win, because two clustering declarations
4912    // in one directory is a torn directory and the only question is which of them is the lie.
4913    let mut clustering = None;
4914    let mut sections = Vec::new();
4915    let mut seen_sections = false;
4916    // Zero until a section table says otherwise, which is what a format 22 table gets and what
4917    // makes every section stamp fail to match on one, because real generations start at one.
4918    let mut generation = 0;
4919    while cur.at != bytes.len() {
4920        let mut tag = [0u8; 8];
4921        tag.copy_from_slice(cur.take(8)?);
4922        if &tag == CLUSTERING {
4923            if clustering.is_some() {
4924                return Err(invalid("directory names two clustering declarations"));
4925            }
4926            let bucket = Width::from_tag(cur.u8()?)
4927                .ok_or_else(|| invalid("clustering width tag differs"))?;
4928            let count = cur.u16()? as usize;
4929            let mut columns = Vec::with_capacity(count.min(fields.len()));
4930            for _ in 0..count {
4931                columns.push(u32::from(cur.u16()?));
4932            }
4933            // Through the constructor and not built by hand, so that a file claiming a column the
4934            // table does not have is caught at open rather than at the first scan that trusted it.
4935            clustering = Some(Clustering::new(columns, bucket, &fields).map_err(|_| {
4936                invalid("stored clustering declaration does not match the table it is on")
4937            })?);
4938        } else if &tag == SECTIONS {
4939            if seen_sections {
4940                return Err(invalid("directory names two section tables"));
4941            }
4942            seen_sections = true;
4943            generation = cur.u64()?;
4944            let count = cur.u16()? as usize;
4945            if count > MAX_SECTIONS {
4946                return Err(invalid("section count exceeds its bound"));
4947            }
4948            sections = Vec::with_capacity(count);
4949            // entry at a time: a malformed section entry is refused rather than turned into an
4950            // offset.
4951            for _ in 0..count {
4952                sections.push(Section::decode(cur.take(section::ENTRY_BYTES)?)?);
4953            }
4954            for held in &sections {
4955                let Some(end) = held.extent_page.checked_add(u64::from(held.extent_bytes)) else {
4956                    return Err(invalid("a section's extent table overflows the file"));
4957                };
4958                // The bound check is here and not in `section`, because only the caller knows how
4959                // big the file is. A section pointing past the end is a torn directory, and reading
4960                // the payload it names would be reading whatever else is at that offset.
4961                if held.extent_bytes != 0 && (held.extent_page < HEADER || end > size) {
4962                    return Err(invalid("a section's extent table is outside the file"));
4963                }
4964                if held.extents == 0 && held.extent_bytes != 0 {
4965                    return Err(invalid("a section with no extents names an extent table"));
4966                }
4967            }
4968        } else {
4969            return Err(invalid("directory extension magic differs"));
4970        }
4971    }
4972    if cur.at != bytes.len() {
4973        return Err(invalid("directory has trailing bytes"));
4974    }
4975    Ok(Table {
4976        name,
4977        fields,
4978        stripes,
4979        rows,
4980        dictionaries,
4981        distincts,
4982        frequencies,
4983        clustering,
4984        generation,
4985        sections,
4986    })
4987}
4988
4989/// A zone map's end, in the layout `rudb_common::bounds` defines. See [`Cursor::bound`].
4990fn put_bound(out: &mut Vec<u8>, bound: Option<&Bound>) -> Result<()> {
4991    bounds::put(out, bound)
4992}
4993
4994/// Which cascades are worth trying on a run of dictionary codes.
4995///
4996/// The exhaustive chooser encodes every candidate at every level of a cascade three deep and keeps
4997/// the smallest, which on a part of 1024 codes is around a hundred full encodes to decide something
4998/// three candidates were always going to win. It is the right default for a crate that does not
4999/// know what it is looking at. Here we do know. Codes are counted from zero in the order the values
5000/// were first seen, so a part of them is one value, or a narrow band, or a few long runs, and those
5001/// are constant, frame of reference and run length. Nothing else has ever come first on this data.
5002///
5003/// A dictionary of dictionary codes is the one candidate that can never pay, because the codes are
5004/// already the dictionary, and it is also the most expensive one to try. Below the top level the
5005/// streams are an RLE's run values and run lengths, which are integers in their own right with no
5006/// runs left in them, so only the two flat candidates go down there.
5007///
5008/// This is size given up for time on purpose, and the ablation is this chooser against
5009/// [`chooser::EXHAUSTIVE`] on the same file.
5010#[derive(Debug)]
5011struct Codes;
5012
5013impl chooser::Chooser for Codes {
5014    fn name(&self) -> &'static str {
5015        "codes"
5016    }
5017
5018    fn narrow_strings(
5019        &self,
5020        _values: &[&[u8]],
5021        offered: &[string::Kind],
5022        _depth: u8,
5023    ) -> Vec<string::Kind> {
5024        // Never reached, because nothing here encodes strings through the cascade. The trait asks
5025        // for it and the honest answer to a question we have no opinion on is the whole list.
5026        offered.to_vec()
5027    }
5028
5029    fn narrow_integers(
5030        &self,
5031        _values: &[i64],
5032        offered: &[integer::Kind],
5033        depth: u8,
5034    ) -> Vec<integer::Kind> {
5035        let keep: &[integer::Kind] = if depth == 0 {
5036            &[integer::Kind::Constant, integer::Kind::Packed, integer::Kind::Rle]
5037        } else {
5038            &[integer::Kind::Constant, integer::Kind::Packed]
5039        };
5040        let narrowed: Vec<integer::Kind> =
5041            offered.iter().copied().filter(|kind| keep.contains(kind)).collect();
5042        // The contract is a non empty subset, and a chunk that offers none of the three is a chunk
5043        // this has no opinion about rather than one that cannot be written.
5044        if narrowed.is_empty() { offered.to_vec() } else { narrowed }
5045    }
5046}
5047
5048/// Which cascades are worth trying on a part of plain integers.
5049///
5050/// Wider than [`Codes`] because the values are not codes and carry whatever shape the column has.
5051/// A timestamp column climbs, so delta is the one that matters and is the reason this exists at
5052/// all: three timestamp columns in ClickBench were coming out at exactly eight bytes a row with
5053/// nothing asked of them. The same three columns are why the stride is here, since a timestamp
5054/// loaded from a source that recorded whole seconds is microseconds with twenty zero bits under
5055/// every value. A column that is one value with a handful of exceptions is sparse. What is still
5056/// left out is the dictionary, for the same reason as in [`Codes`]: it is the most
5057/// expensive candidate to try and this file already puts the columns that want one through a
5058/// dictionary of their own before they ever reach here.
5059#[derive(Debug)]
5060struct Fixed;
5061
5062impl chooser::Chooser for Fixed {
5063    fn name(&self) -> &'static str {
5064        "fixed"
5065    }
5066
5067    fn narrow_strings(
5068        &self,
5069        _values: &[&[u8]],
5070        offered: &[string::Kind],
5071        _depth: u8,
5072    ) -> Vec<string::Kind> {
5073        offered.to_vec()
5074    }
5075
5076    fn narrow_integers(
5077        &self,
5078        _values: &[i64],
5079        offered: &[integer::Kind],
5080        depth: u8,
5081    ) -> Vec<integer::Kind> {
5082        let keep: &[integer::Kind] = if depth == 0 {
5083            &[
5084                integer::Kind::Constant,
5085                integer::Kind::Packed,
5086                integer::Kind::Delta,
5087                integer::Kind::Rle,
5088                integer::Kind::Sparse,
5089                integer::Kind::Strided,
5090            ]
5091        } else {
5092            &[integer::Kind::Constant, integer::Kind::Packed, integer::Kind::Delta]
5093        };
5094        let narrowed: Vec<integer::Kind> =
5095            offered.iter().copied().filter(|kind| keep.contains(kind)).collect();
5096        if narrowed.is_empty() { offered.to_vec() } else { narrowed }
5097    }
5098}
5099
5100/// Every value of an integer part as an `i64`, or `None` for a part this cannot widen without
5101/// losing one.
5102///
5103/// `UBIGINT` is the only integer type left out, because half its range does not fit and a page that
5104/// silently wrapped would be worse than a page that stays plain. Booleans and strings are not
5105/// integers and have their own ways of being small.
5106fn widened(data: &Data) -> Option<Vec<i64>> {
5107    match data {
5108        Data::Int8(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
5109        Data::UInt8(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
5110        Data::Int16(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
5111        Data::UInt16(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
5112        Data::Int32(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
5113        Data::UInt32(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
5114        Data::Int64(values) => Some(values.to_vec()),
5115        _ => None,
5116    }
5117}
5118
5119/// An integer type a cascaded page can be read back into, and how to tell whether a value fits.
5120///
5121/// This exists so that the check and the conversion can be two loops instead of one. `TryFrom` puts
5122/// them together, which is the right shape for one value and the wrong one for a page: a fallible
5123/// conversion a value at a time is a branch a value at a time, the branch decides whether the loop
5124/// keeps going, and a loop like that is one no compiler will widen.
5125trait Narrow: Copy {
5126    /// How wide this type is, and what to add to a value to put its range at the bottom of a `u64`.
5127    ///
5128    /// Half the width for a signed type, which is what moves its smallest value to zero, and nothing
5129    /// for an unsigned one, whose smallest value is already there.
5130    const BIASED: (u32, u64);
5131
5132    /// The value narrowed, which the caller has already shown fits.
5133    fn narrow(value: i64) -> Self;
5134}
5135
5136/// The bits of `value` a `T` cannot hold, and zero when the value fits.
5137///
5138/// The question is asked this way round because the answers or together. A page fits when every
5139/// residue in it is zero, so the loop is an or into an accumulator and the decision is one test
5140/// after it, where asking whether each value is between a floor and a ceiling gives an answer that
5141/// does not combine and turns into a running minimum and maximum.
5142///
5143/// Biasing and shifting is what the answer is made of, rather than anything that reads more like the
5144/// question, because those are the operations a machine has four of. A 64 bit integer minimum is
5145/// AVX-512. So is a 64 bit arithmetic shift right, which is how the sign extension this could be
5146/// written as would have to be done. An add and a logical shift right are AVX2 and are on every
5147/// machine this runs on, so this is the form that gets four values a cycle instead of one.
5148///
5149/// Adding the bias moves the type's range to `0..=2^bits`, wrapping, so everything in range shifts
5150/// away to nothing and everything outside it leaves something behind. A negative value under an
5151/// unsigned type is caught by the same shift, because a negative `i64` read as a `u64` is enormous.
5152#[allow(clippy::cast_sign_loss, reason = "a residue is a bit pattern and not a number")]
5153fn residue<T: Narrow>(value: i64) -> u64 {
5154    let (bits, bias) = T::BIASED;
5155    (value as u64).wrapping_add(bias) >> bits
5156}
5157
5158/// Says a primitive integer narrows with `as`, and where the bottom of its range is.
5159///
5160/// `as` is a truncation and is the right operation here only because [`fit`] has already found every
5161/// residue zero, and it is what makes the second loop a narrowing store with no branch in it.
5162macro_rules! narrows {
5163    ($($ty:ty => $bias:expr),* $(,)?) => {$(
5164        impl Narrow for $ty {
5165            const BIASED: (u32, u64) = (<$ty>::BITS, $bias);
5166
5167            #[allow(
5168                clippy::cast_possible_truncation,
5169                clippy::cast_sign_loss,
5170                reason = "the caller has checked the bits this truncates away"
5171            )]
5172            fn narrow(value: i64) -> Self {
5173                value as Self
5174            }
5175        }
5176    )*};
5177}
5178
5179narrows! {
5180    i8 => 1 << 7,
5181    u8 => 0,
5182    i16 => 1 << 15,
5183    u16 => 0,
5184    i32 => 1 << 31,
5185    u32 => 0,
5186}
5187
5188/// Narrows a page's values, refusing the page if any of them does not fit.
5189///
5190/// The check first and the conversion second, rather than a fallible conversion a value at a time.
5191/// Both loops here are ones a compiler widens: [`residue`] is three instructions a lane and a
5192/// narrowing store is one. The version before this was a `TryFrom` and a `collect` into a `Result`,
5193/// which is a compare, a branch and a short circuit a value at a time, and on ClickBench 39 it was
5194/// seven percent of the query. The version after that kept a running minimum and maximum, which is
5195/// the obvious way to ask and needs a 64 bit integer minimum that AVX2 does not have, so it stayed
5196/// a value at a time and was still ten percent of the same query.
5197///
5198/// An empty page has nothing to refuse, which falls out of the accumulator starting at zero rather
5199/// than needing a case of its own.
5200fn fit<T: Narrow>(values: &[i64]) -> Result<Vec<T>> {
5201    let mut spilled = 0u64;
5202    for value in values {
5203        spilled |= residue::<T>(*value);
5204    }
5205    if spilled != 0 {
5206        return Err(invalid("page value is not of its type"));
5207    }
5208    Ok(values.iter().map(|value| T::narrow(*value)).collect())
5209}
5210
5211/// The same values back in the width the column is declared at.
5212///
5213/// A value that does not fit is a page that disagrees with the directory about what the column is,
5214/// which is a damaged file rather than a caller error, so it is refused rather than truncated.
5215fn narrowed(ty: &LogicalType, values: Vec<i64>) -> Result<Data> {
5216    Ok(match ty {
5217        LogicalType::TinyInt => Data::Int8(fit::<i8>(&values)?.into()),
5218        LogicalType::UTinyInt => Data::UInt8(fit::<u8>(&values)?.into()),
5219        LogicalType::SmallInt => Data::Int16(fit::<i16>(&values)?.into()),
5220        LogicalType::USmallInt => Data::UInt16(fit::<u16>(&values)?.into()),
5221        LogicalType::Integer | LogicalType::Date => Data::Int32(fit::<i32>(&values)?.into()),
5222        LogicalType::UInteger => Data::UInt32(fit::<u32>(&values)?.into()),
5223        LogicalType::BigInt
5224        | LogicalType::Timestamp
5225        | LogicalType::Time
5226        | LogicalType::TimeTz
5227        | LogicalType::TimestampTz
5228        | LogicalType::TimestampS
5229        | LogicalType::TimestampMs
5230        | LogicalType::TimestampNs => Data::Int64(values.into()),
5231        // A decimal is an integer of unscaled units, so the cascade reads back into whichever
5232        // integer the declared width says the column is stored as.
5233        LogicalType::Decimal { .. } => match ty.physical() {
5234            PhysicalType::Int16 => Data::Int16(fit::<i16>(&values)?.into()),
5235            PhysicalType::Int32 => Data::Int32(fit::<i32>(&values)?.into()),
5236            PhysicalType::Int64 => Data::Int64(values.into()),
5237            _ => return Err(invalid("cascade codec belongs to a decimal that is not an integer")),
5238        },
5239        _ => return Err(invalid("cascade codec belongs to a page that is not integers")),
5240    })
5241}
5242
5243/// How many bytes a part of this type costs written out plainly, which is what the cascade has to
5244/// beat before it is worth the decode.
5245fn plain_width(ty: &LogicalType) -> Option<usize> {
5246    Some(match ty {
5247        LogicalType::TinyInt | LogicalType::UTinyInt => 1,
5248        LogicalType::SmallInt | LogicalType::USmallInt => 2,
5249        LogicalType::Integer | LogicalType::UInteger | LogicalType::Date => 4,
5250        LogicalType::BigInt
5251        | LogicalType::Timestamp
5252        | LogicalType::Time
5253        | LogicalType::TimeTz
5254        | LogicalType::TimestampTz
5255        | LogicalType::TimestampS
5256        | LogicalType::TimestampMs
5257        | LogicalType::TimestampNs => 8,
5258        LogicalType::Decimal { .. } => match ty.physical() {
5259            PhysicalType::Int16 => 2,
5260            PhysicalType::Int32 => 4,
5261            PhysicalType::Int64 => 8,
5262            // The widest decimals are stored as `i128`, which the cascade does not widen into, so
5263            // they take the plain path and there is nothing here to compare against.
5264            _ => return None,
5265        },
5266        _ => return None,
5267    })
5268}
5269
5270/// A part's plain integers through the cascade, or `None` when nothing it offers is worth it.
5271///
5272/// What it has to beat is whatever the page would otherwise have cost, which is the bit packed form
5273/// where there is one and the plain width where there is not. Both are cheaper to decode than a
5274/// cascade, so a tie goes to them.
5275fn cascaded(
5276    flat: &Vector,
5277    ty: &LogicalType,
5278    packed: Option<&Packed<'_>>,
5279) -> Result<Option<Vec<u8>>> {
5280    let (Some(width), Some(data)) = (plain_width(ty), flat.data()) else { return Ok(None) };
5281    let Some(values) = widened(data) else { return Ok(None) };
5282    let plain = values.len().saturating_mul(width);
5283    let best = match packed {
5284        // The tag, the base, the word count and the words, which is what the codec 2 branch writes.
5285        Some(packed) => plain.min(21 + size_of_val(packed.words())),
5286        None => plain,
5287    };
5288    let out = integer::encode_with(&values, &Fixed)?;
5289    Ok((out.len() < best).then_some(out))
5290}
5291
5292/// A part's dictionary codes through the integer cascade, or `None` when the cascade did not pay.
5293///
5294/// Until now this stream was a `u32` a row with nothing asked of it, and on ClickBench that was
5295/// 400,185,326 bytes for every one of the 28 varchar columns, the same count for `URL` as for a
5296/// column holding the empty string in nearly every row. Codes are dense integers counted from zero
5297/// and a part holds 1024 of them, which is the shape frame of reference is best at, and a column
5298/// with one value everywhere comes back a constant costing nothing per row rather than four bytes.
5299///
5300/// The result is taken only when it is smaller than the plain form. A cascade is allowed to come
5301/// out larger on a part whose codes are genuinely wide, `URL` has about sixty million distinct
5302/// values, and there is no reason to pay for the decode when it does.
5303/// A varchar page as one FSST layer, or `None` when it did not pay.
5304///
5305/// Until now a varchar page that neither the global dictionary nor the per page dictionary claimed
5306/// was written out raw: four bytes of offset a row and then the bytes. That is the right answer for
5307/// a page of values with nothing in common and the wrong one for a page of English, and a column of
5308/// comments is the case this exists for.
5309///
5310/// One layer and not the full string cascade, which is what the payload blocks of a global
5311/// dictionary go through. The cascade is a search: it encodes the page under every candidate it has
5312/// and recurses into the integer cascade for the lengths of each one, and on TPC-H `orders` that
5313/// took the write from 6.9 s to 48.3 s. It reads back no faster than the dictionary it replaced
5314/// either, 1.807 G instructions against 1.810 G for `select o_comment from orders`, because
5315/// unpicking a nest of layers a value at a time costs what the dictionary's payload block decode
5316/// cost. Raw pages of the same column read in 0.686 G, which says the whole of the difference is
5317/// what the page has to be put back together from.
5318///
5319/// FSST alone keeps most of what the cascade found and gives all of that back. Decoding it is one
5320/// pass over the payload into one buffer, the values are laid end to end in it the way the raw form
5321/// already lays them out, and what the reader hands a chunk is views over that buffer.
5322///
5323/// The page dictionary gets first refusal because it is cheaper still, and it wins on a page whose
5324/// values repeat. What is left for this is the page whose values mostly do not, which is exactly the
5325/// page that was being written raw.
5326///
5327/// Taken only when it comes out smaller than the raw form, so a page of incompressible values pays
5328/// nothing at read time for having been offered.
5329fn text_compressed(flat: &Vector) -> Result<Option<Vec<u8>>> {
5330    let mut values: Vec<&[u8]> = Vec::with_capacity(flat.len());
5331    let mut payload = 0_usize;
5332    for row in 0..flat.len() {
5333        let text = flat.text_at(row).unwrap_or("").as_bytes();
5334        payload = payload.saturating_add(text.len());
5335        values.push(text);
5336    }
5337    // What codec 0 writes for a varchar page: an offset a row and one more, then the payload.
5338    let plain = (flat.len() + 1).saturating_mul(4).saturating_add(payload);
5339    let Some(out) = string::encode_only(string::Kind::Fsst, &values)? else {
5340        return Ok(None);
5341    };
5342    Ok((out.len() < plain).then_some(out))
5343}
5344
5345fn encoded_codes(codes: &[u32]) -> Result<Option<Vec<u8>>> {
5346    let wide: Vec<i64> = codes.iter().map(|code| i64::from(*code)).collect();
5347    let coded = integer::encode_with(&wide, &Codes)?;
5348    let plain = codes.len().saturating_mul(size_of::<u32>());
5349    Ok((coded.len() < plain).then_some(coded))
5350}
5351
5352fn encode(
5353    vector: &Vector,
5354    global: Option<&mut GlobalDictionary>,
5355) -> Result<(Vec<u8>, Option<Vec<u32>>)> {
5356    let ty = vector.logical_type();
5357    // flatten: the file writer needs a uniform scalar page and does it once per loaded chunk.
5358    let flat = vector.flatten()?;
5359    let mut out = Vec::new();
5360    let mut global_codes = None;
5361    if let Some(global) = global {
5362        let mut codes = Vec::with_capacity(flat.len());
5363        for row in 0..flat.len() {
5364            let text = flat.text_at(row).unwrap_or("");
5365            let code = global.code(text)?;
5366            global.observe(code, flat.is_null_at(row))?;
5367            codes.push(code);
5368        }
5369        global_codes = Some(codes);
5370    }
5371    let membership = global_codes.as_deref().map(unique_codes);
5372    let dictionary = if global_codes.is_none() && ty == &LogicalType::Varchar {
5373        string_dictionary(&flat)?
5374    } else {
5375        None
5376    };
5377    let compressed_text =
5378        if global_codes.is_none() && dictionary.is_none() && ty == &LogicalType::Varchar {
5379            text_compressed(&flat)?
5380        } else {
5381            None
5382        };
5383    let packed_vector = if dictionary.is_none() && global_codes.is_none() {
5384        Some(flat.bit_packed()?)
5385    } else {
5386        None
5387    };
5388    let packed = packed_vector.as_ref().and_then(Vector::packed_parts);
5389    let coded = match global_codes.as_deref() {
5390        Some(codes) => encoded_codes(codes)?,
5391        None => None,
5392    };
5393    // Only where nothing else has claimed the page, which is the plain integer case. A packed part
5394    // is still on the table because the cascade has to beat it too: the bit pack takes a part only
5395    // when it halves it, so a column that shrinks by a third was coming out whole.
5396    let cascade = if dictionary.is_none() && global_codes.is_none() {
5397        cascaded(&flat, ty, packed.as_ref())?
5398    } else {
5399        None
5400    };
5401    out.push(if coded.is_some() {
5402        4
5403    } else if cascade.is_some() {
5404        5
5405    } else if global_codes.is_some() {
5406        3
5407    } else if dictionary.is_some() {
5408        1
5409    } else if compressed_text.is_some() {
5410        6
5411    } else if packed.is_some() {
5412        2
5413    } else {
5414        0
5415    });
5416    let nulls = flat.validity();
5417    let flag = match nulls {
5418        Validity::AllValid => 0,
5419        Validity::AllInvalid => 1,
5420        Validity::Mask(_) => 2,
5421    };
5422    out.push(flag);
5423    if flag == 2 {
5424        for group in (0..vector.len()).step_by(8) {
5425            let mut bits = 0_u8;
5426            for bit in 0..8 {
5427                if group + bit < vector.len() && !flat.is_null_at(group + bit) {
5428                    bits |= 1 << bit;
5429                }
5430            }
5431            out.push(bits);
5432        }
5433    }
5434    if let Some(coded) = coded {
5435        out.extend_from_slice(&coded);
5436        return Ok((out, membership));
5437    }
5438    if let Some(cascade) = cascade {
5439        out.extend_from_slice(&cascade);
5440        return Ok((out, membership));
5441    }
5442    if let Some(codes) = global_codes {
5443        for code in codes {
5444            put_u32(&mut out, code);
5445        }
5446        return Ok((out, membership));
5447    }
5448    if let Some(dictionary) = dictionary {
5449        out.extend_from_slice(&dictionary);
5450        return Ok((out, membership));
5451    }
5452    if let Some(compressed_text) = compressed_text {
5453        out.extend_from_slice(&compressed_text);
5454        return Ok((out, membership));
5455    }
5456    if let Some(packed) = packed {
5457        if packed.offset() != 0 {
5458            return Err(invalid("writer received a sliced packed vector"));
5459        }
5460        out.push(u8::try_from(packed.width()).map_err(|_| invalid("packed width overflow"))?);
5461        out.extend_from_slice(&packed.base().to_le_bytes());
5462        put_u32(
5463            &mut out,
5464            u32::try_from(packed.words().len()).map_err(|_| invalid("too many packed words"))?,
5465        );
5466        for word in packed.words() {
5467            put_u64(&mut out, *word);
5468        }
5469        return Ok((out, membership));
5470    }
5471    let data = flat.data().ok_or_else(|| invalid("scalar column did not flatten"))?;
5472    match (ty, data) {
5473        (LogicalType::TinyInt, Data::Int8(values)) => {
5474            for value in &**values {
5475                out.extend_from_slice(&value.to_le_bytes());
5476            }
5477        }
5478        (LogicalType::UTinyInt, Data::UInt8(values)) => {
5479            for value in &**values {
5480                out.extend_from_slice(&value.to_le_bytes());
5481            }
5482        }
5483        (LogicalType::SmallInt, Data::Int16(values)) => {
5484            for value in &**values {
5485                out.extend_from_slice(&value.to_le_bytes());
5486            }
5487        }
5488        (LogicalType::USmallInt, Data::UInt16(values)) => {
5489            for value in &**values {
5490                out.extend_from_slice(&value.to_le_bytes());
5491            }
5492        }
5493        (LogicalType::UInteger, Data::UInt32(values)) => {
5494            for value in &**values {
5495                out.extend_from_slice(&value.to_le_bytes());
5496            }
5497        }
5498        (LogicalType::UBigInt, Data::UInt64(values)) => {
5499            for value in &**values {
5500                out.extend_from_slice(&value.to_le_bytes());
5501            }
5502        }
5503        (LogicalType::Integer | LogicalType::Date, Data::Int32(values)) => {
5504            for value in &**values {
5505                out.extend_from_slice(&value.to_le_bytes());
5506            }
5507        }
5508        (
5509            LogicalType::BigInt
5510            | LogicalType::Timestamp
5511            | LogicalType::Time
5512            | LogicalType::TimeTz
5513            | LogicalType::TimestampTz
5514            | LogicalType::TimestampS
5515            | LogicalType::TimestampMs
5516            | LogicalType::TimestampNs,
5517            Data::Int64(values),
5518        ) => {
5519            for value in &**values {
5520                out.extend_from_slice(&value.to_le_bytes());
5521            }
5522        }
5523        // A hugeint and a uuid are both the 128 bit lane, and a uuid's bits are the ones the rest of
5524        // the engine already carries it in, so nothing about the value changes on the way down.
5525        (LogicalType::HugeInt | LogicalType::Uuid, Data::Int128(values)) => {
5526            for value in &**values {
5527                out.extend_from_slice(&value.to_le_bytes());
5528            }
5529        }
5530        (LogicalType::UHugeInt, Data::UInt128(values)) => {
5531            for value in &**values {
5532                out.extend_from_slice(&value.to_le_bytes());
5533            }
5534        }
5535        // Plainly, in the IEEE bytes. The integer encodings do not apply to a float and none of the
5536        // float codecs is worth having before somebody has measured a corpus of them.
5537        (LogicalType::Float, Data::Float32(values)) => {
5538            for value in &**values {
5539                out.extend_from_slice(&value.to_le_bytes());
5540            }
5541        }
5542        (LogicalType::Double, Data::Float64(values)) => {
5543            for value in &**values {
5544                out.extend_from_slice(&value.to_le_bytes());
5545            }
5546        }
5547        // Three counts and not one number. Months, days and microseconds stay apart on disk because
5548        // they are apart in the value: a month is not a fixed number of days and a day is not a
5549        // fixed number of microseconds, which is the whole reason the type has three fields.
5550        (LogicalType::Interval, Data::Interval(values)) => {
5551            for (months, days, micros) in &**values {
5552                out.extend_from_slice(&months.to_le_bytes());
5553                out.extend_from_slice(&days.to_le_bytes());
5554                out.extend_from_slice(&micros.to_le_bytes());
5555            }
5556        }
5557        (LogicalType::Boolean, Data::Bool(values)) => {
5558            for value in &**values {
5559                out.push(u8::from(*value));
5560            }
5561        }
5562        // The unscaled integer and nothing else. Scale is a property of the column and it is in the
5563        // directory already, so writing it a value at a time would be paying for it twice.
5564        (LogicalType::Decimal { .. }, Data::Int16(values)) => {
5565            for value in &**values {
5566                out.extend_from_slice(&value.to_le_bytes());
5567            }
5568        }
5569        (LogicalType::Decimal { .. }, Data::Int32(values)) => {
5570            for value in &**values {
5571                out.extend_from_slice(&value.to_le_bytes());
5572            }
5573        }
5574        (LogicalType::Decimal { .. }, Data::Int64(values)) => {
5575            for value in &**values {
5576                out.extend_from_slice(&value.to_le_bytes());
5577            }
5578        }
5579        (LogicalType::Decimal { .. }, Data::Int128(values)) => {
5580            for value in &**values {
5581                out.extend_from_slice(&value.to_le_bytes());
5582            }
5583        }
5584        // A blob and a bit string go down the way a varchar does, because the layout is the same
5585        // one: an offset a value and then the bytes. What is not the same is that nothing here may
5586        // read the payload as text, which is why this arm asks the column for bytes rather than for
5587        // a string, and why the codecs above that do read text are all asked of a varchar by name.
5588        (LogicalType::Varchar | LogicalType::Blob | LogicalType::Bit, Data::Varlen(values)) => {
5589            let mut bytes = Vec::new();
5590            put_u32(&mut out, 0);
5591            for row in 0..vector.len() {
5592                let value = values.bytes(row).ok_or_else(|| invalid("string view is invalid"))?;
5593                bytes.extend_from_slice(value);
5594                put_u32(
5595                    &mut out,
5596                    u32::try_from(bytes.len())
5597                        .map_err(|_| invalid("string payload exceeds 4GiB"))?,
5598                );
5599            }
5600            out.extend_from_slice(&bytes);
5601        }
5602        _ => return Err(Error::not_implemented(format!("native page for {ty}"))),
5603    }
5604    Ok((out, membership))
5605}
5606
5607fn put_varint(out: &mut Vec<u8>, mut value: u32) {
5608    while value >= 0x80 {
5609        out.push((value as u8 & 0x7f) | 0x80);
5610        value >>= 7;
5611    }
5612    out.push(value as u8);
5613}
5614
5615/// The distinct codes of one part, which is what a stripe's membership index is merged from.
5616fn unique_codes(codes: &[u32]) -> Vec<u32> {
5617    let mut unique = codes.to_vec();
5618    unique.sort_unstable();
5619    unique.dedup();
5620    unique
5621}
5622
5623/// The union of the sorted distinct codes of every part in a stripe.
5624///
5625/// Pairwise up a tree rather than one long list concatenated and sorted. Both are the same order of
5626/// work on paper and the tree is the one that does not sort what is already in order: sixty four
5627/// sorted lists become one in six passes over the values.
5628fn merged_codes(lists: Vec<Vec<u32>>) -> Vec<u32> {
5629    let mut lists = lists;
5630    while lists.len() > 1 {
5631        let mut next = Vec::with_capacity(lists.len().div_ceil(2));
5632        for pair in lists.chunks(2) {
5633            match pair {
5634                [left, right] => next.push(merged_pair(left, right)),
5635                [only] => next.push(only.clone()),
5636                _ => {}
5637            }
5638        }
5639        lists = next;
5640    }
5641    lists.pop().unwrap_or_default()
5642}
5643
5644fn merged_pair(left: &[u32], right: &[u32]) -> Vec<u32> {
5645    let mut out = Vec::with_capacity(left.len().saturating_add(right.len()));
5646    let mut at = 0;
5647    let mut to = 0;
5648    while at < left.len() && to < right.len() {
5649        match left[at].cmp(&right[to]) {
5650            Ordering::Less => {
5651                out.push(left[at]);
5652                at += 1;
5653            }
5654            Ordering::Greater => {
5655                out.push(right[to]);
5656                to += 1;
5657            }
5658            Ordering::Equal => {
5659                out.push(left[at]);
5660                at += 1;
5661                to += 1;
5662            }
5663        }
5664    }
5665    out.extend_from_slice(&left[at..]);
5666    out.extend_from_slice(&right[to..]);
5667    out
5668}
5669
5670/// The widest bounds and the total null count of a stripe, from the bounds of its parts.
5671///
5672/// A bound that is missing from any part is missing from the stripe, because a missing bound means
5673/// nothing is known and a stripe that holds an unknown cannot claim one.
5674fn merged_range(ranges: impl Iterator<Item = Range>) -> Range {
5675    let mut merged = Range::default();
5676    let mut first = true;
5677    for range in ranges {
5678        merged.nulls = merged.nulls.saturating_add(range.nulls);
5679        // Both of these have to survive every part, so one part that could not say anything makes
5680        // the stripe unable to say it either. A sum is dropped on overflow rather than wrapped,
5681        // which leaves the stripe with exact ends and no total, which is a true thing to say.
5682        merged.sum = match (merged.sum.take(), range.sum) {
5683            (Some(held), Some(next)) if !first => held.checked_add(next),
5684            (_, next) if first => next,
5685            _ => None,
5686        };
5687        merged.exact = if first { range.exact } else { merged.exact && range.exact };
5688        if first {
5689            merged.low = range.low;
5690            merged.high = range.high;
5691            first = false;
5692            continue;
5693        }
5694        merged.low = match (merged.low.take(), range.low) {
5695            (Some(held), Some(next)) => Some(held.smaller(next)),
5696            _ => None,
5697        };
5698        merged.high = match (merged.high.take(), range.high) {
5699            (Some(held), Some(next)) => Some(held.larger(next)),
5700            _ => None,
5701        };
5702    }
5703    merged
5704}
5705
5706/// One stripe's sieves for one column: the part count, a length for each part, then their bytes.
5707///
5708/// One page for the whole stripe rather than one per part, because a part's sieve is a few hundred
5709/// bytes and sixty four of those are sixty four directory entries and sixty four reads for something
5710/// a scan walks straight through. A part with no sieve writes a length of zero and costs four bytes.
5711/// `bound` cut down to [`PART_BOUND_BYTES`], still a bound of the side it was.
5712///
5713/// A prefix of a string sorts at or before the string, so cutting one down leaves a low end that is
5714/// still a low end. A high end has to go the other way, so the cut prefix is stepped up at the last
5715/// byte that can carry it, and a prefix of nothing but `0xFF` has no such byte and gives up the
5716/// bound rather than claiming one that is too small. Anything that is not a string is already a
5717/// fixed width and is left alone.
5718fn shortened(bound: Option<Bound>, high: bool) -> Option<Bound> {
5719    match bound {
5720        Some(Bound::Bytes(mut value)) if value.len() > PART_BOUND_BYTES => {
5721            value.truncate(PART_BOUND_BYTES);
5722            if !high {
5723                return Some(Bound::Bytes(value));
5724            }
5725            while let Some(last) = value.pop() {
5726                if last < u8::MAX {
5727                    value.push(last + 1);
5728                    return Some(Bound::Bytes(value));
5729                }
5730            }
5731            None
5732        }
5733        other => other,
5734    }
5735}
5736
5737/// The ranges of one column's parts of one stripe, as a page.
5738///
5739/// The two ends and the null count, and not `exact` or the total. Those two answer a `MIN` or a
5740/// `SUM` out of the directory, and the directory already answers those per stripe, where the same
5741/// number costs sixty times less to keep. What a part range is for is skipping the part, and
5742/// skipping needs the ends. So a range read back from here says it is not exact, which is true of a
5743/// string end that was cut down anyway.
5744fn encode_part_ranges(ranges: &[Range]) -> Result<Vec<u8>> {
5745    let mut out = Vec::new();
5746    put_u32(
5747        &mut out,
5748        u32::try_from(ranges.len()).map_err(|_| invalid("too many parts in a stripe"))?,
5749    );
5750    for range in ranges {
5751        put_bound(&mut out, shortened(range.low.clone(), false).as_ref())?;
5752        put_bound(&mut out, shortened(range.high.clone(), true).as_ref())?;
5753        put_u32(&mut out, u32::try_from(range.nulls).map_err(|_| invalid("null count overflow"))?);
5754    }
5755    Ok(out)
5756}
5757
5758/// The ranges one encoded page holds, one entry per part of the stripe.
5759fn decode_part_ranges(bytes: &[u8]) -> Result<Vec<Range>> {
5760    let mut cur = Cursor { bytes, at: 0 };
5761    let parts = cur.u32()? as usize;
5762    let mut out = Vec::new();
5763    for _ in 0..parts {
5764        let low = cur.bound()?;
5765        let high = cur.bound()?;
5766        let nulls = cur.u32()? as usize;
5767        out.push(Range { low, high, nulls, exact: false, sum: None });
5768    }
5769    Ok(out)
5770}
5771
5772fn encode_sieves<'a>(sieves: impl Iterator<Item = &'a Option<Sieve>>) -> Result<Vec<u8>> {
5773    let held: Vec<&Option<Sieve>> = sieves.collect();
5774    let mut out = Vec::new();
5775    put_u32(
5776        &mut out,
5777        u32::try_from(held.len()).map_err(|_| invalid("too many parts in a stripe"))?,
5778    );
5779    for sieve in &held {
5780        let length = sieve.as_ref().map_or(0, Sieve::len);
5781        put_u32(&mut out, u32::try_from(length).map_err(|_| invalid("sieve length overflow"))?);
5782    }
5783    // flatten: a part with no sieve wrote a length of zero above and contributes no bytes here.
5784    for sieve in held.into_iter().flatten() {
5785        out.extend_from_slice(&sieve.to_bytes());
5786    }
5787    Ok(out)
5788}
5789
5790/// The sieves one encoded page holds, one entry per part of the stripe.
5791///
5792/// A part whose bytes are not a sieve this version understands comes back as `None`, which is a part
5793/// that gets read. That is how a file written by a later version of the sieve stays readable rather
5794/// than being a corrupt page.
5795fn decode_sieves(bytes: &[u8]) -> Result<Vec<Option<Sieve>>> {
5796    let parts = u32::from_le_bytes(
5797        bytes
5798            .get(..4)
5799            .ok_or_else(|| invalid("sieve page is truncated"))?
5800            .try_into()
5801            .map_err(|_| invalid("sieve page is truncated"))?,
5802    ) as usize;
5803    let mut lengths = Vec::with_capacity(parts);
5804    for part in 0..parts {
5805        let at = 4 + part * 4;
5806        let field = bytes.get(at..at + 4).ok_or_else(|| invalid("sieve page is truncated"))?;
5807        lengths.push(u32::from_le_bytes(
5808            field.try_into().map_err(|_| invalid("sieve page is truncated"))?,
5809        ) as usize);
5810    }
5811    let mut at = 4 + parts * 4;
5812    let mut out = Vec::with_capacity(parts);
5813    for length in lengths {
5814        if length == 0 {
5815            out.push(None);
5816            continue;
5817        }
5818        let end = at.checked_add(length).ok_or_else(|| invalid("sieve page is truncated"))?;
5819        let field = bytes.get(at..end).ok_or_else(|| invalid("sieve page is truncated"))?;
5820        out.push(Sieve::from_bytes(field));
5821        at = end;
5822    }
5823    if at != bytes.len() {
5824        return Err(invalid("sieve page has trailing bytes"));
5825    }
5826    Ok(out)
5827}
5828
5829/// One stripe's membership index: the code count and then the codes as ascending deltas.
5830///
5831/// The codes have to be sorted and distinct already, which is what [`unique_codes`] and
5832/// [`merged_codes`] hand over. Anything else decodes as different codes, so neither of those two is
5833/// a step a caller can skip.
5834fn encode_membership(unique: &[u32]) -> Vec<u8> {
5835    let mut out = Vec::with_capacity(unique.len().saturating_mul(2).saturating_add(5));
5836    put_varint(&mut out, u32::try_from(unique.len()).unwrap_or(u32::MAX));
5837    let mut previous = 0;
5838    for (at, &code) in unique.iter().enumerate() {
5839        put_varint(&mut out, if at == 0 { code } else { code - previous });
5840        previous = code;
5841    }
5842    out
5843}
5844
5845fn take_varint(bytes: &[u8], at: &mut usize) -> Result<u32> {
5846    let mut value = 0_u32;
5847    for shift in (0..35).step_by(7) {
5848        let byte = *bytes.get(*at).ok_or_else(|| invalid("membership varint is truncated"))?;
5849        *at += 1;
5850        let part = u32::from(byte & 0x7f);
5851        if shift == 28 && part > 0x0f {
5852            return Err(invalid("membership varint overflow"));
5853        }
5854        value = value
5855            .checked_add(
5856                part.checked_shl(shift).ok_or_else(|| invalid("membership varint overflow"))?,
5857            )
5858            .ok_or_else(|| invalid("membership varint overflow"))?;
5859        if byte & 0x80 == 0 {
5860            return Ok(value);
5861        }
5862    }
5863    Err(invalid("membership varint is too long"))
5864}
5865
5866fn decode_membership(bytes: &[u8]) -> Result<Vec<u32>> {
5867    let mut at = 0;
5868    let count = take_varint(bytes, &mut at)? as usize;
5869    let mut codes = Vec::with_capacity(count);
5870    let mut previous = 0_u32;
5871    for index in 0..count {
5872        let delta = take_varint(bytes, &mut at)?;
5873        let code = if index == 0 {
5874            delta
5875        } else {
5876            previous.checked_add(delta).ok_or_else(|| invalid("membership code overflow"))?
5877        };
5878        if index > 0 && code <= previous {
5879            return Err(invalid("membership codes are not increasing"));
5880        }
5881        codes.push(code);
5882        previous = code;
5883    }
5884    if at != bytes.len() {
5885        return Err(invalid("membership page has trailing bytes"));
5886    }
5887    Ok(codes)
5888}
5889
5890fn string_dictionary(vector: &Vector) -> Result<Option<Vec<u8>>> {
5891    let mut by_text = HashMap::new();
5892    let mut values = Vec::new();
5893    let mut codes = Vec::with_capacity(vector.len());
5894    let mut plain_bytes = 0_usize;
5895    for row in 0..vector.len() {
5896        let text = vector.text_at(row).unwrap_or("");
5897        plain_bytes = plain_bytes.saturating_add(text.len());
5898        let code = match by_text.get(text) {
5899            Some(&code) => code,
5900            None => {
5901                let code = u32::try_from(values.len())
5902                    .map_err(|_| invalid("too many dictionary values"))?;
5903                by_text.insert(text, code);
5904                values.push(text);
5905                code
5906            }
5907        };
5908        codes.push(code);
5909    }
5910    let dictionary_bytes = values.iter().map(|value| value.len()).sum::<usize>();
5911    let encoded = 8_usize
5912        .saturating_add((values.len() + 1).saturating_mul(4))
5913        .saturating_add(dictionary_bytes)
5914        .saturating_add(codes.len().saturating_mul(4));
5915    let plain = (vector.len() + 1).saturating_mul(4).saturating_add(plain_bytes);
5916    if encoded >= plain {
5917        return Ok(None);
5918    }
5919    let mut out = Vec::with_capacity(encoded);
5920    put_u32(
5921        &mut out,
5922        u32::try_from(values.len()).map_err(|_| invalid("too many dictionary values"))?,
5923    );
5924    put_u32(
5925        &mut out,
5926        u32::try_from(dictionary_bytes).map_err(|_| invalid("dictionary payload exceeds 4GiB"))?,
5927    );
5928    let mut offset = 0_u32;
5929    put_u32(&mut out, offset);
5930    for value in &values {
5931        offset = offset
5932            .checked_add(
5933                u32::try_from(value.len()).map_err(|_| invalid("dictionary value is too long"))?,
5934            )
5935            .ok_or_else(|| invalid("dictionary payload exceeds 4GiB"))?;
5936        put_u32(&mut out, offset);
5937    }
5938    for value in values {
5939        out.extend_from_slice(value.as_bytes());
5940    }
5941    for code in codes {
5942        put_u32(&mut out, code);
5943    }
5944    Ok(Some(out))
5945}
5946
5947struct EncodedDictionary {
5948    index: Vec<u8>,
5949    ranks: Vec<u8>,
5950    /// The payload as the blocks it is written as, kept apart rather than joined because joining
5951    /// them is a second copy of a thing that is already gigabytes on the columns that matter.
5952    payload: Vec<Vec<u8>>,
5953}
5954
5955/// The first eight bytes of a value as an integer that sorts the way the bytes sort.
5956fn head(bytes: &[u8]) -> u64 {
5957    let mut word = [0; 8];
5958    let take = bytes.len().min(8);
5959    word[..take].copy_from_slice(&bytes[..take]);
5960    u64::from_be_bytes(word)
5961}
5962
5963/// The sorted order of every global dictionary, one entry per column and empty where there is no
5964/// dictionary.
5965///
5966/// One column's sort has nothing to do with another's, and a table like `hits` has fifteen string
5967/// columns, so this runs across threads the way the numeric synopses above do. It is the only part
5968/// of committing a file that is more than bookkeeping, and doing it serially would show up as a
5969/// pause at the end of a load that thirty two threads had been busy with until then.
5970fn rankings(dictionaries: &[Option<GlobalDictionary>]) -> Result<Vec<Vec<(u64, u32)>>> {
5971    let present =
5972        dictionaries.iter().enumerate().filter(|(_, held)| held.is_some()).map(|(at, _)| at);
5973    let present = present.collect::<Vec<_>>();
5974    let mut orders = vec![Vec::new(); dictionaries.len()];
5975    let workers = std::thread::available_parallelism()
5976        .map_or(1, usize::from)
5977        .min(MAX_FREQUENCY_WORKERS)
5978        .min(present.len());
5979    if workers <= 1 {
5980        for at in present {
5981            if let Some(dictionary) = &dictionaries[at] {
5982                orders[at] = dictionary.ranked();
5983            }
5984        }
5985        return Ok(orders);
5986    }
5987    let width = present.len().div_ceil(workers);
5988    let pieces = std::thread::scope(|scope| {
5989        present
5990            .chunks(width)
5991            .map(|columns| {
5992                scope.spawn(|| {
5993                    columns
5994                        .iter()
5995                        .filter_map(|&at| dictionaries[at].as_ref().map(|held| (at, held.ranked())))
5996                        .collect::<Vec<_>>()
5997                })
5998            })
5999            .collect::<Vec<_>>()
6000            .into_iter()
6001            .map(|handle| {
6002                handle.join().map_err(|_| Error::internal("a dictionary sort worker panicked"))
6003            })
6004            .collect::<Result<Vec<_>>>()
6005    })?;
6006    for piece in pieces {
6007        for (at, order) in piece {
6008            orders[at] = order;
6009        }
6010    }
6011    Ok(orders)
6012}
6013
6014fn encode_global_dictionary(
6015    dictionary: GlobalDictionary,
6016    order: &[(u64, u32)],
6017) -> Result<EncodedDictionary> {
6018    let values = dictionary.offsets.len() - 1;
6019    if order.len() != values {
6020        return Err(invalid("global dictionary order does not cover its values"));
6021    }
6022    let blocks = values.div_ceil(TEXT_PAYLOAD_VALUES);
6023    let payload = encode_payload(&dictionary)?;
6024    if payload.len() != blocks {
6025        return Err(invalid("global dictionary payload is not the blocks it says it is"));
6026    }
6027    let (ranks, rank_ends) = encode_ranks(order, code_width(values))?;
6028    let rank_blocks = values.div_ceil(TEXT_RANK_BLOCK);
6029    let offset_bits = offset_width(&dictionary.offsets);
6030    let mut index = Vec::with_capacity(
6031        DICTIONARY_HEADER + offset_bytes(values, offset_bits) + (blocks + rank_blocks) * 16,
6032    );
6033    put_u32(
6034        &mut index,
6035        u32::try_from(values).map_err(|_| invalid("global dictionary has too many values"))?,
6036    );
6037    put_u32(&mut index, TEXT_PAYLOAD_VALUES as u32);
6038    put_u32(
6039        &mut index,
6040        u32::try_from(blocks).map_err(|_| invalid("global dictionary has too many blocks"))?,
6041    );
6042    put_u32(&mut index, offset_bits as u32);
6043    encode_offsets(&dictionary.offsets, offset_bits, &mut index)?;
6044    // Where each block ends, so a reader can find one. The stored blocks are shorter than the
6045    // decoded ones and by a different amount each, so this is the one thing the offsets above no
6046    // longer say.
6047    let mut at = 0_u64;
6048    for block in &payload {
6049        at = at
6050            .checked_add(block.len() as u64)
6051            .ok_or_else(|| invalid("global dictionary payload overflow"))?;
6052        put_u64(&mut index, at);
6053    }
6054    for block in &payload {
6055        put_u64(&mut index, checksum(block));
6056    }
6057    // The same two lists for the sorted order. A rank block is packed at whatever width its own
6058    // heads need, so where one ends is no longer arithmetic on the block number.
6059    if rank_ends.len() != rank_blocks {
6060        return Err(invalid("global dictionary order is not the blocks it says it is"));
6061    }
6062    for end in &rank_ends {
6063        put_u64(&mut index, *end);
6064    }
6065    let mut at = 0_usize;
6066    for end in &rank_ends {
6067        let end = usize::try_from(*end).map_err(|_| invalid("global dictionary order overflow"))?;
6068        put_u64(&mut index, checksum(&ranks[at..end]));
6069        at = end;
6070    }
6071    Ok(EncodedDictionary { index, ranks, payload })
6072}
6073
6074/// How many blocks of the payload the shape is settled on.
6075///
6076/// Eight blocks is 8,192 values, which is the sample `chooser::Sampled` draws and is that size for
6077/// the same reason. They are spread across the dictionary rather than taken off the front, because
6078/// a dictionary is in the order values were first seen and the front of it is the first morsel of
6079/// the load.
6080const PAYLOAD_SAMPLE_BLOCKS: usize = 8;
6081
6082/// The shapes the payload encoder picks between.
6083///
6084/// Narrow on purpose. The exhaustive search encodes every candidate at every level and runs at two
6085/// to six megabytes a second on this data, which over the twelve gigabytes of dictionary `hits`
6086/// carries is about an hour of processor time, so it cannot be what a load does. Each of these
6087/// settles the outer level and the one below it, which is where almost all of that hour goes, and
6088/// leaves the levels under them to the exhaustive search where the chunks are small enough for it
6089/// to cost nothing.
6090///
6091/// Measured on the five ClickBench columns that have a dictionary worth the name, at 1,024 values a
6092/// block, against the exhaustive search over the same blocks:
6093///
6094/// | column | exhaustive | FRONT then LZ | LZ then FSST | LZ then PLAIN |
6095/// |---|---|---|---|---|
6096/// | 2 | 2.923 at 4.3 MB/s | 2.587 at 21.2 | 2.593 at 36.1 | 2.538 at 53.6 |
6097/// | 13 | 3.093 at 3.1 | 3.029 at 36.4 | 2.921 at 35.7 | 2.770 at 82.9 |
6098/// | 14 | 2.330 at 2.1 | 2.283 at 24.3 | 2.213 at 23.5 | 2.113 at 67.6 |
6099/// | 39 | 2.459 at 5.3 | 2.147 at 10.6 | 2.145 at 29.3 | 2.088 at 43.1 |
6100/// | 56 | 4.694 at 6.3 | 4.381 at 51.0 | 4.172 at 50.6 | 3.983 at 86.8 |
6101///
6102/// The best of the three per column is 98 percent of the exhaustive ratio for a tenth of the time.
6103/// `FSST` and `PLAIN` on their own are in the list as a floor rather than to win. `FSST` is the
6104/// right answer for text that does not share prefixes with its neighbours, and `PLAIN` is there so
6105/// that a column nothing compresses is found out in the sample and written at a gigabyte a second
6106/// rather than searched for an answer that does not exist.
6107fn payload_shapes() -> Vec<chooser::Settled> {
6108    let integers = vec![integer::Kind::Packed];
6109    [
6110        vec![string::Kind::Front, string::Kind::Lz],
6111        vec![string::Kind::Lz, string::Kind::Fsst],
6112        vec![string::Kind::Lz, string::Kind::Plain],
6113        vec![string::Kind::Fsst],
6114        vec![string::Kind::Plain],
6115    ]
6116    .into_iter()
6117    .map(|strings| chooser::Settled::new(strings, integers.clone()))
6118    .collect()
6119}
6120
6121/// The payload as encoded blocks of [`TEXT_PAYLOAD_VALUES`] values each.
6122///
6123/// Across threads because this is the only part of committing a file that is real work rather than
6124/// bookkeeping. The blocks are the same size and cost about the same, so an index each is enough of
6125/// a queue and there is nothing to weight the way the numeric synopses are weighted.
6126fn encode_payload(dictionary: &GlobalDictionary) -> Result<Vec<Vec<u8>>> {
6127    let values = dictionary.offsets.len() - 1;
6128    let blocks = values.div_ceil(TEXT_PAYLOAD_VALUES);
6129    let run = |block: usize| {
6130        let first = block * TEXT_PAYLOAD_VALUES;
6131        let last = (first + TEXT_PAYLOAD_VALUES).min(values);
6132        (first..last)
6133            .map(|value| {
6134                let from = dictionary.offsets[value] as usize;
6135                let to = dictionary.offsets[value + 1] as usize;
6136                &dictionary.payload[from..to]
6137            })
6138            .collect::<Vec<_>>()
6139    };
6140    // A dictionary small enough to be the sample is small enough to search in full, and searching
6141    // it costs less than deciding not to.
6142    let shape = (blocks > PAYLOAD_SAMPLE_BLOCKS).then(|| settle_shape(&run, blocks)).transpose()?;
6143    let one = |block: usize| match &shape {
6144        Some(shape) => string::encode_with(&run(block), shape),
6145        None => string::encode(&run(block)),
6146    };
6147    let workers = std::thread::available_parallelism()
6148        .map_or(1, usize::from)
6149        .min(MAX_FREQUENCY_WORKERS)
6150        .min(blocks);
6151    if workers <= 1 {
6152        return (0..blocks).map(one).collect();
6153    }
6154    let next = AtomicUsize::new(0);
6155    let pieces = std::thread::scope(|scope| {
6156        (0..workers)
6157            .map(|_| {
6158                scope.spawn(|| {
6159                    let mut mine = Vec::new();
6160                    loop {
6161                        let block = next.fetch_add(1, Atomic::Relaxed);
6162                        if block >= blocks {
6163                            break;
6164                        }
6165                        mine.push((block, one(block)?));
6166                    }
6167                    Ok(mine)
6168                })
6169            })
6170            .collect::<Vec<_>>()
6171            .into_iter()
6172            .map(|handle| {
6173                handle.join().map_err(|_| Error::internal("a dictionary encode worker panicked"))?
6174            })
6175            .collect::<Result<Vec<_>>>()
6176    })?;
6177    let mut payload = vec![Vec::new(); blocks];
6178    for piece in pieces {
6179        for (block, bytes) in piece {
6180            payload[block] = bytes;
6181        }
6182    }
6183    Ok(payload)
6184}
6185
6186/// Which of [`payload_shapes`] comes out smallest over a sample of the blocks.
6187///
6188/// Every shape is encoded over the same sample and the smallest wins, which is the exhaustive
6189/// search moved up a level: over shapes of a column rather than over candidates of a chunk. The
6190/// sample is spread across the dictionary so that the first and last blocks are both in it, because
6191/// a dictionary written in first seen order has its common values at the front and its long tail at
6192/// the back, and those do not compress alike.
6193fn settle_shape<'a>(
6194    run: &dyn Fn(usize) -> Vec<&'a [u8]>,
6195    blocks: usize,
6196) -> Result<chooser::Settled> {
6197    let last = blocks - 1;
6198    let sample = (0..PAYLOAD_SAMPLE_BLOCKS)
6199        .map(|region| run(region * last / (PAYLOAD_SAMPLE_BLOCKS - 1)))
6200        .collect::<Vec<_>>();
6201    let mut best: Option<(chooser::Settled, usize)> = None;
6202    for shape in payload_shapes() {
6203        let mut size = 0;
6204        for block in &sample {
6205            size += string::encode_with(block, &shape)?.len();
6206        }
6207        if best.as_ref().is_none_or(|(_, smallest)| size < *smallest) {
6208            best = Some((shape, size));
6209        }
6210    }
6211    best.map(|(shape, _)| shape)
6212        .ok_or_else(|| invalid("no shape applies to a global dictionary payload"))
6213}
6214
6215/// The sorted order laid out the way a reader reads it, in blocks of [`TEXT_RANK_BLOCK`] entries.
6216///
6217/// Each block holds its heads first and then its codes, rather than pairing them, because a search
6218/// asks for a head at every probe and for a code about once a search. Keeping the heads together
6219/// means a probe touches eight bytes of a block rather than twelve spread over it, and the last few
6220/// probes of a search, which are the ones that land in the same block, touch the same cache line.
6221fn encode_ranks(order: &[(u64, u32)], code_bits: usize) -> Result<(Vec<u8>, Vec<u64>)> {
6222    let mut out = Vec::with_capacity(order.len() * 4);
6223    let mut ends = Vec::with_capacity(order.len().div_ceil(TEXT_RANK_BLOCK));
6224    let mut heads = Vec::with_capacity(TEXT_RANK_BLOCK);
6225    let mut codes = Vec::with_capacity(TEXT_RANK_BLOCK);
6226    for block in order.chunks(TEXT_RANK_BLOCK) {
6227        // The order is sorted by value and a head is a prefix of a value, so the heads of a block
6228        // rise, the smallest is the first and the largest is the last.
6229        let base = block.first().map_or(0, |&(head, _)| head);
6230        let span = block.last().map_or(0, |&(head, _)| head.wrapping_sub(base));
6231        let width = (u64::BITS - span.leading_zeros()) as usize;
6232        heads.clear();
6233        codes.clear();
6234        for &(head, code) in block {
6235            heads.push(head.wrapping_sub(base));
6236            codes.push(u64::from(code));
6237        }
6238        put_u64(&mut out, base);
6239        out.push(width as u8);
6240        bitpack::pack_tail(&heads, width, &mut out)
6241            .map_err(|_| invalid("global dictionary heads do not pack"))?;
6242        bitpack::pack_tail(&codes, code_bits, &mut out)
6243            .map_err(|_| invalid("global dictionary codes do not pack"))?;
6244        ends.push(out.len() as u64);
6245    }
6246    Ok((out, ends))
6247}
6248
6249/// Opens a column's global dictionary, which reads its index and none of its payload.
6250///
6251/// `keep_budget` is how many decoded payload bytes this dictionary may hold on to, and every
6252/// caller bar the test of the ceiling passes [`TEXT_KEEP_BUDGET`]. It is a parameter rather than
6253/// the constant read where it is used because a test of a ceiling that cannot be moved has to build
6254/// a quarter of a gigabyte of dictionary to reach it.
6255fn open_global_dictionary(
6256    file: Arc<File>,
6257    page: Page,
6258    ty: &LogicalType,
6259    keep_budget: usize,
6260) -> Result<Vector> {
6261    if ty != &LogicalType::Varchar {
6262        return Err(invalid("global dictionary belongs to a non-string column"));
6263    }
6264    let mut header = [0; DICTIONARY_HEADER];
6265    read_at(&file, page.offset, &mut header)?;
6266    let count = u32::from_le_bytes(header[0..4].try_into().expect("four bytes")) as usize;
6267    let per_block = u32::from_le_bytes(header[4..8].try_into().expect("four bytes")) as usize;
6268    let blocks = u32::from_le_bytes(header[8..12].try_into().expect("four bytes")) as usize;
6269    let offset_bits = u32::from_le_bytes(header[12..16].try_into().expect("four bytes")) as usize;
6270    if per_block != TEXT_PAYLOAD_VALUES {
6271        return Err(invalid("global dictionary block width differs"));
6272    }
6273    if blocks != count.div_ceil(TEXT_PAYLOAD_VALUES) {
6274        return Err(invalid("global dictionary block count differs from its value count"));
6275    }
6276    if offset_bits > u32::BITS as usize {
6277        return Err(invalid("global dictionary packs offsets past a payload"));
6278    }
6279    let offset_len = offset_bytes(count, offset_bits);
6280    // The sorted order is kept out of the index on purpose. The index is read and checksummed in
6281    // full the moment the column is first touched, and the order is half again the size of the
6282    // offsets, so putting it there would make every query that reads a string column pay for a
6283    // search that most of them never make.
6284    let ranks = count;
6285    let rank_blocks = ranks.div_ceil(TEXT_RANK_BLOCK);
6286    // Two words a payload block, one for where it ends in the file and one for its checksum, and the
6287    // same two a rank block.
6288    let hash_len = blocks
6289        .checked_add(rank_blocks)
6290        .and_then(|words| words.checked_mul(16))
6291        .ok_or_else(|| invalid("global dictionary block count overflow"))?;
6292    let index_len = DICTIONARY_HEADER
6293        .checked_add(offset_len)
6294        .and_then(|len| len.checked_add(hash_len))
6295        .ok_or_else(|| invalid("global dictionary header overflow"))?;
6296    if index_len > page.length as usize {
6297        return Err(invalid("global dictionary offset index exceeds its page"));
6298    }
6299    let mut index = vec![0; index_len];
6300    index[..DICTIONARY_HEADER].copy_from_slice(&header);
6301    read_at(&file, page.offset + DICTIONARY_HEADER as u64, &mut index[DICTIONARY_HEADER..])?;
6302    if checksum(&index) != page.hash {
6303        return Err(invalid("global dictionary index checksum differs"));
6304    }
6305    let offsets = index[DICTIONARY_HEADER..DICTIONARY_HEADER + offset_len].to_vec();
6306    let mut words = index[DICTIONARY_HEADER + offset_len..]
6307        .chunks_exact(8)
6308        .map(|part| u64::from_le_bytes(part.try_into().expect("eight bytes")))
6309        .collect::<Vec<_>>();
6310    let mut hashes = words.split_off(blocks);
6311    let mut rank_ends = hashes.split_off(blocks);
6312    let rank_hashes = rank_ends.split_off(rank_blocks);
6313    let ends = words;
6314    // A rank block packs its heads at whatever width its own values need, so its length is no longer
6315    // arithmetic on the block number and the reader has to be told where each one ends.
6316    if rank_ends.windows(2).any(|pair| pair[0] >= pair[1]) {
6317        return Err(invalid("global dictionary order blocks do not rise"));
6318    }
6319    let rank_len = usize::try_from(rank_ends.last().copied().unwrap_or_default())
6320        .map_err(|_| invalid("global dictionary rank overflow"))?;
6321    let body_len = index_len
6322        .checked_add(rank_len)
6323        .ok_or_else(|| invalid("global dictionary header overflow"))?;
6324    if body_len > page.length as usize {
6325        return Err(invalid("global dictionary order exceeds its page"));
6326    }
6327    // What the offsets bound is the decoded payload, and what the page holds is the stored one, so
6328    // the last block end is the only thing that ties the index to the length of the page.
6329    let stored_len = page.length as usize - body_len;
6330    if ends.last().copied().unwrap_or_default() as usize != stored_len
6331        || ends.windows(2).any(|pair| pair[0] > pair[1])
6332    {
6333        return Err(invalid("global dictionary blocks do not bound the payload"));
6334    }
6335    Vector::external_text(
6336        LogicalType::Varchar,
6337        Arc::new(NativeText {
6338            file,
6339            values: count,
6340            offsets,
6341            offset_bits,
6342            ranks,
6343            rank_at: page.offset + index_len as u64,
6344            rank_ends,
6345            rank_hashes,
6346            rank_blocks: (0..rank_blocks).map(|_| OnceLock::new()).collect(),
6347            code_bits: code_width(count),
6348            code_ranks: OnceLock::new(),
6349            payload: page.offset + body_len as u64,
6350            ends,
6351            hashes,
6352            blocks: (0..blocks).map(|_| OnceLock::new()).collect(),
6353            keep_budget,
6354            payload_kept: AtomicUsize::new(0),
6355            searched: Mutex::new(HashMap::new()),
6356        }),
6357    )
6358}
6359
6360/// What a stored page is, without decoding a value out of it.
6361///
6362/// Two layers, and both of them belong in the answer. The codec byte at the front of every page is
6363/// the format's own choice, and it is what says whether the column came back as codes into a table
6364/// wide dictionary, as a bit packed page, as an encoding cascade or as the bytes themselves. Under
6365/// the cascade codecs there is a second choice the encoder made per chunk, and that is what
6366/// [`integer::describe`] and [`string::describe`] already write out as `DICT(PACKED, PACKED)`.
6367///
6368/// This mirrors the tags [`decode`] reads and has to be kept beside it. A page whose header this
6369/// cannot walk comes back as text rather than as an error, because a caller asking what a file
6370/// looks like is usually asking because something is wrong with it, and a report that stops at the
6371/// first bad page is a report that says nothing about the other nine hundred.
6372fn page_encoding(ty: &LogicalType, rows: usize, bytes: &[u8]) -> String {
6373    /// The page header is the codec, the validity tag and, for a page that stores a mask, the mask.
6374    fn cascade_at(rows: usize, bytes: &[u8]) -> Result<(u8, usize)> {
6375        let mut cur = Cursor { bytes, at: 0 };
6376        let codec = cur.u8()?;
6377        if cur.u8()? == 2 {
6378            cur.take(rows.div_ceil(8))?;
6379        }
6380        Ok((codec, cur.at))
6381    }
6382    let Ok((codec, at)) = cascade_at(rows, bytes) else {
6383        return "UNREADABLE".to_string();
6384    };
6385    let tail = &bytes[at..];
6386    let described = |described: Result<String>| described.unwrap_or_else(|_| "UNREADABLE".into());
6387    match codec {
6388        0 => match ty {
6389            LogicalType::Varchar | LogicalType::Blob => "PLAIN".to_string(),
6390            _ => "FIXED".to_string(),
6391        },
6392        1 => "DICT(PLAIN)".to_string(),
6393        2 => "FOR+BITPACK".to_string(),
6394        3 => "TABLE DICT".to_string(),
6395        4 => format!("TABLE DICT({})", described(integer::describe(tail))),
6396        5 => described(integer::describe(tail)),
6397        6 => described(string::describe(tail)),
6398        other => format!("CODEC {other}"),
6399    }
6400}
6401
6402fn decode(
6403    ty: &LogicalType,
6404    rows: usize,
6405    bytes: &[u8],
6406    global: Option<Arc<Vector>>,
6407) -> Result<Vector> {
6408    let mut cur = Cursor { bytes, at: 0 };
6409    let codec = cur.u8()?;
6410    let flag = cur.u8()?;
6411    let validity = match flag {
6412        0 => Validity::AllValid,
6413        1 => Validity::AllInvalid,
6414        2 => {
6415            let mask = cur.take(rows.div_ceil(8))?;
6416            Validity::from_iter(rows, |row| mask[row / 8] >> (row % 8) & 1 == 1)
6417        }
6418        _ => return Err(invalid("page validity tag differs")),
6419    };
6420    if codec == 1 {
6421        if ty != &LogicalType::Varchar {
6422            return Err(invalid("dictionary codec belongs to a non-string page"));
6423        }
6424        let count = cur.u32()? as usize;
6425        let payload_len = cur.u32()? as usize;
6426        let offset_bytes = cur.take(
6427            (count + 1)
6428                .checked_mul(4)
6429                .ok_or_else(|| invalid("dictionary offset count overflow"))?,
6430        )?;
6431        let offsets = offset_bytes
6432            .chunks_exact(4)
6433            .map(|part| u32::from_le_bytes(part.try_into().expect("four bytes")))
6434            .collect::<Vec<_>>();
6435        let payload = cur.take(payload_len)?.to_vec();
6436        if offsets.first() != Some(&0)
6437            || offsets.last().copied().map(|last| last as usize) != Some(payload.len())
6438            || offsets.windows(2).any(|pair| pair[0] > pair[1])
6439        {
6440            return Err(invalid("dictionary offsets do not bound the payload"));
6441        }
6442        // A page, because every chunk cut out of this dictionary points at the same payload and a
6443        // page is what lets a cut be the views and nothing else.
6444        let mut strings = StringColumn::over(Buffer::from_vec(payload).into_page());
6445        for pair in offsets.windows(2) {
6446            strings.push_in_place(pair[0] as usize, (pair[1] - pair[0]) as usize)?;
6447        }
6448        let mut codes = Vec::with_capacity(rows);
6449        for _ in 0..rows {
6450            codes.push(cur.u32()?);
6451        }
6452        if codes.iter().any(|code| *code as usize >= count) {
6453            return Err(invalid("dictionary code is out of range"));
6454        }
6455        if cur.at != bytes.len() {
6456            return Err(invalid("dictionary page has trailing bytes"));
6457        }
6458        let dictionary = Vector::flat(LogicalType::Varchar, Data::Varlen(strings))?;
6459        return Ok(Vector::dictionary(codes, dictionary)?.with_validity(validity));
6460    }
6461    if codec == 3 || codec == 4 {
6462        let dictionary = global.ok_or_else(|| invalid("global code page has no dictionary"))?;
6463        let codes = if codec == 4 {
6464            // The cascade holds the whole tail of the page and says how long it is itself, so the
6465            // check that nothing is left over is the one the decoder already makes.
6466            let wide = integer::decode(&bytes[cur.at..])?;
6467            if wide.len() != rows {
6468                return Err(invalid("encoded code page holds the wrong number of rows"));
6469            }
6470            // Converted in one pass and checked in the same one, rather than a fallible conversion
6471            // per code. A `Result` an element is a short circuit the loop cannot be vectorized past,
6472            // and it was costing about twelve instructions a row to narrow a number that already
6473            // fits. Every code a file holds is inside a `u32` or the file is corrupt, so the check
6474            // belongs once at the end: or the codes together and the answer has a bit set above the
6475            // low thirty two, or the sign bit, exactly when one of them did.
6476            let mut codes = Vec::with_capacity(wide.len());
6477            let mut seen = 0_i64;
6478            for &code in &wide {
6479                seen |= code;
6480                codes.push(code as u32);
6481            }
6482            if seen < 0 || seen > i64::from(u32::MAX) {
6483                return Err(invalid("code is not a code"));
6484            }
6485            codes
6486        } else {
6487            let mut codes = Vec::with_capacity(rows);
6488            for _ in 0..rows {
6489                codes.push(cur.u32()?);
6490            }
6491            if cur.at != bytes.len() {
6492                return Err(invalid("global code page has trailing bytes"));
6493            }
6494            codes
6495        };
6496        let highest = codes.iter().copied().max();
6497        return Ok(Vector::stable_dictionary_validated(codes, dictionary, highest)?
6498            .with_validity(validity));
6499    }
6500    if codec == 6 {
6501        if ty != &LogicalType::Varchar {
6502            return Err(invalid("compressed text codec belongs to a non-string page"));
6503        }
6504        // As codec 5, the layer holds the whole tail of the page and says how long it is itself.
6505        // It comes back as one buffer with the values laid end to end and where each one ends, which
6506        // is the raw form's layout, so what is left to do here is what codec 0 does.
6507        let (payload, ends) = string::decode_flat(&bytes[cur.at..])?.into_parts();
6508        if ends.len() != rows {
6509            return Err(invalid("compressed text page holds the wrong number of rows"));
6510        }
6511        // A page, because this is read once and handed out a chunk at a time, and a cut of a paged
6512        // payload moves views rather than bytes.
6513        let mut values = StringColumn::over(Buffer::from_vec(payload).into_page());
6514        let mut start = 0;
6515        for end in ends {
6516            let len = end
6517                .checked_sub(start)
6518                .ok_or_else(|| invalid("compressed text value ends before it starts"))?;
6519            values.push_in_place(start, len)?;
6520            start = end;
6521        }
6522        return Ok(Vector::flat(ty.clone(), Data::Varlen(values))?.with_validity(validity));
6523    }
6524    if codec == 5 {
6525        // The cascade holds the whole tail of the page and says how long it is itself.
6526        let values = integer::decode(&bytes[cur.at..])?;
6527        if values.len() != rows {
6528            return Err(invalid("cascade page holds the wrong number of rows"));
6529        }
6530        let data = narrowed(ty, values)?;
6531        return Ok(Vector::flat(ty.clone(), data)?.with_validity(validity));
6532    }
6533    if codec == 2 {
6534        let width = u32::from(cur.u8()?);
6535        let base = i128::from_le_bytes(cur.take(16)?.try_into().expect("sixteen bytes"));
6536        let count = cur.u32()? as usize;
6537        let mut words = Vec::with_capacity(count);
6538        for _ in 0..count {
6539            words.push(cur.u64()?);
6540        }
6541        if cur.at != bytes.len() {
6542            return Err(invalid("packed page has trailing bytes"));
6543        }
6544        return Ok(Vector::packed(ty.clone(), words, width, base, rows)?.with_validity(validity));
6545    }
6546    if codec != 0 {
6547        return Err(invalid("page codec is unknown"));
6548    }
6549    let data = match ty {
6550        LogicalType::TinyInt => {
6551            let values = cur.take(rows)?;
6552            Data::Int8(values.iter().map(|item| *item as i8).collect::<Vec<_>>().into())
6553        }
6554        LogicalType::UTinyInt => Data::UInt8(cur.take(rows)?.to_vec().into()),
6555        LogicalType::SmallInt => {
6556            let values =
6557                cur.take(rows.checked_mul(2).ok_or_else(|| invalid("page size overflow"))?)?;
6558            Data::Int16(
6559                values
6560                    .chunks_exact(2)
6561                    .map(|item| i16::from_le_bytes(item.try_into().expect("two bytes")))
6562                    .collect::<Vec<_>>()
6563                    .into(),
6564            )
6565        }
6566        LogicalType::USmallInt => {
6567            let values =
6568                cur.take(rows.checked_mul(2).ok_or_else(|| invalid("page size overflow"))?)?;
6569            Data::UInt16(
6570                values
6571                    .chunks_exact(2)
6572                    .map(|item| u16::from_le_bytes(item.try_into().expect("two bytes")))
6573                    .collect::<Vec<_>>()
6574                    .into(),
6575            )
6576        }
6577        LogicalType::UInteger => {
6578            let values =
6579                cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
6580            Data::UInt32(
6581                values
6582                    .chunks_exact(4)
6583                    .map(|item| u32::from_le_bytes(item.try_into().expect("four bytes")))
6584                    .collect::<Vec<_>>()
6585                    .into(),
6586            )
6587        }
6588        LogicalType::UBigInt => {
6589            let values =
6590                cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
6591            Data::UInt64(
6592                values
6593                    .chunks_exact(8)
6594                    .map(|item| u64::from_le_bytes(item.try_into().expect("eight bytes")))
6595                    .collect::<Vec<_>>()
6596                    .into(),
6597            )
6598        }
6599        LogicalType::Integer | LogicalType::Date => {
6600            let values =
6601                cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
6602            Data::Int32(
6603                values
6604                    .chunks_exact(4)
6605                    .map(|item| i32::from_le_bytes(item.try_into().expect("four bytes")))
6606                    .collect::<Vec<_>>()
6607                    .into(),
6608            )
6609        }
6610        LogicalType::BigInt
6611        | LogicalType::Timestamp
6612        | LogicalType::Time
6613        | LogicalType::TimeTz
6614        | LogicalType::TimestampTz
6615        | LogicalType::TimestampS
6616        | LogicalType::TimestampMs
6617        | LogicalType::TimestampNs => {
6618            let values =
6619                cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
6620            Data::Int64(
6621                values
6622                    .chunks_exact(8)
6623                    .map(|item| i64::from_le_bytes(item.try_into().expect("eight bytes")))
6624                    .collect::<Vec<_>>()
6625                    .into(),
6626            )
6627        }
6628        LogicalType::HugeInt | LogicalType::Uuid => {
6629            let values =
6630                cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
6631            Data::Int128(
6632                values
6633                    .chunks_exact(16)
6634                    .map(|item| i128::from_le_bytes(item.try_into().expect("sixteen bytes")))
6635                    .collect::<Vec<_>>()
6636                    .into(),
6637            )
6638        }
6639        LogicalType::UHugeInt => {
6640            let values =
6641                cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
6642            Data::UInt128(
6643                values
6644                    .chunks_exact(16)
6645                    .map(|item| u128::from_le_bytes(item.try_into().expect("sixteen bytes")))
6646                    .collect::<Vec<_>>()
6647                    .into(),
6648            )
6649        }
6650        LogicalType::Float => {
6651            let values =
6652                cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
6653            Data::Float32(
6654                values
6655                    .chunks_exact(4)
6656                    .map(|item| f32::from_le_bytes(item.try_into().expect("four bytes")))
6657                    .collect::<Vec<_>>()
6658                    .into(),
6659            )
6660        }
6661        LogicalType::Double => {
6662            let values =
6663                cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
6664            Data::Float64(
6665                values
6666                    .chunks_exact(8)
6667                    .map(|item| f64::from_le_bytes(item.try_into().expect("eight bytes")))
6668                    .collect::<Vec<_>>()
6669                    .into(),
6670            )
6671        }
6672        LogicalType::Interval => {
6673            let values =
6674                cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
6675            Data::Interval(
6676                values
6677                    .chunks_exact(16)
6678                    .map(|item| {
6679                        (
6680                            i32::from_le_bytes(item[..4].try_into().expect("four bytes")),
6681                            i32::from_le_bytes(item[4..8].try_into().expect("four bytes")),
6682                            i64::from_le_bytes(item[8..].try_into().expect("eight bytes")),
6683                        )
6684                    })
6685                    .collect::<Vec<_>>()
6686                    .into(),
6687            )
6688        }
6689        LogicalType::Boolean => {
6690            let values = cur.take(rows)?;
6691            if values.iter().any(|value| *value > 1) {
6692                return Err(invalid("boolean page has another value"));
6693            }
6694            Data::Bool(values.iter().map(|value| *value == 1).collect::<Vec<_>>().into())
6695        }
6696        // Whichever integer the declared width says, which is the mapping the rest of the engine
6697        // already uses for a decimal in memory.
6698        LogicalType::Decimal { .. } => match ty.physical() {
6699            PhysicalType::Int16 => {
6700                let values =
6701                    cur.take(rows.checked_mul(2).ok_or_else(|| invalid("page size overflow"))?)?;
6702                Data::Int16(
6703                    values
6704                        .chunks_exact(2)
6705                        .map(|item| i16::from_le_bytes(item.try_into().expect("two bytes")))
6706                        .collect::<Vec<_>>()
6707                        .into(),
6708                )
6709            }
6710            PhysicalType::Int32 => {
6711                let values =
6712                    cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
6713                Data::Int32(
6714                    values
6715                        .chunks_exact(4)
6716                        .map(|item| i32::from_le_bytes(item.try_into().expect("four bytes")))
6717                        .collect::<Vec<_>>()
6718                        .into(),
6719                )
6720            }
6721            PhysicalType::Int64 => {
6722                let values =
6723                    cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
6724                Data::Int64(
6725                    values
6726                        .chunks_exact(8)
6727                        .map(|item| i64::from_le_bytes(item.try_into().expect("eight bytes")))
6728                        .collect::<Vec<_>>()
6729                        .into(),
6730                )
6731            }
6732            _ => {
6733                let values =
6734                    cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
6735                Data::Int128(
6736                    values
6737                        .chunks_exact(16)
6738                        .map(|item| i128::from_le_bytes(item.try_into().expect("sixteen bytes")))
6739                        .collect::<Vec<_>>()
6740                        .into(),
6741                )
6742            }
6743        },
6744        LogicalType::Varchar | LogicalType::Blob | LogicalType::Bit => {
6745            let offset_bytes = cur
6746                .take((rows + 1).checked_mul(4).ok_or_else(|| invalid("offset count overflow"))?)?;
6747            let offsets = offset_bytes
6748                .chunks_exact(4)
6749                .map(|part| u32::from_le_bytes(part.try_into().expect("four bytes")))
6750                .collect::<Vec<_>>();
6751            let payload = cur.take(bytes.len() - cur.at)?.to_vec();
6752            if offsets.first() != Some(&0)
6753                || offsets.last().copied().map(|last| last as usize) != Some(payload.len())
6754                || offsets.windows(2).any(|pair| pair[0] > pair[1])
6755            {
6756                return Err(invalid("string offsets do not bound the payload"));
6757            }
6758            // A page for the reason the dictionary payload above is one: the page is read once and
6759            // handed out a chunk at a time, and a cut of a paged payload moves views rather than
6760            // bytes.
6761            //
6762            // A varchar is checked for text on the way in and a blob and a bit string are not,
6763            // because the second pair never claimed to hold any. Reading them through the checking
6764            // seam would refuse a column for holding exactly what it was told to hold.
6765            let mut values = StringColumn::over(Buffer::from_vec(payload).into_page());
6766            let text = ty == &LogicalType::Varchar;
6767            for pair in offsets.windows(2) {
6768                let (at, len) = (pair[0] as usize, (pair[1] - pair[0]) as usize);
6769                if text {
6770                    values.push_in_place(at, len)?;
6771                } else {
6772                    values.push_bytes_in_place(at, len)?;
6773                }
6774            }
6775            Data::Varlen(values)
6776        }
6777        _ => return Err(Error::not_implemented(format!("native page for {ty}"))),
6778    };
6779    if cur.at != bytes.len() {
6780        return Err(invalid("page has trailing bytes"));
6781    }
6782    Ok(Vector::flat(ty.clone(), data)?.with_validity(validity))
6783}
6784
6785#[cfg(test)]
6786mod tests {
6787    use std::fs;
6788    use std::io::{Seek, SeekFrom, Write};
6789    use std::path::PathBuf;
6790    use std::time::{SystemTime, UNIX_EPOCH};
6791
6792    use rudb_common::Stat;
6793    use rudb_common::Value;
6794    use rudb_common::bounds::{Frequencies, Op, Remainder, Zones};
6795    use rudb_common::stat::Provenance;
6796
6797    use super::*;
6798
6799    #[test]
6800    fn checksum_matches_fixed_vectors() {
6801        assert_eq!(checksum(b""), 0xef46_db37_51d8_e999);
6802        assert_eq!(checksum(b"a"), 0xd24e_c4f1_a98c_6e5b);
6803        assert_eq!(checksum(b"abc"), 0x44bc_2cf5_ad77_0999);
6804    }
6805
6806    fn path(label: &str) -> PathBuf {
6807        let stamp = SystemTime::now().duration_since(UNIX_EPOCH).expect("time advances").as_nanos();
6808        std::env::temp_dir().join(format!("rudb-native-{label}-{}-{stamp}.rdb", std::process::id()))
6809    }
6810
6811    /// A read names the offset it wants, so a cursor somebody else moved cannot reach it.
6812    #[test]
6813    fn a_read_at_an_offset_ignores_where_another_thread_left_the_cursor() {
6814        const SPANS: usize = 64;
6815        const SPAN: usize = 512;
6816        let path = path("positional");
6817        let content: Vec<u8> =
6818            (0..SPANS).flat_map(|span| std::iter::repeat_n(span as u8, SPAN)).collect();
6819        fs::write(&path, &content).expect("the file is written");
6820        let file = Arc::new(File::open(&path).expect("the file opens"));
6821        std::thread::scope(|scope| {
6822            for _ in 0..8 {
6823                let file = Arc::clone(&file);
6824                scope.spawn(move || {
6825                    for _ in 0..64 {
6826                        for span in 0..SPANS {
6827                            let mut bytes = [0_u8; SPAN];
6828                            read_at(&file, (span * SPAN) as u64, &mut bytes)
6829                                .expect("the span reads");
6830                            assert!(
6831                                bytes.iter().all(|byte| *byte == span as u8),
6832                                "span {span} came back as {}",
6833                                bytes[0],
6834                            );
6835                        }
6836                    }
6837                });
6838            }
6839        });
6840        let mut past = [0_u8; SPAN];
6841        let end = (SPANS * SPAN) as u64;
6842        let error = read_at(&file, end, &mut past).expect_err("a read past the end is refused");
6843        assert!(error.message().contains("ends before its declared length"), "{error}");
6844        drop(file);
6845        let _ = fs::remove_file(&path);
6846    }
6847
6848    /// The writer records where it put a page and puts it there, whatever the cursor is doing.
6849    ///
6850    /// The cursor is moved between the steps that record an offset, which is what reading the pages
6851    /// back to build the frequencies does on a platform with no `pread`. Without the fix the
6852    /// directory lands on top of a page and the file fails to reopen.
6853    #[test]
6854    fn a_writer_puts_a_page_where_it_said_it_did_wherever_the_cursor_has_got_to() {
6855        let path = path("cursor");
6856        let mut writer = Writer::create(
6857            &path,
6858            "items",
6859            vec![
6860                Field::required("id", LogicalType::Integer),
6861                Field::new("text", LogicalType::Varchar),
6862            ],
6863        )
6864        .expect("new file");
6865        writer.append(&sample()).expect("first part");
6866        writer.file.seek(SeekFrom::Start(0)).expect("the cursor goes back to the header");
6867        writer.append(&sample()).expect("second part");
6868        writer.file.seek(SeekFrom::Start(1)).expect("and somewhere useless again");
6869        writer.finish().expect("commit");
6870        let reader = Reader::open(&path).expect("reopen from disk");
6871        assert_eq!(reader.table().rows(), 6);
6872        let ids = reader.read(0, &[0]).expect("the integer page reads back");
6873        assert_eq!(ids.value_at(0, 0), Value::Integer(4));
6874        assert_eq!(ids.value_at(2, 0), Value::Integer(-2));
6875        let text = reader.read(1, &[1]).expect("the text page reads back");
6876        assert_eq!(text.value_at(1, 0), Value::Null);
6877        assert_eq!(text.value_at(2, 0), Value::Varchar("long text after a slash".into()));
6878        // Nothing the directory points at may run past the end of the file, which is the shape the
6879        // failure took: a page recorded at an offset the directory had already been written over.
6880        let end = reader.table().stripes().iter().flat_map(|stripe| {
6881            stripe
6882                .pages
6883                .iter()
6884                .map(|page| page.offset + u64::from(page.length))
6885                .chain(std::iter::once(stripe.index.offset + u64::from(stripe.index.length)))
6886        });
6887        let last = end.fold(HEADER, u64::max);
6888        let directory = fs::metadata(&path).expect("the file is there").len();
6889        assert!(last <= directory, "a page runs to {last} in a file of {directory} bytes");
6890        fs::remove_file(path).expect("remove scratch file");
6891    }
6892
6893    /// How long a global dictionary index is, read out of the page's own header.
6894    ///
6895    /// The tests below damage a byte of the order or of the payload, so they need to know where each
6896    /// one starts, and working it out here rather than writing a number down means adding something
6897    /// to the index does not quietly turn one of them into a test that damages the index instead.
6898    fn dictionary_index_len(header: &[u8; DICTIONARY_HEADER]) -> u64 {
6899        let count = u64::from(u32::from_le_bytes(header[0..4].try_into().expect("four bytes")));
6900        let blocks = u64::from(u32::from_le_bytes(header[8..12].try_into().expect("four bytes")));
6901        let bits = u32::from_le_bytes(header[12..16].try_into().expect("four bytes")) as usize;
6902        let rank_blocks = count.div_ceil(TEXT_RANK_BLOCK as u64);
6903        DICTIONARY_HEADER as u64
6904            + offset_bytes(count as usize, bits) as u64
6905            + (blocks + rank_blocks) * 16
6906    }
6907
6908    /// How long the sorted order is, which is where its last block ends.
6909    fn last_rank_end(file: &File, offset: u64, header: &[u8; DICTIONARY_HEADER]) -> u64 {
6910        let count = u64::from(u32::from_le_bytes(header[0..4].try_into().expect("four bytes")));
6911        let blocks = u64::from(u32::from_le_bytes(header[8..12].try_into().expect("four bytes")));
6912        let bits = u32::from_le_bytes(header[12..16].try_into().expect("four bytes")) as usize;
6913        let rank_blocks = count.div_ceil(TEXT_RANK_BLOCK as u64);
6914        let at = offset
6915            + DICTIONARY_HEADER as u64
6916            + offset_bytes(count as usize, bits) as u64
6917            + blocks * 16
6918            + (rank_blocks - 1) * 8;
6919        let mut end = [0; 8];
6920        read_at(file, at, &mut end).expect("the last rank block end");
6921        u64::from_le_bytes(end)
6922    }
6923
6924    fn sample() -> Chunk {
6925        Chunk::new(vec![
6926            Vector::from_values(
6927                LogicalType::Integer,
6928                &[Value::Integer(4), Value::Integer(9), Value::Integer(-2)],
6929            )
6930            .expect("integers"),
6931            Vector::from_values(
6932                LogicalType::Varchar,
6933                &[
6934                    Value::Varchar("alpha".into()),
6935                    Value::Null,
6936                    Value::Varchar("long text after a slash".into()),
6937                ],
6938            )
6939            .expect("strings"),
6940        ])
6941        .expect("matching rows")
6942    }
6943
6944    fn sample_ids() -> Chunk {
6945        Chunk::new(vec![
6946            Vector::flat(LogicalType::Integer, Data::Int32(vec![7, 8, 9].into()))
6947                .expect("integers"),
6948        ])
6949        .expect("one column")
6950    }
6951
6952    #[test]
6953    fn the_planner_gets_the_null_count_off_the_same_directory_the_bounds_are_in() {
6954        // Six rows, two of them null. `IS NULL` used to get the same fifth any unreadable
6955        // condition gets, and the number was in the stripe entry next to the bounds all along.
6956        let path = path("nulls_for_the_planner");
6957        let mut writer =
6958            Writer::create(&path, "items", vec![Field::new("a", LogicalType::Integer)])
6959                .expect("new file");
6960        let rows = Chunk::new(vec![
6961            Vector::from_values(
6962                LogicalType::Integer,
6963                &[
6964                    Value::Integer(4),
6965                    Value::Null,
6966                    Value::Integer(9),
6967                    Value::Null,
6968                    Value::Integer(1),
6969                    Value::Integer(2),
6970                ],
6971            )
6972            .expect("integers"),
6973        ])
6974        .expect("one column");
6975        writer.append(&rows).expect("the only part");
6976        writer.finish().expect("commit");
6977        let reader = Reader::open(&path).expect("reopen from disk");
6978        let stripes = Stripes::new(reader);
6979        let column = stripes.column("a").expect("the file has that column");
6980        assert_eq!(stripes.nulls(column), Stat::exact(2, Provenance::NullCount));
6981        // A column the file does not have. Zero here would be a fact about a column that is not
6982        // there, which the planner would then divide by.
6983        assert_eq!(stripes.nulls(column + 1), Stat::Unknown);
6984        fs::remove_file(&path).expect("clean up");
6985    }
6986
6987    #[test]
6988    fn the_planner_gets_a_row_count_per_value_off_a_complete_synopsis() {
6989        // The whole of the frequency half of #1106, end to end over a real file. Six rows, three
6990        // of one value and two of another, and a complete synopsis because six rows is well inside
6991        // what the writer can account for. The estimate for `id = 4` is three rows rather than a
6992        // sixth of the table, and for a value the file does not hold it is none.
6993        let path = path("frequencies_for_the_planner");
6994        let mut writer =
6995            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
6996                .expect("new file");
6997        let rows = Chunk::new(vec![
6998            Vector::from_values(
6999                LogicalType::Integer,
7000                &[
7001                    Value::Integer(4),
7002                    Value::Integer(4),
7003                    Value::Integer(4),
7004                    Value::Integer(9),
7005                    Value::Integer(9),
7006                    Value::Integer(1),
7007                ],
7008            )
7009            .expect("integers"),
7010        ])
7011        .expect("one column");
7012        writer.append(&rows).expect("the only part");
7013        writer.finish().expect("commit");
7014        let reader = Reader::open(&path).expect("reopen from disk");
7015        let common = Common::new(reader);
7016        assert_eq!(common.rows(), 6);
7017        let column = common.column("id").expect("the file has that column");
7018        assert_eq!(common.column("nothing"), None);
7019        assert_eq!(
7020            common.rows_with(column, &Bound::Int(4)),
7021            Stat::exact(3, Provenance::FrequencySynopsis)
7022        );
7023        // Not in the file, and a synopsis that accounts for all six rows proves it.
7024        assert_eq!(
7025            common.rows_with(column, &Bound::Int(7)),
7026            Stat::exact(0, Provenance::FrequencySynopsis)
7027        );
7028        // A constant of another domain against an integer column. Nothing in the list compares
7029        // with it, so the zero above would be an artefact of the mismatch rather than a fact.
7030        assert_eq!(common.rows_with(column, &Bound::Bytes(b"four".to_vec())), Stat::Unknown);
7031        // A complete list has no remainder. Answering one of no rows over no values would hand the
7032        // caller a division to special case, and the counts above already answer this column.
7033        assert_eq!(common.remainder(column), None);
7034        fs::remove_file(&path).expect("clean up");
7035    }
7036
7037    /// A table directory with nothing in it but a name and one column, for the section tests.
7038    ///
7039    /// The section table is orthogonal to everything else in a directory, so the tests that pin it
7040    /// say so by starting from the emptiest table that encodes.
7041    fn bare_table(sections: Vec<Section>) -> Table {
7042        Table {
7043            name: "linked".to_owned(),
7044            fields: vec![Field::required("id", LogicalType::Integer)],
7045            stripes: Vec::new(),
7046            rows: 0,
7047            dictionaries: vec![None],
7048            distincts: vec![None],
7049            frequencies: vec![None],
7050            clustering: None,
7051            generation: 1,
7052            sections,
7053        }
7054    }
7055
7056    fn a_key_map_section() -> Section {
7057        Section {
7058            kind: *section::KEY_MAP,
7059            id: 1,
7060            generation: 3,
7061            extents: 1,
7062            extent_page: HEADER,
7063            extent_bytes: section::EXTENT_BYTES as u32,
7064            hash: 0x1234_5678_9abc_def0,
7065            flags: 0,
7066            header_bytes: 24,
7067        }
7068    }
7069
7070    #[test]
7071    fn a_section_table_round_trips_through_a_directory() {
7072        let mut later = a_key_map_section();
7073        later.kind = *b"RUDBZZ9\0";
7074        later.id = 2;
7075        let table = bare_table(vec![a_key_map_section(), later]);
7076        let directory = encode_directory(&table).expect("directory");
7077        let decoded = decode_directory(&directory, 1 << 20).expect("reopen");
7078        assert_eq!(decoded.sections(), &[a_key_map_section(), later]);
7079        // The second is a kind this build has no name for, and it survived the round trip anyway.
7080        // That is what keeps an old build from silently discarding a newer build's work when it
7081        // rewrites a directory.
7082        assert!(decoded.sections()[0].known());
7083        assert!(!decoded.sections()[1].known());
7084    }
7085
7086    #[test]
7087    fn a_directory_written_before_the_section_table_reads_as_a_table_with_none() {
7088        // The G1 exit criterion, at the directory level. A format 22 directory is exactly this
7089        // build's directory with the trailing section block cut off, so cutting it off is the
7090        // honest way to make one: no fixture to go stale, and no separate encoder to drift.
7091        let directory = encode_directory(&bare_table(Vec::new())).expect("directory");
7092        let block = SECTIONS.len() + size_of::<u64>() + size_of::<u16>();
7093        let older = &directory[..directory.len() - block];
7094        let decoded = decode_directory(older, 1 << 20).expect("a directory from before sections");
7095        assert!(decoded.sections().is_empty());
7096        assert_eq!(decoded.generation(), 0, "a format 22 table recorded no generation");
7097        assert_eq!(decoded.name(), "linked");
7098        assert_eq!(decoded.fields().len(), 1, "everything before the block still decodes");
7099    }
7100
7101    #[test]
7102    fn a_file_stamped_with_the_previous_format_still_opens_and_reads() {
7103        // The same criterion end to end, which is the one the milestone actually asks for: a build
7104        // that knows about sections opens a file written by a build that did not, with no rewrite
7105        // and no repair, and answers from it. The version field is patched rather than a file
7106        // committed by an old binary because the bytes either side of it are identical: format 22
7107        // and format 23 differ only in a trailing directory block, and a reader that stops before
7108        // that block gets a table with no sections.
7109        let path = path("format_twenty_two");
7110        let mut writer =
7111            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
7112                .expect("new file");
7113        let rows = Chunk::new(vec![
7114            Vector::from_values(
7115                LogicalType::Integer,
7116                &[Value::Integer(1), Value::Integer(2), Value::Integer(3)],
7117            )
7118            .expect("integers"),
7119        ])
7120        .expect("one column");
7121        writer.append(&rows).expect("the only part");
7122        writer.finish().expect("commit");
7123
7124        let file = OpenOptions::new().write(true).open(&path).expect("reopen to patch");
7125        write_at(&file, 8, &22_u32.to_le_bytes()).expect("stamp the older format");
7126        drop(file);
7127
7128        let reader = Reader::open(&path).expect("a format 22 file opens unchanged");
7129        assert_eq!(reader.table().rows(), 3);
7130        assert!(reader.table().sections().is_empty());
7131
7132        // And a format this build has never written is still refused, so the accept set is a list
7133        // and not an absence of a check.
7134        let file = OpenOptions::new().write(true).open(&path).expect("reopen to patch");
7135        write_at(&file, 8, &21_u32.to_le_bytes()).expect("stamp an unreadable format");
7136        drop(file);
7137        let error = Reader::open(&path).expect_err("format 21 is not readable");
7138        assert!(error.to_string().contains("format 21"), "{error}");
7139
7140        fs::remove_file(&path).expect("clean up");
7141    }
7142
7143    #[test]
7144    fn a_section_whose_extent_table_is_outside_the_file_is_refused() {
7145        // The bound the format has to check and `section` cannot, because only the reader knows how
7146        // big the file is. Reading the payload a section like this names would be reading whatever
7147        // else happens to be at that offset, which is the one way a graph section could turn into a
7148        // wrong answer rather than a slow one.
7149        let mut past = a_key_map_section();
7150        past.extent_page = 1 << 30;
7151        let directory = encode_directory(&bare_table(vec![past])).expect("directory");
7152        let error = decode_directory(&directory, 1 << 20).expect_err("refused");
7153        assert!(error.to_string().contains("outside the file"), "{error}");
7154
7155        let mut inside_the_header = a_key_map_section();
7156        inside_the_header.extent_page = 8;
7157        let directory = encode_directory(&bare_table(vec![inside_the_header])).expect("directory");
7158        assert!(
7159            decode_directory(&directory, 1 << 20).is_err(),
7160            "a section may not overlap a header"
7161        );
7162    }
7163
7164    #[test]
7165    fn a_section_recorded_as_not_built_is_legal_and_names_no_bytes() {
7166        // Section 3.7: a relationship that does not fit the budget is recorded with its size so
7167        // that `rudb_links()` can report what a larger budget would buy. That record is a section
7168        // entry with no extents, so it has to survive a round trip while naming nothing.
7169        let not_built = Section {
7170            kind: *section::FORWARD_LINK,
7171            id: 9,
7172            generation: 3,
7173            extents: 0,
7174            extent_page: 0,
7175            extent_bytes: 0,
7176            hash: 0,
7177            flags: 0,
7178            header_bytes: 0,
7179        };
7180        let directory = encode_directory(&bare_table(vec![not_built])).expect("directory");
7181        let decoded = decode_directory(&directory, 1 << 20).expect("reopen");
7182        assert_eq!(decoded.sections(), &[not_built]);
7183
7184        // But a section with no extents that still names an extent table is incoherent, and an
7185        // incoherent entry is a torn directory rather than a relationship that was skipped.
7186        let mut incoherent = not_built;
7187        incoherent.extent_bytes = 28;
7188        incoherent.extent_page = HEADER;
7189        let directory = encode_directory(&bare_table(vec![incoherent])).expect("directory");
7190        assert!(decode_directory(&directory, 1 << 20).is_err());
7191    }
7192
7193    #[test]
7194    fn a_directory_naming_more_sections_than_the_bound_is_refused() {
7195        let directory = encode_directory(&bare_table(Vec::new())).expect("directory");
7196        let mut torn = directory.clone();
7197        let count_at = torn.len() - size_of::<u16>();
7198        torn[count_at..].copy_from_slice(&u16::MAX.to_le_bytes());
7199        // Not an allocation of sixty five thousand entries off a torn count: either the bound
7200        // refuses it or the bytes run out, and both are errors rather than a read past the end.
7201        assert!(decode_directory(&torn, 1 << 20).is_err());
7202    }
7203
7204    /// A committed one column file of `rows` integers, for the attach tests.
7205    fn linked_file(label: &str, rows: i32) -> PathBuf {
7206        let path = path(label);
7207        let mut writer =
7208            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
7209                .expect("new file");
7210        let values = (0..rows).map(Value::Integer).collect::<Vec<_>>();
7211        let chunk =
7212            Chunk::new(vec![Vector::from_values(LogicalType::Integer, &values).expect("integers")])
7213                .expect("one column");
7214        writer.append(&chunk).expect("the only part");
7215        writer.finish().expect("commit");
7216        path
7217    }
7218
7219    fn a_key_map_payload() -> Vec<u8> {
7220        // Shaped like one without being one: this crate never reads a payload, so what matters here
7221        // is that every byte comes back and that the header the entry measures is at the front.
7222        (0..512_u32).flat_map(u32::to_le_bytes).collect()
7223    }
7224
7225    #[test]
7226    fn a_section_attached_to_a_committed_file_reads_back_byte_for_byte() {
7227        let path = linked_file("attach", 64);
7228        let payload = a_key_map_payload();
7229        let table = attach(
7230            &path,
7231            "items",
7232            &[section::Attachment {
7233                kind: *section::KEY_MAP,
7234                id: 0,
7235                flags: 2,
7236                header_bytes: 40,
7237                bytes: &payload,
7238            }],
7239        )
7240        .expect("attach a key map");
7241        assert_eq!(table.sections().len(), 1);
7242
7243        let reader = Reader::open(&path).expect("reopen after the attach");
7244        let held = reader.table().sections();
7245        assert_eq!(held.len(), 1);
7246        assert_eq!(held[0].kind, *section::KEY_MAP);
7247        assert_eq!(held[0].flags, 2, "the form a reader must not have to guess");
7248        assert_eq!(held[0].header_bytes, 40);
7249        // The generation is the one the pages were written at, not the one the attach committed at.
7250        // Attaching a section moved no row, so a section written by it is current, and a second
7251        // table added to this file later would not make it stale.
7252        assert_eq!(held[0].generation, 1);
7253        assert!(held[0].usable(reader.table().generation()));
7254        assert_eq!(reader.payload(&held[0]).expect("read the payload"), payload);
7255        assert_eq!(reader.extents(&held[0]).expect("extent table").len(), 1);
7256
7257        fs::remove_file(&path).expect("clean up");
7258    }
7259
7260    #[test]
7261    fn attaching_a_section_answers_every_row_exactly_as_before() {
7262        // Section 3.1 end to end, and the reason the whole layer is safe to build incrementally. A
7263        // file with a section in it and the same file without one have to agree row for row, so the
7264        // comparison is made against the answers taken before the attach rather than against a
7265        // constant somebody typed.
7266        let path = linked_file("attach_changes_nothing", 300);
7267        let before = Reader::open(&path).expect("open before");
7268        let rows = before.table().rows();
7269        let first = before.read(0, &[0]).expect("read before");
7270        let values = (0..rows).map(|at| first.value_at(at, 0)).collect::<Vec<_>>();
7271        let layout = before.layout().columns_total();
7272        drop(before);
7273
7274        let payload = a_key_map_payload();
7275        attach(
7276            &path,
7277            "items",
7278            &[section::Attachment {
7279                kind: *section::KEY_MAP,
7280                id: 0,
7281                flags: 0,
7282                header_bytes: 0,
7283                bytes: &payload,
7284            }],
7285        )
7286        .expect("attach");
7287
7288        let after = Reader::open(&path).expect("open after");
7289        assert_eq!(after.table().rows(), rows);
7290        let read = after.read(0, &[0]).expect("read after");
7291        for (at, value) in values.iter().enumerate() {
7292            assert_eq!(&read.value_at(at, 0), value, "row {at} moved");
7293        }
7294        assert_eq!(
7295            after.layout().columns_total(),
7296            layout,
7297            "an attach appends and does not rewrite a column page"
7298        );
7299
7300        fs::remove_file(&path).expect("clean up");
7301    }
7302
7303    #[test]
7304    fn a_rebuilt_section_replaces_the_one_it_supersedes() {
7305        // Rebuilding a key map has to be a write and not a question. If an attach added rather than
7306        // replaced, a table rebuilt a few times would name several maps for one column and a reader
7307        // would have to pick, which is a decision with no right answer in it.
7308        let path = linked_file("attach_twice", 32);
7309        let one = a_key_map_payload();
7310        let two = vec![7_u8; 1024];
7311        let entry = |bytes| section::Attachment {
7312            kind: *section::KEY_MAP,
7313            id: 4,
7314            flags: 1,
7315            header_bytes: 0,
7316            bytes,
7317        };
7318        attach(&path, "items", &[entry(&one)]).expect("first build");
7319        attach(&path, "items", &[entry(&two)]).expect("rebuild");
7320
7321        let reader = Reader::open(&path).expect("reopen");
7322        let held = reader.table().sections();
7323        assert_eq!(held.len(), 1, "one map per column and not one per build");
7324        assert_eq!(reader.payload(&held[0]).expect("payload"), two);
7325
7326        fs::remove_file(&path).expect("clean up");
7327    }
7328
7329    #[test]
7330    fn an_attach_carries_through_a_kind_it_does_not_know() {
7331        // The first of section 3.2's three rules, at the point where it is easiest to break: a build
7332        // that rewrites a directory has to carry an entry it has no name for, or opening a file with
7333        // an older binary and attaching one section quietly deletes the work of a newer one.
7334        let path = linked_file("attach_unknown", 16);
7335        let payload = vec![3_u8; 96];
7336        attach(
7337            &path,
7338            "items",
7339            &[section::Attachment {
7340                kind: *b"RUDBZZ9\0",
7341                id: 1,
7342                flags: 0,
7343                header_bytes: 0,
7344                bytes: &payload,
7345            }],
7346        )
7347        .expect("a kind this build does not know still writes");
7348        let key_map = a_key_map_payload();
7349        attach(
7350            &path,
7351            "items",
7352            &[section::Attachment {
7353                kind: *section::KEY_MAP,
7354                id: 0,
7355                flags: 0,
7356                header_bytes: 0,
7357                bytes: &key_map,
7358            }],
7359        )
7360        .expect("attach beside it");
7361
7362        let reader = Reader::open(&path).expect("reopen");
7363        let held = reader.table().sections();
7364        assert_eq!(held.len(), 2, "the unfamiliar entry survived a directory rewrite");
7365        let unknown = held.iter().find(|one| !one.known()).expect("the unfamiliar one");
7366        assert_eq!(reader.payload(unknown).expect("its bytes are still there"), payload);
7367
7368        fs::remove_file(&path).expect("clean up");
7369    }
7370
7371    #[test]
7372    fn a_payload_of_nothing_is_a_relationship_recorded_as_not_built() {
7373        let path = linked_file("attach_not_built", 8);
7374        attach(
7375            &path,
7376            "items",
7377            &[section::Attachment {
7378                kind: *section::FORWARD_LINK,
7379                id: 2,
7380                flags: 0,
7381                header_bytes: 0,
7382                bytes: &[],
7383            }],
7384        )
7385        .expect("record a link that did not fit the budget");
7386
7387        let reader = Reader::open(&path).expect("reopen");
7388        let held = reader.table().sections();
7389        assert_eq!(held.len(), 1);
7390        assert_eq!(held[0].extents, 0);
7391        assert_eq!(held[0].extent_page, 0, "an entry that names no bytes points at none");
7392        assert!(reader.extents(&held[0]).expect("no extent table").is_empty());
7393        assert!(reader.payload(&held[0]).expect("no payload").is_empty());
7394
7395        fs::remove_file(&path).expect("clean up");
7396    }
7397
7398    #[test]
7399    fn a_payload_past_one_extent_is_split_and_joined_back() {
7400        // Issue #745's rule, exercised rather than argued. One byte past the bound is the smallest
7401        // payload that has to be two extents, and it is the case a split written for the common
7402        // size gets wrong.
7403        let path = linked_file("attach_two_extents", 8);
7404        let payload = vec![0x5a_u8; section::MAX_EXTENT as usize + 1];
7405        attach(
7406            &path,
7407            "items",
7408            &[section::Attachment {
7409                kind: *section::KEY_MAP,
7410                id: 0,
7411                flags: 0,
7412                header_bytes: 0,
7413                bytes: &payload,
7414            }],
7415        )
7416        .expect("attach a payload past the bound");
7417
7418        let reader = Reader::open(&path).expect("reopen");
7419        let held = reader.table().sections();
7420        let extents = reader.extents(&held[0]).expect("extent table");
7421        assert_eq!(extents.len(), 2, "one byte past the bound is two extents");
7422        assert_eq!(extents[0].length, section::MAX_EXTENT);
7423        assert_eq!(extents[1].length, 1);
7424        assert_eq!(extents[1].first, u64::from(section::MAX_EXTENT));
7425        // And the extent the caller wants is readable on its own, which is the point of the split.
7426        assert_eq!(reader.extent(&extents[1]).expect("the last extent"), vec![0x5a]);
7427        assert_eq!(reader.payload(&held[0]).expect("the whole payload").len(), payload.len());
7428
7429        fs::remove_file(&path).expect("clean up");
7430    }
7431
7432    #[test]
7433    fn a_torn_extent_is_refused_rather_than_decoded() {
7434        let path = linked_file("attach_torn", 8);
7435        let payload = a_key_map_payload();
7436        attach(
7437            &path,
7438            "items",
7439            &[section::Attachment {
7440                kind: *section::KEY_MAP,
7441                id: 0,
7442                flags: 0,
7443                header_bytes: 0,
7444                bytes: &payload,
7445            }],
7446        )
7447        .expect("attach");
7448
7449        let reader = Reader::open(&path).expect("reopen");
7450        let extent = reader.extents(&reader.table().sections()[0]).expect("extent table")[0];
7451        let file = OpenOptions::new().write(true).open(&path).expect("reopen to corrupt");
7452        write_at(&file, extent.offset + 7, &[0xff]).expect("flip a byte of the payload");
7453        drop(file);
7454
7455        let reader = Reader::open(&path).expect("the table still opens");
7456        let error = reader
7457            .payload(&reader.table().sections()[0])
7458            .expect_err("a corrupt payload is not handed out");
7459        assert!(error.to_string().contains("checksum"), "{error}");
7460        // And the table is still readable, which is section 3.1: a section that cannot be trusted
7461        // costs the query its shortcut and nothing else.
7462        assert_eq!(reader.read(0, &[0]).expect("the column is untouched").width(), 1);
7463
7464        fs::remove_file(&path).expect("clean up");
7465    }
7466
7467    #[test]
7468    fn attaching_to_a_file_of_the_previous_format_is_refused_rather_than_done() {
7469        // Readable is not writable. A format 22 directory has no section block, and adding one
7470        // without moving the number in the header would leave a file claiming a format it is not.
7471        let path = linked_file("attach_old_format", 8);
7472        let file = OpenOptions::new().write(true).open(&path).expect("reopen to patch");
7473        write_at(&file, 8, &22_u32.to_le_bytes()).expect("stamp the older format");
7474        drop(file);
7475
7476        let payload = a_key_map_payload();
7477        let error = attach(
7478            &path,
7479            "items",
7480            &[section::Attachment {
7481                kind: *section::KEY_MAP,
7482                id: 0,
7483                flags: 0,
7484                header_bytes: 0,
7485                bytes: &payload,
7486            }],
7487        )
7488        .expect_err("format 22 cannot gain a section");
7489        assert!(error.to_string().contains("format 22"), "{error}");
7490        assert!(Reader::open(&path).expect("and the file is untouched").table().rows() == 8);
7491
7492        fs::remove_file(&path).expect("clean up");
7493    }
7494
7495    #[test]
7496    fn a_section_header_longer_than_its_payload_is_refused_at_the_write() {
7497        let path = linked_file("attach_bad_header", 8);
7498        let error = attach(
7499            &path,
7500            "items",
7501            &[section::Attachment {
7502                kind: *section::KEY_MAP,
7503                id: 0,
7504                flags: 0,
7505                header_bytes: 40,
7506                bytes: &[1, 2, 3],
7507            }],
7508        )
7509        .expect_err("a writer's bug stops at the write");
7510        assert!(error.to_string().contains("header is longer"), "{error}");
7511
7512        fs::remove_file(&path).expect("clean up");
7513    }
7514
7515    #[test]
7516    fn attaching_to_a_name_the_file_does_not_hold_says_so() {
7517        let path = linked_file("attach_wrong_name", 8);
7518        let error = attach(&path, "orders", &[]).expect_err("no such table");
7519        assert!(error.to_string().contains("orders"), "{error}");
7520        fs::remove_file(&path).expect("clean up");
7521    }
7522
7523    #[test]
7524    fn the_planner_gets_an_exact_count_for_a_leading_value_of_an_incomplete_synopsis() {
7525        // The case a complete synopsis does not cover, and the one worth the most. 16,000 rows over
7526        // 601 distinct values, 10,000 of them holding a single value and the rest spread ten apiece
7527        // over six hundred more. The writer holds 512 values, so the list is a prefix and most of
7528        // the tail is outside it. The counts inside it are still exact, because the pass recounts
7529        // the candidates that survived it, so `id = 1` is ten thousand rows rather than the
7530        // twenty six a distinct count of 601 would divide its way to.
7531        let path = path("frequency_prefix_for_the_planner");
7532        let mut writer =
7533            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
7534                .expect("new file");
7535        let mut values = vec![Value::Integer(1); 10_000];
7536        for _ in 0..10 {
7537            values.extend((0..600).map(|tail| Value::Integer(1_000 + tail)));
7538        }
7539        // A vector holds 8,192 rows, so this goes in as several parts. The pass that takes the
7540        // synopsis walks the whole column rather than a part, so the counts are the same either way.
7541        for part in values.chunks(8_000) {
7542            let rows = Chunk::new(vec![
7543                Vector::from_values(LogicalType::Integer, part).expect("integers"),
7544            ])
7545            .expect("one column");
7546            writer.append(&rows).expect("a part");
7547        }
7548        writer.finish().expect("commit");
7549        let reader = Reader::open(&path).expect("reopen from disk");
7550        let prefix =
7551            reader.frequency_prefix(0).expect("a readable synopsis").expect("the column has one");
7552        // A prefix and not the whole column, and the writer said how many rows anything left out of
7553        // it can hold.
7554        assert_eq!(prefix.entries.len(), 512);
7555        assert_eq!(prefix.omitted_max, 10);
7556        let common = Common::new(reader);
7557        assert_eq!(common.rows(), 16_000);
7558        let column = common.column("id").expect("the file has that column");
7559        assert_eq!(
7560            common.rows_with(column, &Bound::Int(1)),
7561            Stat::exact(10_000, Provenance::FrequencySynopsis)
7562        );
7563        // In the prefix, because ties go to the smaller value and the prefix reaches 1,510.
7564        assert_eq!(
7565            common.rows_with(column, &Bound::Int(1_100)),
7566            Stat::exact(10, Provenance::FrequencySynopsis)
7567        );
7568        // Outside it, and a prefix says nothing about a value it does not list. Not zero, which is
7569        // what a complete list would say, and the file holds ten rows of this one.
7570        assert_eq!(common.rows_with(column, &Bound::Int(1_550)), Stat::Unknown);
7571        // Not in the file at all, and still nothing rather than a zero. A prefix cannot tell the
7572        // two apart, which is the whole of what it gives up.
7573        assert_eq!(common.rows_with(column, &Bound::Int(9_999)), Stat::Unknown);
7574        // What the prefix left out, which is what turns the unknown above into a number. The 512
7575        // entries account for 15,110 rows, so 890 are left for the 89 values the writer dropped,
7576        // and 890 over 89 is the ten rows each of them really holds.
7577        let remainder = common.remainder(column).expect("the list is a prefix");
7578        assert_eq!(remainder, Remainder { rows: 890, listed: 512, most: 10 });
7579        assert_eq!(remainder.rows / (601 - remainder.listed), 10);
7580        fs::remove_file(&path).expect("clean up");
7581    }
7582
7583    /// A file with no table in it is a file, and opening it says so rather than failing.
7584    #[test]
7585    fn a_file_holding_no_table_commits_and_opens_and_a_table_can_be_added_to_it() {
7586        let path = path("empty");
7587        Writer::empty(&path).expect("a file with nothing in it");
7588        let catalog = Catalog::open(&path).expect("the empty file opens");
7589        assert_eq!(catalog.len(), 0);
7590        assert!(catalog.is_empty());
7591        assert_eq!(catalog.names().count(), 0);
7592        // The next generation goes over the top of it the way it goes over any other, which is what
7593        // says this is a committed file and not a special case somebody has to know about.
7594        let mut writer =
7595            Writer::open(&path, "items", vec![Field::required("id", LogicalType::Integer)])
7596                .expect("a table goes into the empty file");
7597        writer.append(&sample_ids()).expect("rows");
7598        writer.finish().expect("commit");
7599        let catalog = Catalog::open(&path).expect("the file opens again");
7600        assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
7601        fs::remove_file(&path).expect("clean up");
7602    }
7603
7604    /// A committed table with no rows is a name the next generation takes over, and one with rows
7605    /// is a name it refuses.
7606    ///
7607    /// The refusal is what it always was and it is load bearing: carrying a table that holds rows
7608    /// forward means reading and rewriting its pages, and a writer that quietly wrote a second
7609    /// entry under the same name would leave a file with two tables a reader cannot tell apart. An
7610    /// empty one has no pages and no reader, so there is nothing to carry and nothing to lose, and
7611    /// taking its place is what lets a schema committed by an earlier session be loaded by a stream
7612    /// instead of through memory.
7613    #[test]
7614    fn a_committed_empty_table_gives_up_its_name_and_one_with_rows_does_not() {
7615        let path = path("empty-name");
7616        let field = || vec![Field::required("id", LogicalType::Integer)];
7617        Writer::create(&path, "items", field()).expect("new file").finish().expect("commit");
7618        let catalog = Catalog::open(&path).expect("the file opens");
7619        assert_eq!(catalog.rows().collect::<Vec<_>>(), vec![("items", 0)]);
7620
7621        let mut writer = Writer::open(&path, "items", field()).expect("the empty name is free");
7622        writer.append(&sample_ids()).expect("rows");
7623        writer.finish().expect("commit");
7624        let catalog = Catalog::open(&path).expect("the file opens again");
7625        // One entry and not two. The generation replaced the empty table rather than joining it.
7626        assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
7627        let held = catalog.rows().collect::<Vec<_>>();
7628        assert_eq!(held.len(), 1);
7629        assert!(held[0].1 > 0, "the rows that were appended are the ones the catalog counts");
7630
7631        // The same call against the same name now that it holds rows, which is still refused.
7632        let error = Writer::open(&path, "items", field()).expect_err("a name with rows is taken");
7633        assert!(error.to_string().contains("same name"), "{error}");
7634        fs::remove_file(&path).expect("clean up");
7635    }
7636
7637    #[test]
7638    fn committed_file_reopens_and_reads_only_requested_columns() {
7639        let path = path("reopen");
7640        let mut writer = Writer::create(
7641            &path,
7642            "items",
7643            vec![
7644                Field::required("id", LogicalType::Integer),
7645                Field::new("text", LogicalType::Varchar),
7646            ],
7647        )
7648        .expect("new file");
7649        writer.append(&sample()).expect("first part");
7650        writer.append(&sample()).expect("second part");
7651        writer.finish().expect("commit");
7652        let reader = Reader::open(&path).expect("reopen from disk");
7653        assert_eq!(reader.table().rows(), 6);
7654        // Two appends below the stripe bound are two parts of one stripe, which is the whole point
7655        // of the split: the directory describes the stripe and the scan still reads a part.
7656        assert_eq!(reader.table().stripes().len(), 1);
7657        assert_eq!(reader.parts(), 2);
7658        assert_eq!(reader.part_rows(0), 3);
7659        assert_eq!(reader.part_rows(1), 3);
7660        let text = reader.read(1, &[1]).expect("only text page");
7661        assert_eq!(text.width(), 1);
7662        assert_eq!(text.value_at(1, 0), Value::Null);
7663        assert_eq!(text.value_at(2, 0), Value::Varchar("long text after a slash".into()));
7664        let sparse = reader.read_sparse(1, &[1]).expect("one part without its whole page");
7665        assert_eq!(sparse.width(), 1);
7666        assert_eq!(sparse.value_at(1, 0), Value::Null);
7667        assert_eq!(sparse.value_at(2, 0), Value::Varchar("long text after a slash".into()));
7668        assert!(!reader.skips_codes(0, 1, &[0]).expect("alpha is in the stripe"));
7669        assert!(!reader.skips_codes(0, 1, &[2]).expect("long text is in the stripe"));
7670        assert!(reader.skips_codes(0, 1, &[3]).expect("unknown code is absent"));
7671        let count = reader.read(0, &[]).expect("no page is needed for count");
7672        assert_eq!(count.len(), 3);
7673        assert!(reader.skips(0, &[Probe { column: 0, op: Op::Greater, value: Bound::Int(100) }]));
7674        assert!(!reader.skips(0, &[Probe { column: 0, op: Op::Greater, value: Bound::Int(0) }]));
7675        let integers = reader.top_frequencies(0, 1).expect("valid integer synopsis").expect("kept");
7676        assert_eq!(
7677            integers,
7678            vec![(Value::Integer(-2), 2), (Value::Integer(4), 2), (Value::Integer(9), 2),]
7679        );
7680        let strings = reader.top_frequencies(1, 1).expect("valid string synopsis").expect("kept");
7681        assert_eq!(strings.len(), 3);
7682        assert!(strings.contains(&(Value::Null, 2)));
7683        assert!(strings.contains(&(Value::Varchar("alpha".into()), 2)));
7684        assert!(strings.contains(&(Value::Varchar("long text after a slash".into()), 2)));
7685        fs::remove_file(path).expect("remove scratch file");
7686    }
7687
7688    /// Two pipeline instances handing over whole runs, which is what makes the native sink safe to
7689    /// instance.
7690    ///
7691    /// The runs arrive in the order the instances finished reading them rather than in source
7692    /// order, and the second one to finish is the one that read the earlier rows. Each run is still
7693    /// a stripe of its own and the table still reads back in source order, which is the whole of
7694    /// what the writer promises about ordering.
7695    #[test]
7696    fn runs_handed_over_out_of_order_still_read_back_in_source_order() {
7697        let path = path("interleaved-runs");
7698        let mut writer =
7699            Writer::create(&path, "interleaved", vec![Field::new("v", LogicalType::BigInt)])
7700                .expect("new file");
7701        for morsel in [2_u64, 0, 3, 1] {
7702            let parts = (0..4_u64)
7703                .map(|chunk| {
7704                    let first = i64::try_from(morsel * 32 + chunk * 8).expect("small");
7705                    let values =
7706                        (0..8_i64).map(|row| Value::BigInt(first + row)).collect::<Vec<_>>();
7707                    let column =
7708                        Vector::from_values(LogicalType::BigInt, &values).expect("a column");
7709                    ((morsel, chunk), Chunk::new(vec![column]).expect("one column"))
7710                })
7711                .collect::<Vec<_>>();
7712            writer.append_stripe(parts).expect("a stripe");
7713        }
7714        writer.finish().expect("commit");
7715
7716        let reader = Reader::open(&path).expect("valid directory");
7717        assert_eq!(reader.table().stripes().len(), 4, "a run is a stripe of its own");
7718        assert_eq!(reader.table().rows(), 128);
7719        for part in 0..16_usize {
7720            let read = reader.read(part, &[0]).expect("a part back");
7721            for row in 0..8_usize {
7722                let want = i64::try_from(part * 8 + row).expect("small");
7723                assert_eq!(read.value_at(row, 0), Value::BigInt(want), "part {part} row {row}");
7724            }
7725        }
7726        fs::remove_file(path).expect("remove scratch file");
7727    }
7728
7729    /// Runs from different callers may interleave and may not overlap, and the commit is what
7730    /// catches an overlap.
7731    #[test]
7732    fn runs_that_overlap_each_other_are_refused_at_commit() {
7733        let path = path("overlapping-runs");
7734        let mut writer =
7735            Writer::create(&path, "overlapping", vec![Field::new("v", LogicalType::BigInt)])
7736                .expect("new file");
7737        let one = |order: (u64, u64)| {
7738            let column =
7739                Vector::from_values(LogicalType::BigInt, &[Value::BigInt(1)]).expect("a column");
7740            (order, Chunk::new(vec![column]).expect("one column"))
7741        };
7742        // The second run sits inside the first rather than after it, which is a thing no instance
7743        // holding its own contiguous run can produce and a thing the file cannot represent.
7744        writer.append_stripe(vec![one((0, 0)), one((0, 2))]).expect("a stripe");
7745        writer.append_stripe(vec![one((0, 1))]).expect("a stripe");
7746        let error = writer.finish().expect_err("the runs overlap");
7747        assert!(error.message().contains("source order"), "{error}");
7748        fs::remove_file(path).expect("remove scratch file");
7749    }
7750
7751    /// A stripe holds [`STRIPE_PARTS`] parts, so a run longer than that is a caller bug rather than
7752    /// something to split, and the writer says so at the door instead of quietly cutting it in two.
7753    #[test]
7754    fn a_run_longer_than_a_stripe_is_refused() {
7755        let path = path("overlong-run");
7756        let mut writer =
7757            Writer::create(&path, "overlong", vec![Field::new("v", LogicalType::BigInt)])
7758                .expect("new file");
7759        let parts = (0..=STRIPE_PARTS)
7760            .map(|at| {
7761                let column = Vector::from_values(LogicalType::BigInt, &[Value::BigInt(1)])
7762                    .expect("a column");
7763                let chunk = Chunk::new(vec![column]).expect("one column");
7764                ((0, u64::try_from(at).expect("small")), chunk)
7765            })
7766            .collect::<Vec<_>>();
7767        let error = writer.append_stripe(parts).expect_err("one part too many");
7768        assert!(error.message().contains("more parts than it holds"), "{error}");
7769        fs::remove_file(path).expect("remove scratch file");
7770    }
7771
7772    /// Parts past the stripe bound start a new stripe, and every part stays addressable on its own.
7773    ///
7774    /// This is the shape the format exists for, so both ends of the split are checked here. The
7775    /// directory holds three stripes rather than a hundred and thirty one, and a read of any one
7776    /// part still answers with that part's rows rather than with its whole stripe's.
7777    #[test]
7778    fn parts_past_the_stripe_bound_start_a_new_stripe() {
7779        let path = path("stripe-bound");
7780        let mut writer = Writer::create(
7781            &path,
7782            "items",
7783            vec![
7784                Field::required("id", LogicalType::Integer),
7785                Field::new("text", LogicalType::Varchar),
7786            ],
7787        )
7788        .expect("new file");
7789        let parts = STRIPE_PARTS * 2 + 3;
7790        for part in 0..parts {
7791            let id = part as i32;
7792            let chunk = Chunk::new(vec![
7793                Vector::from_values(
7794                    LogicalType::Integer,
7795                    &[Value::Integer(id), Value::Integer(-id)],
7796                )
7797                .expect("integers"),
7798                Vector::from_values(
7799                    LogicalType::Varchar,
7800                    &[Value::Varchar(format!("value {part}")), Value::Null],
7801                )
7802                .expect("strings"),
7803            ])
7804            .expect("matching rows");
7805            writer.append(&chunk).expect("one part");
7806        }
7807        writer.finish().expect("commit");
7808
7809        let reader = Reader::open(&path).expect("reopen from disk");
7810        assert_eq!(reader.parts(), parts);
7811        assert_eq!(reader.table().rows(), parts * 2);
7812        assert_eq!(reader.table().stripes().len(), parts.div_ceil(STRIPE_PARTS));
7813        assert_eq!(reader.table().stripes()[0].parts(), STRIPE_PARTS);
7814        assert_eq!(reader.table().stripes()[0].rows(), STRIPE_PARTS * 2);
7815        assert_eq!(reader.table().stripes()[2].parts(), 3);
7816        // Backwards on purpose. The reader keeps four stripes a column, so a scan that walks the
7817        // table the other way is what catches a cache that only ever holds what it just read.
7818        for part in (0..parts).rev() {
7819            let dense = reader.read(part, &[0, 1]).expect("a whole page read");
7820            let sparse = reader.read_sparse(part, &[0, 1]).expect("one part read");
7821            for chunk in [&dense, &sparse] {
7822                assert_eq!(chunk.len(), 2, "part {part} has its own row count");
7823                assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
7824                assert_eq!(chunk.value_at(1, 0), Value::Integer(-(part as i32)));
7825                assert_eq!(chunk.value_at(0, 1), Value::Varchar(format!("value {part}")));
7826                assert_eq!(chunk.value_at(1, 1), Value::Null);
7827            }
7828        }
7829        // The bounds are merged over the stripe, so they answer for the range the whole stripe
7830        // covers and not for the part that was asked about.
7831        let above = [Probe { column: 0, op: Op::Greater, value: Bound::Int(100) }];
7832        assert!(reader.skips(0, &above), "the first stripe stops at 63");
7833        assert!(!reader.skips(STRIPE_PARTS * 2, &above), "the third stripe reaches 130");
7834        fs::remove_file(path).expect("remove scratch file");
7835    }
7836
7837    /// A scattered value in the column that decides `WHERE UserID = ?`.
7838    fn scattered(n: i64) -> i64 {
7839        n.wrapping_mul(-7_046_029_254_386_353_131)
7840    }
7841
7842    /// A part whose sieve does not hold the constant is skipped, and a range would skip none of them.
7843    ///
7844    /// This is ClickBench query 19 in miniature. The values are spread over the whole of `BIGINT`, so
7845    /// every stripe's bounds cover nearly all of it and rule out nothing, and the part that really
7846    /// holds the value is the only one a scan has to read.
7847    #[test]
7848    fn a_part_is_skipped_when_its_sieve_does_not_hold_the_constant() {
7849        let path = path("sieve-skip");
7850        let mut writer =
7851            Writer::create(&path, "hits", vec![Field::required("id", LogicalType::BigInt)])
7852                .expect("new file");
7853        let parts = STRIPE_PARTS + 3;
7854        // Big enough that the filter is worth its bytes. A part of eight numbers packs to under a
7855        // hundred bytes and the smallest filter there is is sixty nine, so a filter over a part
7856        // that small costs about as much to read as the rows do and is no longer written.
7857        let per_part = 128;
7858        for part in 0..parts {
7859            let held: Vec<Value> = (0..per_part)
7860                .map(|row| Value::BigInt(scattered((part * per_part + row) as i64)))
7861                .collect();
7862            let chunk =
7863                Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
7864                    .expect("one column");
7865            writer.append(&chunk).expect("one part");
7866        }
7867        writer.finish().expect("commit");
7868
7869        let reader = Reader::open(&path).expect("reopen from disk");
7870        let probe = |value: i64| Probe {
7871            column: 0,
7872            op: Op::Equal,
7873            value: Bound::Int(i128::from(scattered(value))),
7874        };
7875        for wanted in [0_i64, (per_part + 1) as i64, (parts * per_part - 1) as i64] {
7876            let tests = [probe(wanted)];
7877            let kept: Vec<usize> = (0..parts).filter(|&part| !reader.skips(part, &tests)).collect();
7878            let home = wanted as usize / per_part;
7879            assert!(kept.contains(&home), "the part holding {wanted} is read");
7880            // A filter answers maybe, so a part it keeps need not hold the value. Sixty seven parts
7881            // of a hundred and twenty eight numbers each, at a dozen bits a value, is about one
7882            // stray part across the whole file and that is what this leaves room for.
7883            assert!(kept.len() <= 2, "{wanted} keeps {kept:?}, which is more than one stray part");
7884        }
7885        let absent = [probe((parts * per_part) as i64 + 1)];
7886        let kept = (0..parts).filter(|&part| !reader.skips(part, &absent)).count();
7887        assert!(kept <= 1, "{kept} parts of {parts} kept a value no part holds");
7888        // The same probes against the bounds alone, which is what this replaces. A column of
7889        // scattered numbers has a range per stripe that covers nearly the whole type.
7890        let tests = [probe(0)];
7891        assert!(
7892            reader.table().stripes().iter().all(|stripe| !stripe.zone.skips(&tests)),
7893            "the bounds rule out no stripe at all"
7894        );
7895        fs::remove_file(path).expect("remove scratch file");
7896    }
7897
7898    /// A part whose own bounds rule out an ordered comparison is skipped where the stripe's keep it.
7899    ///
7900    /// This is the shape of ClickBench 24. Each part covers a narrow stretch of the column and the
7901    /// stripe covers all sixty four of them at once, so a comparison that lands inside the stripe
7902    /// rules out none of it and rules out all but a few parts.
7903    #[test]
7904    fn a_part_is_skipped_when_its_own_bounds_rule_out_a_comparison_the_stripe_keeps() {
7905        let path = path("part-range-skip");
7906        let mut writer =
7907            Writer::create(&path, "hits", vec![Field::required("at", LogicalType::BigInt)])
7908                .expect("new file");
7909        let parts = STRIPE_PARTS + 3;
7910        let per_part = 128;
7911        for part in 0..parts {
7912            // Scattered inside the part's own band rather than a run, because a run of
7913            // consecutive numbers encodes to a stride of a few bytes and then the page of ranges
7914            // costs more than reading the column it indexes, which is the case the writer declines.
7915            let held: Vec<Value> = (0..per_part)
7916                .map(|row| {
7917                    Value::BigInt((part * 1_000) as i64 + (scattered(row as i64).rem_euclid(900)))
7918                })
7919                .collect();
7920            let chunk =
7921                Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
7922                    .expect("one column");
7923            writer.append(&chunk).expect("one part");
7924        }
7925        writer.finish().expect("commit");
7926
7927        let reader = Reader::open(&path).expect("reopen from disk");
7928        let under = [Probe { column: 0, op: Op::Less, value: Bound::Int(3_000) }];
7929        let kept: Vec<usize> = (0..parts).filter(|&part| !reader.skips(part, &under)).collect();
7930        assert_eq!(kept, vec![0, 1, 2], "only the three parts that start under three thousand");
7931        // The same question asked of the stripe alone, which is what this replaces.
7932        assert!(!reader.stripe_skips(0, &under), "the stripe reaches from zero and keeps itself");
7933        fs::remove_file(path).expect("remove scratch file");
7934    }
7935
7936    /// The other half of the same page. A part whose own bounds put every row of it inside the
7937    /// filter is waved through, so the comparison never runs on it, where the stripe's bounds reach
7938    /// across every part and can prove nothing.
7939    #[test]
7940    fn a_part_is_waved_through_when_its_own_bounds_pass_a_comparison_the_stripe_cannot() {
7941        let path = path("part-range-certain");
7942        let mut writer =
7943            Writer::create(&path, "hits", vec![Field::required("at", LogicalType::BigInt)])
7944                .expect("new file");
7945        let parts = STRIPE_PARTS + 3;
7946        let per_part = 128;
7947        for part in 0..parts {
7948            let held: Vec<Value> = (0..per_part)
7949                .map(|row| {
7950                    Value::BigInt((part * 1_000) as i64 + (scattered(row as i64).rem_euclid(900)))
7951                })
7952                .collect();
7953            let chunk =
7954                Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
7955                    .expect("one column");
7956            writer.append(&chunk).expect("one part");
7957        }
7958        writer.finish().expect("commit");
7959
7960        let reader = Reader::open(&path).expect("reopen from disk");
7961        let under = [Probe { column: 0, op: Op::Less, value: Bound::Int(3_000) }];
7962        let waved: Vec<usize> = (0..parts).filter(|&part| reader.certain(part, &under)).collect();
7963        assert_eq!(waved, vec![0, 1, 2], "the three parts that end under three thousand");
7964        // The first stripe reaches from zero to past sixty thousand, so it straddles three thousand
7965        // and settles nothing either way. The three yeses above are the parts' own ends talking.
7966        assert!(!reader.stripe_skips(0, &under), "the stripe straddles the comparison");
7967        fs::remove_file(path).expect("remove scratch file");
7968    }
7969
7970    /// The page is worth its bytes on a column with parts to tell apart and is not written on one
7971    /// that has a single part, where the stripe bounds already are the part's.
7972    #[test]
7973    fn a_stripe_of_one_part_writes_no_range_page_and_a_stripe_of_many_does() {
7974        for (parts, wanted) in [(1_usize, false), (STRIPE_PARTS, true)] {
7975            let path = path("part-range-page");
7976            let mut writer =
7977                Writer::create(&path, "hits", vec![Field::required("at", LogicalType::BigInt)])
7978                    .expect("new file");
7979            for part in 0..parts {
7980                let held: Vec<Value> = (0..128)
7981                    .map(|row| {
7982                        Value::BigInt((part * 1_000) as i64 + scattered(row as i64).rem_euclid(900))
7983                    })
7984                    .collect();
7985                let chunk = Chunk::new(vec![
7986                    Vector::from_values(LogicalType::BigInt, &held).expect("numbers"),
7987                ])
7988                .expect("one column");
7989                writer.append(&chunk).expect("one part");
7990            }
7991            writer.finish().expect("commit");
7992            let reader = Reader::open(&path).expect("reopen from disk");
7993            let bytes = reader.layout().columns[0].part_ranges;
7994            assert_eq!(bytes > 0, wanted, "{parts} parts wrote {bytes} bytes of ranges");
7995            fs::remove_file(path).expect("remove scratch file");
7996        }
7997    }
7998
7999    /// A cut down string end is still an end on the side it was, which is the only thing that keeps
8000    /// a shortened bound from turning a skip into a wrong answer.
8001    #[test]
8002    fn a_string_end_that_is_cut_down_still_covers_the_value_it_came_from() {
8003        let long = vec![b'a'; PART_BOUND_BYTES * 2];
8004        let low = shortened(Some(Bound::Bytes(long.clone())), false).expect("a low end");
8005        let high = shortened(Some(Bound::Bytes(long.clone())), true).expect("a high end");
8006        let Bound::Bytes(low) = low else { panic!("a string stays a string") };
8007        let Bound::Bytes(high) = high else { panic!("a string stays a string") };
8008        assert!(low.len() <= PART_BOUND_BYTES && high.len() <= PART_BOUND_BYTES);
8009        assert!(low.as_slice() <= long.as_slice(), "the low end is at or under the value");
8010        assert!(high.as_slice() >= long.as_slice(), "the high end is at or over the value");
8011    }
8012
8013    /// A string of nothing but the largest byte has no prefix that can be stepped up, so the high
8014    /// end is given up rather than claimed too small. No end keeps the part, which is always safe.
8015    #[test]
8016    fn a_string_end_with_no_room_to_step_up_gives_up_the_bound() {
8017        let long = vec![u8::MAX; PART_BOUND_BYTES * 2];
8018        assert_eq!(shortened(Some(Bound::Bytes(long.clone())), true), None);
8019        let low = shortened(Some(Bound::Bytes(long)), false).expect("a low end is still a prefix");
8020        assert_eq!(low, Bound::Bytes(vec![u8::MAX; PART_BOUND_BYTES]));
8021    }
8022
8023    /// What a column is stored as, asked of two files holding the same rows in a different order.
8024    ///
8025    /// This is the question the report exists to answer and it is the one the directory cannot. The
8026    /// two files have the same rows, the same schema and the same number of parts, and the column
8027    /// comes out four times smaller in one of them, because ascending keys delta encode to a few
8028    /// bits a row and shuffled ones do not. Nothing about the file's shape says so. The page header
8029    /// says so, and reading it is what this does.
8030    ///
8031    /// It is q18 on TPC-H in miniature: clustering lineitem by ship date leaves `l_orderkey`
8032    /// ascending inside a partition but sparse, its deltas go from six bits to twelve, and the scan
8033    /// pays for the wider ones.
8034    #[test]
8035    fn what_a_column_is_stored_as_follows_the_order_the_rows_were_written_in() {
8036        let parts = 4;
8037        let per_part = 1024;
8038        let rows = parts * per_part;
8039        let written = |name: &str, keys: &[i64]| {
8040            let path = path(name);
8041            let fields = vec![Field::required("key", LogicalType::BigInt)];
8042            let mut writer = Writer::create(&path, "keys", fields).expect("new file");
8043            for part in 0..parts {
8044                let values: Vec<Value> = keys[part * per_part..(part + 1) * per_part]
8045                    .iter()
8046                    .map(|key| Value::BigInt(*key))
8047                    .collect();
8048                let chunk = Chunk::new(vec![
8049                    Vector::from_values(LogicalType::BigInt, &values).expect("numbers"),
8050                ])
8051                .expect("one column");
8052                writer.append(&chunk).expect("one part");
8053            }
8054            writer.finish().expect("commit");
8055            path
8056        };
8057        // Ascending with a small irregular step, which is what a key column in arrival order looks
8058        // like: an order has one to seven line items, so the key repeats and then moves on by one.
8059        let climbing = |step: &dyn Fn(usize) -> i64| {
8060            let mut key = 0;
8061            (0..rows)
8062                .map(|row| {
8063                    key += step(row);
8064                    key
8065                })
8066                .collect::<Vec<i64>>()
8067        };
8068        let ascending = climbing(&|row| (row % 3) as i64);
8069        // The same rows in the same direction over a range a thousand times wider, which is what a
8070        // partition of a clustered table holds: still ascending, and far enough apart that the
8071        // deltas no longer fit in a handful of bits.
8072        let sparse = climbing(&|row| ((row * 2_654_435_761) % 4096) as i64);
8073        let near_path = written("stored-near", &ascending);
8074        let far_path = written("stored-far", &sparse);
8075
8076        let one = Reader::open(&near_path).expect("reopen from disk");
8077        let other = Reader::open(&far_path).expect("reopen from disk");
8078        let near = one.stored(0).expect("the column is stored");
8079        let far = other.stored(0).expect("the column is stored");
8080        assert_eq!(near.len(), parts, "one row per part");
8081        assert_eq!(far.len(), parts);
8082        // The bytes are the same bytes the directory totals, which is the check that this is
8083        // reading the pages the file really holds rather than some other pages.
8084        let total = |stored: &[StoredPart]| stored.iter().map(|part| part.bytes).sum::<u64>();
8085        assert_eq!(total(&near), one.layout().columns[0].pages);
8086        assert_eq!(total(&far), other.layout().columns[0].pages);
8087        assert!(
8088            total(&near) * 2 < total(&far),
8089            "the sparse keys cost more, {} against {}",
8090            total(&far),
8091            total(&near)
8092        );
8093        // Every part accounted for, in order, with the row it starts at following the one before.
8094        for (at, part) in near.iter().enumerate() {
8095            assert_eq!(part.part, at);
8096            assert_eq!(part.row, at * per_part);
8097            assert_eq!(part.rows, per_part);
8098            let held = &ascending[at * per_part..(at + 1) * per_part];
8099            assert_eq!(part.low, Some(Value::BigInt(held[0])));
8100            assert_eq!(part.high, Some(Value::BigInt(held[per_part - 1])));
8101            assert_eq!(part.nulls, Some(0));
8102        }
8103        // And the encoding is a line of text that names what the encoder chose, which is the whole
8104        // point. Both are a cascade over deltas and the widths inside them are what differ.
8105        assert!(near[0].encoding.contains("DELTA"), "{}", near[0].encoding);
8106        assert!(far[0].encoding.contains("DELTA"), "{}", far[0].encoding);
8107        assert_ne!(near[0].encoding, far[0].encoding);
8108        fs::remove_file(near_path).expect("remove scratch file");
8109        fs::remove_file(far_path).expect("remove scratch file");
8110    }
8111
8112    /// A sieve bigger than the part it indexes is not written, and one smaller than it still is.
8113    ///
8114    /// Both columns hold values spread over the whole of `BIGINT`, so neither gets a bitmap and both
8115    /// reach the filter. They differ in what the part costs to read. `spread` is a thousand distinct
8116    /// numbers and packs to eight kilobytes, so a filter of about thirteen hundred bytes is a good
8117    /// trade. `repeated` is the same thousand rows over four numbers in runs and encodes to
8118    /// almost nothing, but the filter is sized for the rows rather than the values it turns out to
8119    /// hold, so it comes out larger than the data. Reading it to decide whether to read the part spends more than
8120    /// the part, every time, and that is the case this drops.
8121    #[test]
8122    fn a_sieve_larger_than_the_part_it_indexes_is_not_written() {
8123        let path = path("sieve-pays");
8124        let fields = vec![
8125            Field::required("spread", LogicalType::BigInt),
8126            Field::required("repeated", LogicalType::BigInt),
8127        ];
8128        let mut writer = Writer::create(&path, "hits", fields).expect("new file");
8129        let parts = 3;
8130        let per_part = 1024;
8131        for part in 0..parts {
8132            let base = (part * per_part) as i64;
8133            let spread: Vec<Value> =
8134                (0..per_part).map(|row| Value::BigInt(scattered(base + row as i64))).collect();
8135            let repeated: Vec<Value> =
8136                (0..per_part).map(|row| Value::BigInt(scattered((row / 256) as i64))).collect();
8137            let chunk = Chunk::new(vec![
8138                Vector::from_values(LogicalType::BigInt, &spread).expect("numbers"),
8139                Vector::from_values(LogicalType::BigInt, &repeated).expect("numbers"),
8140            ])
8141            .expect("two columns");
8142            writer.append(&chunk).expect("one part");
8143        }
8144        writer.finish().expect("commit");
8145
8146        let reader = Reader::open(&path).expect("reopen from disk");
8147        let layout = reader.layout();
8148        let spread = &layout.columns[0];
8149        let repeated = &layout.columns[1];
8150        assert!(spread.sieves > 0, "a column whose parts are worth a filter keeps one");
8151        assert_eq!(
8152            repeated.sieves, 0,
8153            "a column whose filter costs more than its parts keeps none"
8154        );
8155        // Per part this is the rule itself, so it holds over the column as well: a part without a
8156        // sieve adds to one side of this and to nothing on the other.
8157        for column in &layout.columns {
8158            assert!(
8159                column.sieves < column.pages,
8160                "{} spends {} on sieves over {} of data",
8161                column.name,
8162                column.sieves,
8163                column.pages
8164            );
8165        }
8166        // The filter that was kept still does what it is for.
8167        let absent = [Probe {
8168            column: 0,
8169            op: Op::Equal,
8170            value: Bound::Int(i128::from(scattered((parts * per_part) as i64 + 1))),
8171        }];
8172        assert!((0..parts).all(|part| reader.skips(part, &absent)), "no part holds it");
8173        fs::remove_file(path).expect("remove scratch file");
8174    }
8175
8176    /// A damaged sieve page is a part that gets read, not a query that fails.
8177    ///
8178    /// A sieve is an index over rows that are still there and still correct, so losing one costs
8179    /// time and costs no answers. That is the opposite of the membership index beside it, which is
8180    /// the only thing standing between a string page and a wrong answer.
8181    #[test]
8182    fn a_damaged_sieve_page_is_read_through_rather_than_refused() {
8183        let path = path("sieve-damaged");
8184        let mut writer =
8185            Writer::create(&path, "hits", vec![Field::required("id", LogicalType::BigInt)])
8186                .expect("new file");
8187        let rows = 128;
8188        let held: Vec<Value> = (0..rows).map(|row| Value::BigInt(scattered(row))).collect();
8189        let chunk =
8190            Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
8191                .expect("one column");
8192        writer.append(&chunk).expect("one part");
8193        writer.finish().expect("commit");
8194
8195        let page =
8196            Reader::open(&path).expect("reopen").table.stripes[0].sieves[0].expect("a sieve page");
8197        let mut file = OpenOptions::new().write(true).open(&path).expect("open the sieve page");
8198        file.seek(SeekFrom::Start(page.offset + u64::from(page.length) - 1)).expect("seek");
8199        file.write_all(&[0xff]).expect("damage one byte");
8200        drop(file);
8201
8202        let reader = Reader::open(&path).expect("reopen the damaged file");
8203        let absent =
8204            [Probe { column: 0, op: Op::Equal, value: Bound::Int(i128::from(scattered(99))) }];
8205        assert!(!reader.skips(0, &absent), "a sieve that cannot be read skips nothing");
8206        assert_eq!(
8207            reader.read(0, &[0]).expect("the rows are untouched").len(),
8208            usize::try_from(rows).expect("a small count")
8209        );
8210        fs::remove_file(path).expect("remove scratch file");
8211    }
8212
8213    /// Eight workers over one stripe read it once between them.
8214    ///
8215    /// This is the shape a scan actually has. Parts are handed out in order, so every worker on a
8216    /// column crosses into a stripe within a few parts of the others, and before [`Reader::held`]
8217    /// started sharing the read every one of them read the whole page. On the full ClickBench file
8218    /// that was a `MIN(EventDate), MAX(EventDate)` moving 3.2 GB off the disk to look at 400 MB of
8219    /// column, which is most of what a first touch costs.
8220    ///
8221    /// The workers that lose the race still answer, out of the part reads they do instead, which is
8222    /// what the values below are checking.
8223    #[test]
8224    fn workers_that_want_the_same_stripe_read_it_once() {
8225        let path = path("single-flight");
8226        let mut writer =
8227            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
8228                .expect("new file");
8229        for part in 0..STRIPE_PARTS {
8230            let id = part as i32;
8231            let chunk = Chunk::new(vec![
8232                Vector::from_values(
8233                    LogicalType::Integer,
8234                    &[Value::Integer(id), Value::Integer(-id)],
8235                )
8236                .expect("integers"),
8237            ])
8238            .expect("matching rows");
8239            writer.append(&chunk).expect("one part");
8240        }
8241        writer.finish().expect("commit");
8242
8243        let reader = Reader::open(&path).expect("reopen from disk");
8244        assert_eq!(reader.table().stripes().len(), 1, "one stripe is the point of the test");
8245        let barrier = std::sync::Barrier::new(8);
8246        std::thread::scope(|scope| {
8247            for worker in 0..8 {
8248                let reader = &reader;
8249                let barrier = &barrier;
8250                scope.spawn(move || {
8251                    barrier.wait();
8252                    for part in (worker..STRIPE_PARTS).step_by(8) {
8253                        let chunk = reader.read(part, &[0]).expect("a whole page read");
8254                        assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
8255                        assert_eq!(chunk.value_at(1, 0), Value::Integer(-(part as i32)));
8256                    }
8257                });
8258            }
8259        });
8260        assert_eq!(reader.pages.load(Atomic::Relaxed), 1, "one stripe, one page read, whoever won");
8261        fs::remove_file(path).expect("remove scratch file");
8262    }
8263
8264    /// Opening a file reads the header and the directory, and nothing that depends on the rows.
8265    ///
8266    /// `spec/stats/04-in-memory.md` section 4.2. There are no statistics in the file yet, so this
8267    /// holds today by not having anything to load, and that is exactly why it is worth pinning now.
8268    /// The change that breaks it is the reasonable looking one: summaries are a few hundred bytes,
8269    /// the next query will want them, so read them on the way past. A process that opened the
8270    /// database to run one trivial query pays for all of it and gets nothing.
8271    ///
8272    /// Two files of the same shape and a thousand times the rows in one of them, opened, and the
8273    /// two openings cost the same. The stripe count is held equal so that the directory is the same
8274    /// size in both, which leaves the rows as the only thing that changed. Anything read out of the
8275    /// data would show up here.
8276    #[test]
8277    fn opening_costs_the_same_over_a_thousand_times_the_rows() {
8278        let opened = |label: &str, rows_per_part: i32| {
8279            let path = path(label);
8280            let mut writer =
8281                Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
8282                    .expect("new file");
8283            for part in 0..STRIPE_PARTS * 3 {
8284                // Scrambled rather than sequential, so that the fat file is actually fatter. A run
8285                // of consecutive integers encodes to almost nothing and would leave the two files
8286                // the same size, which would make this test pass for the wrong reason.
8287                let values = (0..rows_per_part)
8288                    .map(|row| {
8289                        Value::Integer((part as i32 * rows_per_part + row).wrapping_mul(2_654_435))
8290                    })
8291                    .collect::<Vec<_>>();
8292                let chunk = Chunk::new(vec![
8293                    Vector::from_values(LogicalType::Integer, &values).expect("integers"),
8294                ])
8295                .expect("matching rows");
8296                writer.append(&chunk).expect("one part");
8297            }
8298            writer.finish().expect("commit");
8299            let reader = Reader::open(&path).expect("reopen from disk");
8300            let size = fs::metadata(&path).expect("the file is there").len();
8301            let out = (reader.reads(), reader.table().stripes().len(), size);
8302            fs::remove_file(path).expect("remove scratch file");
8303            out
8304        };
8305
8306        let (thin, thin_stripes, thin_size) = opened("open-thin", 1);
8307        let (fat, fat_stripes, fat_size) = opened("open-fat", 1000);
8308        assert_eq!(
8309            thin_stripes, fat_stripes,
8310            "the same stripe count is what makes this a fair ask"
8311        );
8312        assert!(
8313            fat_size > thin_size * 50,
8314            "the fat file has to actually be larger, and it is {fat_size} against {thin_size}"
8315        );
8316
8317        assert_eq!(thin.opening.reads, fat.opening.reads, "the same reads either way");
8318        assert_eq!(thin.pages, 0, "opening read a page");
8319        assert_eq!(fat.pages, 0, "opening read a page");
8320        assert_eq!(thin.indexes, 0, "opening read an index");
8321        assert_eq!(fat.indexes, 0, "opening read an index");
8322        // Not exactly equal, because a directory holds offsets and a larger file has larger ones,
8323        // and a handful of bytes of varint is not somebody loading statistics. A factor is.
8324        assert!(
8325            fat.opening.bytes < thin.opening.bytes * 2,
8326            "opening the thin file read {} bytes and the fat one read {}",
8327            thin.opening.bytes,
8328            fat.opening.bytes
8329        );
8330    }
8331
8332    /// The reads a file costs to open are fixed by its shape and not by what ran before.
8333    ///
8334    /// `spec/stats/04-in-memory.md` section 4.3, which is the rule that keeps a plan reproducible:
8335    /// the plan is a function of the data, the generation and the settings, and never of what
8336    /// happened to be in cache. Opening the same file twice in the same process has to cost the
8337    /// same, because a second open that read less would be an open that was about to plan
8338    /// differently.
8339    #[test]
8340    fn two_opens_of_one_file_cost_the_same_and_the_second_is_not_cheaper() {
8341        let path = path("open-twice");
8342        let mut writer =
8343            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
8344                .expect("new file");
8345        for part in 0..STRIPE_PARTS * 3 {
8346            let chunk = Chunk::new(vec![
8347                Vector::from_values(LogicalType::Integer, &[Value::Integer(part as i32)])
8348                    .expect("integers"),
8349            ])
8350            .expect("matching rows");
8351            writer.append(&chunk).expect("one part");
8352        }
8353        writer.finish().expect("commit");
8354
8355        let first = Reader::open(&path).expect("open");
8356        // A whole scan in between, so the operating system's page cache is as warm as it gets and
8357        // anything that consulted it would show up in the second open.
8358        for part in 0..first.parts() {
8359            first.read(part, &[0]).expect("a part");
8360        }
8361        assert!(first.reads().pages > 0, "the scan has to have read something");
8362        let second = Reader::open(&path).expect("open again");
8363
8364        assert_eq!(first.reads().opening, second.reads().opening);
8365        assert_eq!(
8366            second.reads().pages,
8367            0,
8368            "the second open read a page off the back of the first"
8369        );
8370        assert_eq!(second.reads().indexes, 0, "the second open read an index it inherited");
8371        fs::remove_file(path).expect("remove scratch file");
8372    }
8373
8374    /// A scan reads a stripe's index once for the whole scan, not once per part that misses.
8375    ///
8376    /// The page cache holds four stripes and an index used to ride inside it, so a table with more
8377    /// stripes than that read the index again every time a stripe came back around. The index is a
8378    /// few hundred bytes and the page is a quarter of a megabyte, which is why they are now under
8379    /// different budgets. This is the test that keeps them there, since the saving is small enough
8380    /// that nothing in a benchmark would notice it going away again.
8381    #[test]
8382    fn an_index_is_read_once_per_stripe_however_often_the_page_is_evicted() {
8383        let path = path("index-cache");
8384        let mut writer =
8385            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
8386                .expect("new file");
8387        let parts = STRIPE_PARTS * (CACHED_STRIPES_PER_COLUMN + 2);
8388        for part in 0..parts {
8389            let id = part as i32;
8390            let chunk = Chunk::new(vec![
8391                Vector::from_values(LogicalType::Integer, &[Value::Integer(id)]).expect("integers"),
8392            ])
8393            .expect("matching rows");
8394            writer.append(&chunk).expect("one part");
8395        }
8396        writer.finish().expect("commit");
8397
8398        let reader = Reader::open(&path).expect("reopen from disk");
8399        let stripes = reader.table().stripes().len();
8400        assert!(stripes > CACHED_STRIPES_PER_COLUMN, "the page cache has to be too small for this");
8401        // Twice over, so that the second pass finds every page evicted and every index kept.
8402        for _ in 0..2 {
8403            for part in 0..parts {
8404                let chunk = reader.read(part, &[0]).expect("a part");
8405                assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
8406            }
8407        }
8408        assert_eq!(reader.indexes.load(Atomic::Relaxed), stripes, "one index read per stripe");
8409        assert!(
8410            reader.pages.load(Atomic::Relaxed) > stripes,
8411            "the pages are the ones that get read again, which is what makes the index count mean \
8412             something"
8413        );
8414        fs::remove_file(path).expect("remove scratch file");
8415    }
8416
8417    /// A worker per stripe reads its stripe once, once the cache has been told how many there are.
8418    ///
8419    /// This is the shape a scan has when it hands out a whole stripe per morsel rather than a part.
8420    /// Nobody races for a page any more, but every worker holds a different one for the length of a
8421    /// stripe, so a cache that keeps four pages while eight workers are in eight stripes evicts
8422    /// every one of them before its owner has finished with it, and the owner reads a quarter of a
8423    /// megabyte again for the next part. The barrier is what makes that certain rather than likely:
8424    /// without it a worker can run a whole stripe before the next one starts and never collide.
8425    #[test]
8426    fn a_worker_per_stripe_reads_its_page_once_when_the_cache_was_told_to_expect_it() {
8427        let workers = CACHED_STRIPES_PER_COLUMN + 4;
8428        let path = path("stripe-per-worker");
8429        let mut writer =
8430            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
8431                .expect("new file");
8432        for part in 0..STRIPE_PARTS * workers {
8433            let chunk = Chunk::new(vec![
8434                Vector::from_values(LogicalType::Integer, &[Value::Integer(part as i32)])
8435                    .expect("integers"),
8436            ])
8437            .expect("matching rows");
8438            writer.append(&chunk).expect("one part");
8439        }
8440        writer.finish().expect("commit");
8441
8442        let read = |told: bool| {
8443            let reader = Reader::open(&path).expect("reopen from disk");
8444            assert_eq!(reader.table().stripes().len(), workers, "a stripe per worker");
8445            if told {
8446                reader.keep_stripes(workers);
8447            }
8448            let barrier = std::sync::Barrier::new(workers);
8449            std::thread::scope(|scope| {
8450                for (worker, run) in reader.stripe_parts().into_iter().enumerate() {
8451                    let reader = &reader;
8452                    let barrier = &barrier;
8453                    scope.spawn(move || {
8454                        for part in run {
8455                            barrier.wait();
8456                            let chunk = reader.read(part, &[0]).expect("a part of my own stripe");
8457                            assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
8458                        }
8459                        assert!(worker < workers);
8460                    });
8461                }
8462            });
8463            reader.pages.load(Atomic::Relaxed)
8464        };
8465
8466        assert_eq!(read(true), workers, "one page read per stripe and no more");
8467        assert!(read(false) > workers, "a cache that small is read again on every part");
8468        fs::remove_file(path).expect("remove scratch file");
8469    }
8470
8471    /// A damaged index page is caught before anything decodes a part out of it.
8472    ///
8473    /// The index is the one structure a reader trusts to find bytes with, so it carries a checksum
8474    /// per column section rather than one for the page, and this is what says that check runs.
8475    #[test]
8476    fn a_damaged_index_page_is_an_error() {
8477        let path = path("damaged-index");
8478        let mut writer =
8479            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
8480                .expect("new file");
8481        writer.append(&sample_ids()).expect("first part");
8482        writer.append(&sample_ids()).expect("second part");
8483        writer.finish().expect("commit");
8484
8485        let reader = Reader::open(&path).expect("valid directory");
8486        let index = reader.table.stripes[0].index;
8487        let mut byte = [0; 1];
8488        read_at(&reader.file, index.offset, &mut byte).expect("the first part length");
8489        let mut file = OpenOptions::new().write(true).open(&path).expect("open index page");
8490        file.seek(SeekFrom::Start(index.offset)).expect("index start");
8491        file.write_all(&[!byte[0]]).expect("damage the first part length");
8492        let error = reader.read(1, &[0]).expect_err("a damaged index must not be used");
8493        assert!(error.message().contains("index page section checksum differs"), "{error}");
8494        fs::remove_file(path).expect("remove scratch file");
8495    }
8496
8497    /// Every integer width the format knows about, written and read back.
8498    ///
8499    /// The unsigned ones are the reason ClickBench can be stored at all: `hits` types `EventDate`
8500    /// as `USMALLINT`, and one unsupported column meant the whole table was refused. The extremes
8501    /// are in here on purpose, because a width that round trips through the wrong signedness only
8502    /// goes wrong at the end of its range.
8503    #[test]
8504    fn every_integer_width_round_trips_through_a_page() {
8505        let path = path("integer-widths");
8506        let columns = [
8507            (LogicalType::TinyInt, vec![Value::TinyInt(i8::MIN), Value::TinyInt(i8::MAX)]),
8508            (LogicalType::UTinyInt, vec![Value::UTinyInt(0), Value::UTinyInt(u8::MAX)]),
8509            (LogicalType::SmallInt, vec![Value::SmallInt(i16::MIN), Value::SmallInt(i16::MAX)]),
8510            (LogicalType::USmallInt, vec![Value::USmallInt(0), Value::USmallInt(u16::MAX)]),
8511            (LogicalType::Integer, vec![Value::Integer(i32::MIN), Value::Integer(i32::MAX)]),
8512            (LogicalType::UInteger, vec![Value::UInteger(0), Value::UInteger(u32::MAX)]),
8513            (LogicalType::BigInt, vec![Value::BigInt(i64::MIN), Value::BigInt(i64::MAX)]),
8514            (LogicalType::UBigInt, vec![Value::UBigInt(0), Value::UBigInt(u64::MAX)]),
8515        ];
8516        let fields = columns
8517            .iter()
8518            .enumerate()
8519            .map(|(at, (ty, _))| Field::required(format!("c{at}"), ty.clone()))
8520            .collect::<Vec<_>>();
8521        let vectors = columns
8522            .iter()
8523            .map(|(ty, values)| Vector::from_values(ty.clone(), values).expect("a vector"))
8524            .collect::<Vec<_>>();
8525        let mut writer = Writer::create(&path, "widths", fields).expect("new file");
8526        writer.append(&Chunk::new(vectors).expect("matching rows")).expect("one stripe");
8527        writer.finish().expect("commit");
8528
8529        let reader = Reader::open(&path).expect("reopen from disk");
8530        let wanted = (0..columns.len()).collect::<Vec<_>>();
8531        let read = reader.read(0, &wanted).expect("every column");
8532        assert_eq!(read.len(), 2);
8533        // row at a time: each column has its own type and its own pair of extremes.
8534        for (at, (ty, values)) in columns.iter().enumerate() {
8535            assert_eq!(read.value_at(0, at), values[0], "the low end of {ty}");
8536            assert_eq!(read.value_at(1, at), values[1], "the high end of {ty}");
8537        }
8538        fs::remove_file(path).expect("remove scratch file");
8539    }
8540
8541    /// The rest of the fixed width types, and the byte strings, written and read back.
8542    ///
8543    /// The extremes again, and for a float that means more than the ends of the range. Negative
8544    /// zero and a NaN are the two values that go through an encoder unnoticed and come back
8545    /// different, so they are here on purpose, and the NaN is compared by its bits rather than by
8546    /// `==`, which a NaN fails against itself.
8547    ///
8548    /// A blob is here beside them because it is the same round trip asked of bytes that are not
8549    /// text. The value in it is not UTF-8, so a path that reads a payload as a string on the way
8550    /// past turns this test red rather than turning a user's column into nulls.
8551    #[test]
8552    fn every_other_type_the_format_knows_round_trips_through_a_page() {
8553        let path = path("other-types");
8554        let columns = [
8555            (LogicalType::Float, vec![Value::Float(f32::MIN), Value::Float(-0.0)]),
8556            (LogicalType::Double, vec![Value::Double(f64::MIN), Value::Double(f64::MAX)]),
8557            (LogicalType::HugeInt, vec![Value::HugeInt(i128::MIN), Value::HugeInt(i128::MAX)]),
8558            (LogicalType::UHugeInt, vec![Value::UHugeInt(0), Value::UHugeInt(u128::MAX)]),
8559            (LogicalType::Time, vec![Value::Time(0), Value::Time(86_399_999_999)]),
8560            (LogicalType::TimeTz, vec![Value::TimeTz(-50_400_000_000), Value::TimeTz(0)]),
8561            (
8562                LogicalType::TimestampTz,
8563                vec![Value::TimestampTz(i64::MIN + 1), Value::TimestampTz(i64::MAX)],
8564            ),
8565            (
8566                LogicalType::Interval,
8567                vec![
8568                    Value::Interval { months: i32::MIN, days: i32::MAX, micros: i64::MIN },
8569                    Value::Interval { months: 13, days: -1, micros: 1 },
8570                ],
8571            ),
8572            (
8573                LogicalType::Blob,
8574                vec![Value::Blob(vec![0, 0xff, 0x80, 0xfe]), Value::Blob(Vec::new())],
8575            ),
8576        ];
8577        let fields = columns
8578            .iter()
8579            .enumerate()
8580            .map(|(at, (ty, _))| Field::required(format!("c{at}"), ty.clone()))
8581            .collect::<Vec<_>>();
8582        let vectors = columns
8583            .iter()
8584            .map(|(ty, values)| Vector::from_values(ty.clone(), values).expect("a vector"))
8585            .collect::<Vec<_>>();
8586        let mut writer = Writer::create(&path, "others", fields).expect("new file");
8587        writer.append(&Chunk::new(vectors).expect("matching rows")).expect("one stripe");
8588        writer.finish().expect("commit");
8589
8590        let reader = Reader::open(&path).expect("reopen from disk");
8591        let wanted = (0..columns.len()).collect::<Vec<_>>();
8592        let read = reader.read(0, &wanted).expect("every column");
8593        assert_eq!(read.len(), 2);
8594        for (at, (ty, values)) in columns.iter().enumerate() {
8595            assert_eq!(read.value_at(0, at), values[0], "the low end of {ty}");
8596            assert_eq!(read.value_at(1, at), values[1], "the high end of {ty}");
8597        }
8598        // A float keeps its sign through a zero, which `==` says nothing about because negative
8599        // zero and zero compare equal.
8600        let Value::Float(zero) = read.value_at(1, 0) else { panic!("a float stays a float") };
8601        assert!(zero.is_sign_negative(), "a negative zero came back as {zero}");
8602
8603        fs::remove_file(path).expect("remove scratch file");
8604    }
8605
8606    /// A NaN is still a NaN after a trip through a page.
8607    ///
8608    /// Apart from the other floats because it cannot be asserted the same way. A NaN is not equal
8609    /// to itself, so a comparison against the value that was written passes for every NaN and for
8610    /// nothing else, which is the one assertion that would not catch a page that lost it.
8611    #[test]
8612    fn a_nan_survives_being_written_down() {
8613        let path = path("nan");
8614        let nan = Vector::from_values(LogicalType::Double, &[Value::Double(f64::NAN)])
8615            .expect("a NaN vector");
8616        let mut writer =
8617            Writer::create(&path, "nan", vec![Field::required("d", LogicalType::Double)])
8618                .expect("new file");
8619        writer.append(&Chunk::new(vec![nan]).expect("one column")).expect("one stripe");
8620        writer.finish().expect("commit");
8621        let read = Reader::open(&path).expect("reopen").read(0, &[0]).expect("the column");
8622        let Value::Double(back) = read.value_at(0, 0) else { panic!("a double stays a double") };
8623        assert!(back.is_nan(), "a NaN came back as {back}");
8624        fs::remove_file(path).expect("remove scratch file");
8625    }
8626
8627    /// A uuid and a bit string, which have no `Value` arm of their own and are checked as bits.
8628    ///
8629    /// A uuid is the 128 bit lane and a bit string is bytes, and neither of them reads back as
8630    /// anything in `Value` today, so asking for a value here would compare two nulls and pass
8631    /// whatever the file held. The data underneath is what the storage promise is about, so that is
8632    /// what this reads.
8633    #[test]
8634    fn a_uuid_and_a_bit_string_come_back_as_the_bits_that_went_in() {
8635        let path = path("uuid-and-bit");
8636        let uuids = vec![0_i128, i128::MIN, -1];
8637        let mut bits = StringColumn::new();
8638        for value in [&b"\x02\xff"[..], &b""[..], &b"\x00\x01\x02\x03\x04\x05"[..]] {
8639            bits.push_bytes(value);
8640        }
8641        let expected = bits.clone();
8642        let fields =
8643            vec![Field::required("u", LogicalType::Uuid), Field::required("b", LogicalType::Bit)];
8644        let vectors = vec![
8645            Vector::flat(LogicalType::Uuid, Data::Int128(uuids.clone().into())).expect("uuids"),
8646            Vector::flat(LogicalType::Bit, Data::Varlen(bits)).expect("bit strings"),
8647        ];
8648        let mut writer = Writer::create(&path, "ids", fields).expect("new file");
8649        writer.append(&Chunk::new(vectors).expect("matching rows")).expect("one stripe");
8650        writer.finish().expect("commit");
8651
8652        let reader = Reader::open(&path).expect("reopen from disk");
8653        let read = reader.read(0, &[0, 1]).expect("both columns").flatten().expect("flat");
8654        let Some(Data::Int128(back)) = read.column(0).expect("the uuids").data() else {
8655            panic!("a uuid column is the 128 bit lane")
8656        };
8657        assert_eq!(back.as_slice(), uuids.as_slice());
8658        let Some(Data::Varlen(back)) = read.column(1).expect("the bits").data() else {
8659            panic!("a bit column is bytes")
8660        };
8661        for row in 0..expected.len() {
8662            assert_eq!(back.bytes(row), expected.bytes(row), "row {row} of the bit column");
8663        }
8664        fs::remove_file(path).expect("remove scratch file");
8665    }
8666
8667    #[test]
8668    fn numeric_frequency_candidates_keep_bounded_row_ordinals() {
8669        let path = path("frequency-ordinals");
8670        let mut writer =
8671            Writer::create(&path, "items", vec![Field::required("id", LogicalType::BigInt)])
8672                .expect("new file");
8673        let mut values = Vec::new();
8674        for leader in 0..10_i64 {
8675            values.extend(std::iter::repeat_n(leader, 100));
8676        }
8677        values.extend(1_000_i64..41_000);
8678        for part in values.chunks(1_024) {
8679            let vector = Vector::flat(LogicalType::BigInt, Data::Int64(part.to_vec().into()))
8680                .expect("big integers");
8681            writer.append(&Chunk::new(vec![vector]).expect("one column")).expect("one stripe");
8682        }
8683        writer.finish().expect("commit");
8684
8685        let reader = Reader::open(&path).expect("reopen from disk");
8686        let occurrences =
8687            reader.frequency_occurrences(0).expect("valid metadata").expect("bounded ordinals");
8688        assert!(occurrences.omitted_max < 100);
8689        assert!(occurrences.ordinals.len() <= FREQUENCY_ORDINALS);
8690        assert!(occurrences.ordinals.windows(2).all(|pair| pair[0] < pair[1]));
8691        assert_eq!(&occurrences.ordinals[..1_000], &(0_u64..1_000).collect::<Vec<_>>());
8692        fs::remove_file(path).expect("remove scratch file");
8693    }
8694
8695    /// The bug this is here for cost a 43 GB ClickBench table and an hour of reloading it. The
8696    /// format went from 11 to 12, every binary built after that said "magic or major version is
8697    /// unsupported" about the file, and there was no way to tell from the message whether the path
8698    /// was wrong, the file was truncated, or it was ours and simply older. The number this build
8699    /// wants is the whole answer and it was the one thing the message did not carry.
8700    #[test]
8701    fn a_file_from_another_format_says_which_format_it_is() {
8702        let older = path("older-format");
8703        let mut writer =
8704            Writer::create(&older, "items", vec![Field::new("id", LogicalType::Integer)])
8705                .expect("new file");
8706        let chunk = Chunk::new(vec![
8707            Vector::flat(LogicalType::Integer, Data::Int32(vec![1, 2, 3].into()))
8708                .expect("integers"),
8709        ])
8710        .expect("chunk");
8711        writer.append(&chunk).expect("page written");
8712        writer.finish().expect("commit");
8713
8714        // A format below the whole readable set, rather than `FORMAT - 1`, because the set has
8715        // more than one member now: format 22 is deliberately still readable, so the version that
8716        // has to be refused is the one under the oldest one accepted.
8717        let unreadable =
8718            READABLE.iter().copied().min().expect("at least one format is readable") - 1;
8719        let mut file = OpenOptions::new().write(true).open(&older).expect("open for the header");
8720        file.seek(SeekFrom::Start(8)).expect("the version follows the magic");
8721        file.write_all(&unreadable.to_le_bytes()).expect("write an older version");
8722        drop(file);
8723        let complaint = Reader::open(&older).expect_err("an older format is refused").to_string();
8724        assert!(complaint.contains(&format!("format {unreadable}")), "{complaint}");
8725        assert!(complaint.contains(&format!("format {FORMAT}")), "{complaint}");
8726
8727        let mut file = OpenOptions::new().write(true).open(&older).expect("open for the header");
8728        file.seek(SeekFrom::Start(0)).expect("the magic is first");
8729        file.write_all(b"NOTRUDB!").expect("write another engine's magic");
8730        drop(file);
8731        let complaint = Reader::open(&older).expect_err("a foreign file is refused").to_string();
8732        assert!(complaint.contains("magic"), "{complaint}");
8733        assert!(!complaint.contains("format"), "a version has nothing to do with it: {complaint}");
8734        fs::remove_file(older).expect("remove scratch file");
8735    }
8736
8737    #[test]
8738    fn an_unfinished_or_damaged_file_does_not_answer_with_partial_rows() {
8739        let unfinished = path("unfinished");
8740        let mut writer =
8741            Writer::create(&unfinished, "items", vec![Field::new("id", LogicalType::Integer)])
8742                .expect("new file");
8743        let chunk = Chunk::new(vec![
8744            Vector::flat(LogicalType::Integer, Data::Int32(vec![1, 2, 3].into()))
8745                .expect("integers"),
8746        ])
8747        .expect("chunk");
8748        writer.append(&chunk).expect("page written");
8749        drop(writer);
8750        assert!(Reader::open(&unfinished).is_err(), "no directory was committed");
8751        fs::remove_file(unfinished).expect("remove scratch file");
8752
8753        let damaged = path("damaged");
8754        let mut writer =
8755            Writer::create(&damaged, "items", vec![Field::new("id", LogicalType::Integer)])
8756                .expect("new file");
8757        writer.append(&chunk).expect("page written");
8758        writer.finish().expect("commit");
8759        let reader = Reader::open(&damaged).expect("valid directory");
8760        let mut file =
8761            OpenOptions::new().write(true).open(&damaged).expect("open for a damaged page");
8762        file.seek(SeekFrom::Start(HEADER + 1)).expect("inside first page");
8763        file.write_all(&[255]).expect("damage one byte");
8764        assert!(reader.read(0, &[0]).is_err(), "page checksum rejects corruption");
8765        fs::remove_file(damaged).expect("remove scratch file");
8766    }
8767
8768    #[test]
8769    fn damaged_lazy_dictionary_payload_is_an_error() {
8770        let path = path("damaged-dictionary");
8771        let mut writer = Writer::create(
8772            &path,
8773            "items",
8774            vec![
8775                Field::required("id", LogicalType::Integer),
8776                Field::new("text", LogicalType::Varchar),
8777            ],
8778        )
8779        .expect("new file");
8780        writer.append(&sample()).expect("stripe written");
8781        writer.finish().expect("commit");
8782
8783        let reader = Reader::open(&path).expect("valid directory");
8784        let dictionary = reader.table.dictionaries[1].expect("string dictionary page");
8785        // Read the count out of the page rather than writing it here, so that adding something
8786        // else to the index does not silently turn this into a test that damages the index.
8787        let mut header = [0; DICTIONARY_HEADER];
8788        read_at(&reader.file, dictionary.offset, &mut header).expect("dictionary header");
8789        let index_len = dictionary_index_len(&header);
8790        let rank_len = last_rank_end(&reader.file, dictionary.offset, &header);
8791        let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
8792        file.seek(SeekFrom::Start(dictionary.offset + index_len + rank_len))
8793            .expect("inside dictionary payload");
8794        file.write_all(&[255]).expect("damage dictionary payload");
8795
8796        let chunk = reader.read(0, &[1]).expect("code page and dictionary index remain valid");
8797        let error =
8798            chunk.validate_external().expect_err("payload corruption must reach the caller");
8799        assert!(error.message().contains("payload checksum differs"), "{error}");
8800        fs::remove_file(path).expect("remove scratch file");
8801    }
8802
8803    /// A column whose values are all different is written without a dictionary, and one whose
8804    /// values repeat keeps it.
8805    ///
8806    /// The two columns go in the same table and hold the same number of rows, so the only thing
8807    /// separating them is how much of the first stripe was a value it had not seen before. Both have
8808    /// to read back the values that were written, because the decision is about cost and nothing
8809    /// else. The file size is the other half of it: a column written without a dictionary goes
8810    /// through the string cascade instead, so dropping the dictionary must not turn into storing the
8811    /// column raw.
8812    #[test]
8813    fn a_column_of_all_different_values_is_written_without_a_dictionary() {
8814        let path = path("dictionary-decide");
8815        let rows = 20_000;
8816        // Long enough that storing it raw would show, and different in every row.
8817        let unique =
8818            |row: usize| format!("{row:09} a value that appears exactly once in the table");
8819        // The same values in the same shape, each one used forty times over.
8820        let repeated = |row: usize| unique(row / 40);
8821        let mut writer = Writer::create(
8822            &path,
8823            "items",
8824            vec![
8825                Field::required("unique", LogicalType::Varchar),
8826                Field::required("repeated", LogicalType::Varchar),
8827            ],
8828        )
8829        .expect("new file");
8830        for part in (0..rows).step_by(1_000) {
8831            let span = part..(part + 1_000).min(rows);
8832            let left = span.clone().map(|row| Value::Varchar(unique(row))).collect::<Vec<_>>();
8833            let right = span.map(|row| Value::Varchar(repeated(row))).collect::<Vec<_>>();
8834            writer
8835                .append(
8836                    &Chunk::new(vec![
8837                        Vector::from_values(LogicalType::Varchar, &left).expect("strings"),
8838                        Vector::from_values(LogicalType::Varchar, &right).expect("strings"),
8839                    ])
8840                    .expect("two columns"),
8841                )
8842                .expect("a part");
8843        }
8844        writer.finish().expect("commit");
8845
8846        let reader = Reader::open(&path).expect("reopen from disk");
8847        assert!(
8848            reader.table.dictionaries[0].is_none(),
8849            "a column with no repeats has nothing to say twice"
8850        );
8851        assert!(
8852            reader.table.dictionaries[1].is_some(),
8853            "a column whose values come round again keeps its dictionary"
8854        );
8855        let mut first = 0;
8856        for part in 0..reader.parts() {
8857            let chunk = reader.read(part, &[0, 1]).expect("a part");
8858            for row in 0..chunk.len() {
8859                assert_eq!(chunk.value_at(row, 0), Value::Varchar(unique(first + row)));
8860                assert_eq!(chunk.value_at(row, 1), Value::Varchar(repeated(first + row)));
8861            }
8862            first += chunk.len();
8863        }
8864        assert_eq!(first, rows, "every row was read back");
8865        let raw = (0..rows).map(|row| unique(row).len()).sum::<usize>();
8866        let size = fs::metadata(&path).expect("the file is there").len() as usize;
8867        assert!(size < raw, "a column without a dictionary is still encoded: {size} against {raw}");
8868        fs::remove_file(path).expect("remove scratch file");
8869    }
8870
8871    /// A payload of many blocks reads and checks every block of it.
8872    ///
8873    /// The test above has a dictionary of three values, which is one block, so it says nothing
8874    /// about a reader finding the right block among many. This one has thirty two thousand values,
8875    /// which is thirty two blocks, and it reads a value out of the first block and a value out of
8876    /// the last and then damages the last and asks for it again.
8877    ///
8878    /// Forty thousand rows over those thirty two thousand values, because a column the writer finds
8879    /// to be all distinct does not get a dictionary at all and there would be nothing here to test.
8880    /// Four rows in five holding a value the stripe has not seen before is a column that keeps one.
8881    /// The repeats are put at the front so that the values still arrive in order after them, which
8882    /// is what keeps the last part of the table on the last block of the payload.
8883    #[test]
8884    fn a_dictionary_over_many_blocks_checks_every_block_of_it() {
8885        let path = path("dictionary-blocks");
8886        let value = |row: usize| {
8887            let row = row.saturating_sub(8_000);
8888            format!("{row:07} a value long enough to be worth a payload block")
8889        };
8890        let parts = 40;
8891        let per_part = 1000;
8892        let mut writer =
8893            Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
8894                .expect("new file");
8895        for part in 0..parts {
8896            let values = (0..per_part)
8897                .map(|row| Value::Varchar(value(part * per_part + row)))
8898                .collect::<Vec<_>>();
8899            let chunk = Chunk::new(vec![
8900                Vector::from_values(LogicalType::Varchar, &values).expect("strings"),
8901            ])
8902            .expect("matching rows");
8903            writer.append(&chunk).expect("a part");
8904        }
8905        writer.finish().expect("commit");
8906
8907        let reader = Reader::open(&path).expect("reopen from disk");
8908        let dictionary = reader.table.dictionaries[0].expect("string dictionary page");
8909        assert!(
8910            parts * per_part > TEXT_PAYLOAD_VALUES * 4,
8911            "the dictionary has to be several blocks for this to be testing anything"
8912        );
8913        for part in [0, parts - 1] {
8914            let chunk = reader.read(part, &[0]).expect("a part");
8915            chunk.validate_external().expect("every payload block checks out");
8916            assert_eq!(chunk.value_at(0, 0), Value::Varchar(value(part * per_part)));
8917        }
8918
8919        let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
8920        file.seek(SeekFrom::Start(dictionary.offset + u64::from(dictionary.length) - 4))
8921            .expect("the last bytes of the page are payload");
8922        file.write_all(&[255]).expect("damage the last payload block");
8923        let reader = Reader::open(&path).expect("the directory and the index are untouched");
8924        let chunk = reader.read(parts - 1, &[0]).expect("the code page remains valid");
8925        let error = chunk.validate_external().expect_err("the damage must reach the caller");
8926        assert!(error.message().contains("payload checksum differs"), "{error}");
8927        fs::remove_file(path).expect("remove scratch file");
8928    }
8929
8930    /// Values of different lengths read back where the offsets say they do.
8931    ///
8932    /// The offsets are packed at one width for the column, they are relative to the payload block a
8933    /// value lands in, and they go in runs of half a block, so there are two boundaries where the
8934    /// arithmetic could be off by one and neither shows up on values that are all the same length.
8935    /// This writes 5,000 values whose lengths cycle through a wide range and reads every one back,
8936    /// so the first value of a block, the last value of a run and the last value of a block are all
8937    /// covered several times over. An empty value is in the cycle because a zero length span is the
8938    /// case the reader short circuits.
8939    ///
8940    /// Six thousand rows over those 5,000 values, because a column the writer finds to be all
8941    /// distinct is written without a dictionary and then there are no packed offsets to be off by
8942    /// one in.
8943    #[test]
8944    fn values_of_different_lengths_read_back_out_of_packed_offsets() {
8945        let path = path("dictionary-offsets");
8946        let value = |row: usize| {
8947            let row = row % 5_000;
8948            if row % 511 == 3 { String::new() } else { "x".repeat(row % 97) + &format!("{row:05}") }
8949        };
8950        let rows = 6_000;
8951        let mut writer =
8952            Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
8953                .expect("new file");
8954        let values = (0..rows).map(|row| Value::Varchar(value(row))).collect::<Vec<_>>();
8955        for part in values.chunks(1_000) {
8956            let chunk =
8957                Chunk::new(vec![Vector::from_values(LogicalType::Varchar, part).expect("strings")])
8958                    .expect("matching rows");
8959            writer.append(&chunk).expect("a part");
8960        }
8961        writer.finish().expect("commit");
8962
8963        let reader = Reader::open(&path).expect("reopen from disk");
8964        assert!(
8965            rows > TEXT_PAYLOAD_VALUES * 4,
8966            "the dictionary has to be several blocks for this to be testing anything"
8967        );
8968        for part in 0..rows / 1_000 {
8969            let chunk = reader.read(part, &[0]).expect("a part");
8970            for row in 0..1_000 {
8971                let row = part * 1_000 + row;
8972                assert_eq!(
8973                    chunk.value_at(row % 1_000, 0),
8974                    Value::Varchar(value(row)),
8975                    "value {row}"
8976                );
8977            }
8978        }
8979        fs::remove_file(path).expect("remove scratch file");
8980    }
8981
8982    /// Every worker of a scan wants the dictionary at the same moment and one of them fetches it.
8983    ///
8984    /// Asking a `OnceLock` whether it holds something answers the question a worker that already has
8985    /// the dictionary is asking and not the one a worker without it is asking, which is whether
8986    /// somebody is already on their way with it. Sixteen workers that all miss will all read the
8987    /// page, all verify it and all decode it, and fifteen will drop the result. Nothing about that
8988    /// is incorrect, which is why it went unnoticed, and it showed up as ClickBench 38 getting
8989    /// slower when the scan in front of it got faster and stopped staggering the arrivals.
8990    ///
8991    /// The barrier is what makes the test about that rather than about luck. Without it the first
8992    /// thread is usually finished before the last one starts and the count is one either way.
8993    #[test]
8994    fn a_global_dictionary_is_opened_once_however_many_workers_ask_at_once() {
8995        let path = path("dictionary-once");
8996        let parts = 8;
8997        let per_part = 500;
8998        let value =
8999            |row: usize| format!("{row:07} a value long enough to be worth a payload block");
9000        let mut writer =
9001            Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
9002                .expect("new file");
9003        for part in 0..parts {
9004            let values = (0..per_part)
9005                .map(|row| Value::Varchar(value(part * per_part + row)))
9006                .collect::<Vec<_>>();
9007            let chunk = Chunk::new(vec![
9008                Vector::from_values(LogicalType::Varchar, &values).expect("strings"),
9009            ])
9010            .expect("matching rows");
9011            writer.append(&chunk).expect("a part");
9012        }
9013        writer.finish().expect("commit");
9014
9015        let reader = Reader::open(&path).expect("reopen from disk");
9016        assert!(reader.table.dictionaries[0].is_some(), "the column has to have one to share");
9017        assert_eq!(reader.reads().dictionaries, 0, "opening the file does not open a dictionary");
9018
9019        let workers = 16;
9020        let gate = std::sync::Barrier::new(workers);
9021        std::thread::scope(|scope| {
9022            for worker in 0..workers {
9023                let reader = reader.clone();
9024                let gate = &gate;
9025                scope.spawn(move || {
9026                    gate.wait();
9027                    let chunk = reader.read(worker % parts, &[0]).expect("a part");
9028                    assert_eq!(
9029                        chunk.value_at(0, 0),
9030                        Value::Varchar(value((worker % parts) * per_part))
9031                    );
9032                });
9033            }
9034        });
9035
9036        assert_eq!(reader.reads().dictionaries, 1, "sixteen workers, one dictionary, one open");
9037        fs::remove_file(path).expect("remove scratch file");
9038    }
9039
9040    /// The sorted order sits outside the index the page checksum covers, because a query that
9041    /// never searches a dictionary should not read it, so it carries its own checksums and this is
9042    /// what says they are checked. A search that trusted a damaged order would give a wrong answer
9043    /// rather than a slow one.
9044    #[test]
9045    fn a_damaged_sorted_order_is_an_error() {
9046        let path = path("damaged-order");
9047        let mut writer = Writer::create(
9048            &path,
9049            "items",
9050            vec![
9051                Field::required("id", LogicalType::Integer),
9052                Field::new("text", LogicalType::Varchar),
9053            ],
9054        )
9055        .expect("new file");
9056        writer.append(&sample()).expect("stripe written");
9057        writer.finish().expect("commit");
9058
9059        let reader = Reader::open(&path).expect("valid directory");
9060        let page = reader.table.dictionaries[1].expect("string dictionary page");
9061        let mut header = [0; DICTIONARY_HEADER];
9062        read_at(&reader.file, page.offset, &mut header).expect("dictionary header");
9063        let index_len = dictionary_index_len(&header);
9064        let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
9065        file.seek(SeekFrom::Start(page.offset + index_len)).expect("the first head");
9066        file.write_all(&[255]).expect("damage the order");
9067
9068        let dictionary = reader.dictionary(1).expect("read").expect("a string column has one");
9069        let error = dictionary.compare_rank(0, b"anything").expect_err("a damaged order is caught");
9070        assert!(error.message().contains("rank checksum differs"), "{error}");
9071        fs::remove_file(path).expect("remove scratch file");
9072    }
9073
9074    /// Codes stay in first appearance order and the sorted order is written beside them, so a
9075    /// reader can put the values back in order without the writer having had to know them all
9076    /// before it handed out the first code.
9077    #[test]
9078    fn a_global_dictionary_carries_the_sorted_order_of_its_values() {
9079        // Chosen so the sort cannot be decided on the first eight bytes alone. Three values share
9080        // a nine byte prefix, one is a prefix of another, and one is empty.
9081        let spellings = ["overlong1z", "b", "", "overlong1a", "overlong", "ab", "a", "overlong1"];
9082        let path = path("dictionary-order");
9083        let mut writer =
9084            Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
9085                .expect("new file");
9086        writer
9087            .append(
9088                &Chunk::new(vec![
9089                    Vector::from_values(
9090                        LogicalType::Varchar,
9091                        &spellings.map(|text| Value::Varchar(text.into())),
9092                    )
9093                    .expect("strings"),
9094                ])
9095                .expect("one column"),
9096            )
9097            .expect("stripe written");
9098        writer.finish().expect("commit");
9099
9100        let reader = Reader::open(&path).expect("valid directory");
9101        let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
9102        let count = dictionary.ranks().expect("a v10 file stores one");
9103        assert_eq!(count, spellings.len(), "every distinct value has a rank");
9104        let order = (0..count)
9105            .map(|rank| dictionary.code_at_rank(rank).expect("a code"))
9106            .collect::<Vec<_>>();
9107        let mut seen = order.clone();
9108        seen.sort_unstable();
9109        assert_eq!(seen, (0..spellings.len() as u32).collect::<Vec<_>>(), "a permutation of codes");
9110
9111        let ranked = order
9112            .iter()
9113            .map(|&code| {
9114                dictionary.try_bytes_at(code as usize).expect("read").expect("a value").to_vec()
9115            })
9116            .collect::<Vec<_>>();
9117        let mut expected = spellings.map(|text| text.as_bytes().to_vec()).to_vec();
9118        expected.sort();
9119        assert_eq!(ranked, expected, "rank order is value order");
9120
9121        // What a search asks, on the values themselves rather than through a kernel, so that a
9122        // file whose heads disagree with its bytes is caught here rather than as a wrong answer.
9123        for (rank, value) in expected.iter().enumerate() {
9124            assert_eq!(
9125                dictionary.compare_rank(rank, value).expect("compare"),
9126                Ordering::Equal,
9127                "rank {rank} is its own value"
9128            );
9129            if rank > 0 {
9130                assert_eq!(
9131                    dictionary.compare_rank(rank - 1, value).expect("compare"),
9132                    Ordering::Less,
9133                    "rank {rank} follows the one before it"
9134                );
9135            }
9136        }
9137        fs::remove_file(path).expect("remove scratch file");
9138    }
9139
9140    /// A sweep of the dictionary reads every value and keeps what it read, up to the budget.
9141    ///
9142    /// The point of the sweep is the resident size rather than the answer, so both are checked
9143    /// here. A dictionary this small is well under [`TEXT_KEEP_BUDGET`], so it keeps everything and
9144    /// a second sweep decodes nothing, which is what makes the second statement of a session asking
9145    /// the same question cost what it should. The ceiling is the other half of it and it has its own
9146    /// test below, because a ceiling that never binds is not a ceiling anybody checked.
9147    #[test]
9148    fn a_dictionary_sweep_reads_every_value_and_keeps_it_under_the_budget() {
9149        let path = path("dictionary-sweep");
9150        // Two thousand five hundred distinct values is two whole payload blocks and a part of a
9151        // third, so the sweep has to be called more than once and the last call has to stop short.
9152        let spellings = (0..2_500)
9153            .map(|index| Value::Varchar(format!("value {index:08} {}", "x".repeat(index % 40))))
9154            .collect::<Vec<_>>();
9155        let mut writer =
9156            Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
9157                .expect("new file");
9158        // A chunk is a part and a part is at most 1,024 rows, so the values go in three of them.
9159        // The dictionary is table wide and does not care where a value was written.
9160        for part in spellings.chunks(1_024) {
9161            writer
9162                .append(
9163                    &Chunk::new(vec![
9164                        Vector::from_values(LogicalType::Varchar, part).expect("strings"),
9165                    ])
9166                    .expect("one column"),
9167                )
9168                .expect("stripe written");
9169        }
9170        writer.finish().expect("commit");
9171
9172        let reader = Reader::open(&path).expect("valid directory");
9173        let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
9174        assert_eq!(dictionary.len(), spellings.len(), "every value is distinct");
9175
9176        let resting = dictionary.footprint();
9177        let mut swept: Vec<Vec<u8>> = Vec::new();
9178        let mut at = 0;
9179        let mut calls = 0;
9180        while at < dictionary.len() {
9181            let stopped = dictionary
9182                .sweep_text(at, dictionary.len(), &mut |index: usize, text: &[u8]| {
9183                    assert_eq!(index, swept.len(), "a sweep hands its values over in order");
9184                    swept.push(text.to_vec());
9185                    Ok(())
9186                })
9187                .expect("a sweep reads");
9188            assert!(stopped > at, "a sweep moves");
9189            at = stopped;
9190            calls += 1;
9191        }
9192        assert_eq!(calls, 3, "a sweep hands over one block at a time");
9193        let after = dictionary.footprint();
9194        assert!(after > resting, "a sweep under the budget keeps what it decoded");
9195
9196        let read = (0..dictionary.len())
9197            .map(|code| dictionary.try_bytes_at(code).expect("read").expect("a value").to_vec())
9198            .collect::<Vec<_>>();
9199        assert_eq!(swept, read, "a sweep answers what a point read answers");
9200        assert_eq!(dictionary.footprint(), after, "a point read of a kept block decodes nothing");
9201        fs::remove_file(path).expect("remove scratch file");
9202    }
9203
9204    /// A sweep over a block whose second run of offsets is short reads the same values as a point
9205    /// read does.
9206    ///
9207    /// The sweep decodes the offsets of a whole run at a time rather than a value at a time, and a
9208    /// run holds half a block, so the count it asks for is the run length everywhere but at the end
9209    /// of the dictionary. Two thousand five hundred values, which is what the test above writes,
9210    /// never puts a short run second in its block: the last block there begins on a run boundary and
9211    /// holds one run. Two thousand eight hundred does, so the last block is a whole run of five
9212    /// hundred and twelve followed by two hundred and forty, and an off by one in either the count
9213    /// asked for or the slice taken out of the answer shows up as a wrong value or a refusal.
9214    #[test]
9215    fn a_sweep_over_a_block_with_a_short_second_run_reads_what_a_point_read_reads() {
9216        let path = path("dictionary-sweep-short-run");
9217        let spellings = (0..2_800)
9218            .map(|index| Value::Varchar(format!("value {index:08} {}", "x".repeat(index % 40))))
9219            .collect::<Vec<_>>();
9220        let mut writer =
9221            Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
9222                .expect("new file");
9223        for part in spellings.chunks(1_024) {
9224            writer
9225                .append(
9226                    &Chunk::new(vec![
9227                        Vector::from_values(LogicalType::Varchar, part).expect("strings"),
9228                    ])
9229                    .expect("one column"),
9230                )
9231                .expect("stripe written");
9232        }
9233        writer.finish().expect("commit");
9234
9235        let reader = Reader::open(&path).expect("valid directory");
9236        let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
9237        assert_eq!(dictionary.len(), spellings.len(), "every value is distinct");
9238        let last = dictionary.len() % TEXT_PAYLOAD_VALUES;
9239        assert!(last > TEXT_OFFSET_RUN, "the last block has to reach into a second run of offsets");
9240        assert!(last < TEXT_PAYLOAD_VALUES, "and that second run has to be short of a whole one");
9241
9242        let mut swept: Vec<Vec<u8>> = Vec::new();
9243        let mut at = 0;
9244        while at < dictionary.len() {
9245            let stopped = dictionary
9246                .sweep_text(at, dictionary.len(), &mut |index: usize, text: &[u8]| {
9247                    assert_eq!(index, swept.len(), "a sweep hands its values over in order");
9248                    swept.push(text.to_vec());
9249                    Ok(())
9250                })
9251                .expect("a sweep reads");
9252            assert!(stopped > at, "a sweep moves");
9253            at = stopped;
9254        }
9255        let read = (0..dictionary.len())
9256            .map(|code| dictionary.try_bytes_at(code).expect("read").expect("a value").to_vec())
9257            .collect::<Vec<_>>();
9258        assert_eq!(swept, read, "a sweep answers what a point read answers");
9259        fs::remove_file(path).expect("remove scratch file");
9260    }
9261
9262    /// Narrowing a page takes what fits and refuses the page for anything that does not.
9263    ///
9264    /// The edges of the range on both sides and one step past each of them, for every type, because
9265    /// checking a page separately from converting it is only right if the check refuses exactly what
9266    /// `TryFrom` would have refused, and off by one there is a file that reads back a different
9267    /// number than it was given. The check is a bit pattern rather than a comparison, so it is not
9268    /// the shape a reader would guess from the bounds, which is why all six are here. The empty page
9269    /// is here because a check written the obvious way starts with the extremes the wrong way round
9270    /// and refuses it.
9271    #[test]
9272    fn narrowing_a_page_takes_what_fits_and_refuses_what_does_not() {
9273        assert_eq!(fit::<i8>(&[]).expect("an empty page fits anything"), Vec::<i8>::new());
9274        assert_eq!(fit::<i8>(&[-128, 0, 127]).expect("the edges fit"), vec![-128_i8, 0, 127]);
9275        fit::<i8>(&[128]).expect_err("one past the top does not fit");
9276        fit::<i8>(&[-129]).expect_err("one past the bottom does not fit");
9277        assert_eq!(fit::<u8>(&[0, 255]).expect("the edges fit"), vec![0_u8, 255]);
9278        fit::<u8>(&[256]).expect_err("one past the top does not fit");
9279        fit::<u8>(&[-1]).expect_err("a negative does not fit an unsigned page");
9280        assert_eq!(
9281            fit::<i16>(&[-32_768, 0, 32_767]).expect("the edges fit"),
9282            vec![-32_768_i16, 0, 32_767]
9283        );
9284        fit::<i16>(&[32_768]).expect_err("one past the top does not fit");
9285        fit::<i16>(&[-32_769]).expect_err("one past the bottom does not fit");
9286        assert_eq!(fit::<u16>(&[0, 65_535]).expect("the edges fit"), vec![0_u16, 65_535]);
9287        fit::<u16>(&[65_536]).expect_err("one past the top does not fit");
9288        fit::<u16>(&[-1]).expect_err("a negative does not fit an unsigned page");
9289        assert_eq!(
9290            fit::<i32>(&[i64::from(i32::MIN), 0, i64::from(i32::MAX)]).expect("the edges fit"),
9291            vec![i32::MIN, 0, i32::MAX]
9292        );
9293        fit::<i32>(&[i64::from(i32::MAX) + 1]).expect_err("one past the top does not fit");
9294        fit::<i32>(&[i64::from(i32::MIN) - 1]).expect_err("one past the bottom does not fit");
9295        assert_eq!(
9296            fit::<u32>(&[0, 4_294_967_295]).expect("the edges fit"),
9297            vec![0_u32, 4_294_967_295]
9298        );
9299        fit::<u32>(&[4_294_967_296]).expect_err("one past the top does not fit");
9300        fit::<u32>(&[-1]).expect_err("a negative does not fit an unsigned page");
9301
9302        // One value in a page that fits is still a page that does not, which is the thing an or
9303        // into an accumulator could get wrong in a way a page of one value would never show.
9304        fit::<i8>(&[0, 1, 2, 128, 3]).expect_err("one bad value spoils the page");
9305    }
9306
9307    /// The residue says yes to exactly what `TryFrom` says yes to.
9308    ///
9309    /// The edges above are the cases anyone would think to write down. This is the argument that
9310    /// there are no others, made by asking both questions about every value either narrow type could
9311    /// have an opinion about, and then about the values around the wide edges and the ends of an
9312    /// `i64`, which a range that size cannot reach.
9313    #[test]
9314    fn the_residue_agrees_with_a_checked_conversion_everywhere() {
9315        for value in -70_000_i64..70_000 {
9316            assert_eq!(fit::<i8>(&[value]).is_ok(), i8::try_from(value).is_ok(), "{value} as i8");
9317            assert_eq!(fit::<u8>(&[value]).is_ok(), u8::try_from(value).is_ok(), "{value} as u8");
9318            assert_eq!(fit::<i16>(&[value]).is_ok(), i16::try_from(value).is_ok(), "{value} i16");
9319            assert_eq!(fit::<u16>(&[value]).is_ok(), u16::try_from(value).is_ok(), "{value} u16");
9320        }
9321        let wide = [i64::MIN, i64::MIN + 1, i64::from(i32::MIN), 0, i64::from(u32::MAX), i64::MAX];
9322        for edge in wide {
9323            for step in -2_i64..=2 {
9324                let value = edge.saturating_add(step);
9325                assert_eq!(
9326                    fit::<i32>(&[value]).is_ok(),
9327                    i32::try_from(value).is_ok(),
9328                    "{value} as i32"
9329                );
9330                assert_eq!(
9331                    fit::<u32>(&[value]).is_ok(),
9332                    u32::try_from(value).is_ok(),
9333                    "{value} as u32"
9334                );
9335            }
9336        }
9337    }
9338
9339    /// A dictionary at its budget sweeps without keeping, and still answers what it answered.
9340    ///
9341    /// The budget is a quarter of a gigabyte in a running database, which is a fine size for a real
9342    /// column and no size at all for a test, so this opens the same dictionary a second time with a
9343    /// budget of zero. That is the shape of the hundred million row case: `URL` fills the budget
9344    /// somewhere in the middle of itself and everything past that point is read and dropped, which
9345    /// costs the decode again and holds none of it.
9346    #[test]
9347    fn a_dictionary_at_its_budget_sweeps_without_keeping() {
9348        let path = path("dictionary-budget");
9349        let spellings = (0..2_500)
9350            .map(|index| Value::Varchar(format!("value {index:08} {}", "y".repeat(index % 40))))
9351            .collect::<Vec<_>>();
9352        let mut writer =
9353            Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
9354                .expect("new file");
9355        for part in spellings.chunks(1_024) {
9356            writer
9357                .append(
9358                    &Chunk::new(vec![
9359                        Vector::from_values(LogicalType::Varchar, part).expect("strings"),
9360                    ])
9361                    .expect("one column"),
9362                )
9363                .expect("stripe written");
9364        }
9365        writer.finish().expect("commit");
9366
9367        let reader = Reader::open(&path).expect("valid directory");
9368        let page = reader.table.dictionaries[0].expect("a string column has one");
9369        let file = Arc::clone(&reader.file);
9370        let starved = open_global_dictionary(file, page, &LogicalType::Varchar, 0)
9371            .expect("a dictionary opens whatever it may keep");
9372
9373        let resting = starved.footprint();
9374        let mut swept: Vec<Vec<u8>> = Vec::new();
9375        let mut at = 0;
9376        while at < starved.len() {
9377            at = starved
9378                .sweep_text(at, starved.len(), &mut |_index: usize, text: &[u8]| {
9379                    swept.push(text.to_vec());
9380                    Ok(())
9381                })
9382                .expect("a sweep reads");
9383        }
9384        assert_eq!(swept.len(), spellings.len(), "a starved sweep still reads every value");
9385        assert_eq!(starved.footprint(), resting, "and keeps no block it decoded");
9386
9387        let generous = reader.dictionary(0).expect("read").expect("a string column has one");
9388        let read = (0..generous.len())
9389            .map(|code| generous.try_bytes_at(code).expect("read").expect("a value").to_vec())
9390            .collect::<Vec<_>>();
9391        assert_eq!(swept, read, "a starved sweep answers what a point read answers");
9392        fs::remove_file(path).expect("remove scratch file");
9393    }
9394
9395    #[test]
9396    fn damaged_membership_cannot_skip_a_string_page() {
9397        let path = path("damaged-membership");
9398        let mut writer = Writer::create(
9399            &path,
9400            "items",
9401            vec![
9402                Field::required("id", LogicalType::Integer),
9403                Field::new("text", LogicalType::Varchar),
9404            ],
9405        )
9406        .expect("new file");
9407        writer.append(&sample()).expect("stripe written");
9408        writer.finish().expect("commit");
9409
9410        let reader = Reader::open(&path).expect("valid directory");
9411        let membership = reader.table.stripes[0].memberships[1].expect("string membership");
9412        let mut file = OpenOptions::new().write(true).open(&path).expect("open membership page");
9413        file.seek(SeekFrom::Start(membership.offset)).expect("membership start");
9414        file.write_all(&[255]).expect("damage membership");
9415        let error = reader.skips_codes(0, 1, &[3]).expect_err("corruption must not skip rows");
9416        assert!(error.message().contains("membership page checksum differs"), "{error}");
9417        fs::remove_file(path).expect("remove scratch file");
9418    }
9419
9420    #[test]
9421    fn membership_delta_stream_is_sorted_exact_and_bounded() {
9422        let unique = unique_codes(&[900, 4, 4, 72, 9, u32::MAX]);
9423        assert_eq!(unique, [4, 9, 72, 900, u32::MAX]);
9424        let encoded = encode_membership(&unique);
9425        assert_eq!(
9426            decode_membership(&encoded).expect("valid membership"),
9427            [4, 9, 72, 900, u32::MAX]
9428        );
9429        // A stripe's index is the union of its parts', so a code in two of them is in it once and
9430        // the result is still one ascending run of deltas.
9431        let merged = merged_codes(vec![vec![4, 900], vec![9, 900, u32::MAX], vec![72]]);
9432        assert_eq!(merged, [4, 9, 72, 900, u32::MAX]);
9433        assert_eq!(
9434            decode_membership(&encode_membership(&merged)).expect("valid membership"),
9435            unique
9436        );
9437        assert!(decode_membership(&[1, 0x80]).is_err(), "a truncated varint is invalid");
9438        assert!(
9439            decode_membership(&[1, 0xff, 0xff, 0xff, 0xff, 0x10]).is_err(),
9440            "a value past u32 is invalid"
9441        );
9442    }
9443
9444    #[test]
9445    fn a_global_dictionary_may_be_larger_than_one_column_page() {
9446        let dictionary = Page {
9447            offset: HEADER,
9448            length: u32::try_from(MAX_PAGE + 1).expect("the page bound fits on disk"),
9449            hash: 0,
9450        };
9451        let table = Table {
9452            name: "items".to_owned(),
9453            fields: vec![Field::new("text", LogicalType::Varchar)],
9454            stripes: Vec::new(),
9455            rows: 0,
9456            dictionaries: vec![Some(dictionary)],
9457            distincts: vec![None],
9458            frequencies: vec![None],
9459            clustering: None,
9460            generation: 1,
9461            sections: Vec::new(),
9462        };
9463        let directory = encode_directory(&table).expect("directory");
9464        let file_size = dictionary.offset + u64::from(dictionary.length) + 1;
9465
9466        let decoded = decode_directory(&directory, file_size).expect("large lazy dictionary");
9467        assert_eq!(decoded.dictionaries[0].expect("dictionary").length, dictionary.length);
9468    }
9469
9470    #[test]
9471    fn a_column_with_one_value_everywhere_costs_almost_nothing_a_row() {
9472        let path = path("constant-codes");
9473        let mut writer =
9474            Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
9475                .expect("new file");
9476        let empty = vec![Value::Varchar(String::new()); 1024];
9477        for _ in 0..4 {
9478            let column = Vector::from_values(LogicalType::Varchar, &empty).expect("strings");
9479            writer.append(&Chunk::new(vec![column]).expect("one column")).expect("a part");
9480        }
9481        writer.finish().expect("commit");
9482
9483        let reader = Reader::open(&path).expect("valid directory");
9484        let pages = reader.layout().columns.first().expect("one column").pages;
9485        // This column used to cost four bytes a row, 16,384 of them, the same as a column of four
9486        // thousand distinct URLs would. The cascade calls each part a constant, so what is left is
9487        // a tag, a count and the value, and the row count stops being what drives the number.
9488        assert!(pages < 256, "{pages} bytes of pages for 4,096 rows of one value");
9489        let read = reader.read(3, &[0]).expect("the last part back");
9490        assert_eq!(read.value_at(0, 0), Value::Varchar(String::new()));
9491        assert_eq!(read.value_at(1023, 0), Value::Varchar(String::new()));
9492        fs::remove_file(path).expect("remove scratch file");
9493    }
9494
9495    #[test]
9496    fn a_cascade_value_too_wide_for_its_column_is_refused_rather_than_cut() {
9497        // What a damaged page looks like from here: the cascade decoded, so the bytes are not
9498        // truncated, but the values do not belong to the column the directory says they do.
9499        let over = vec![i64::from(i32::MAX) + 1];
9500        let error = narrowed(&LogicalType::Integer, over).expect_err("a page that disagrees");
9501        assert!(format!("{error}").contains("not of its type"), "{error}");
9502        assert!(narrowed(&LogicalType::BigInt, vec![i64::MIN]).is_ok(), "bigint holds all of i64");
9503        assert!(narrowed(&LogicalType::Varchar, vec![0]).is_err(), "strings are not integers");
9504    }
9505
9506    #[test]
9507    fn a_code_stream_the_cascade_cannot_shrink_is_left_alone() {
9508        // A shift register rather than a run, because an arithmetic run is the one wide shape the
9509        // cascade does shrink. This is what a column with tens of millions of distinct values hands
9510        // over: full width codes with no order to them.
9511        let mut state: u32 = 0x9e37_79b9;
9512        let spread: Vec<u32> = (0..1024)
9513            .map(|_| {
9514                state ^= state << 13;
9515                state ^= state >> 17;
9516                state ^= state << 5;
9517                state
9518            })
9519            .collect();
9520        assert_eq!(encoded_codes(&spread).expect("no failure"), None);
9521        let near: Vec<u32> = (0..1024).collect();
9522        let coded = encoded_codes(&near).expect("no failure").expect("counting up is packable");
9523        assert!(coded.len() < near.len() * 4, "{} bytes for a run of 1,024", coded.len());
9524    }
9525
9526    /// The columns of a stripe are encoded on whichever thread got to them, so the one thing that
9527    /// must not depend on which thread that was is the file. Two writes of the same rows are
9528    /// compared byte for byte rather than value for value, because a dictionary that two columns
9529    /// somehow shared would still read back correctly and would hand out its codes in the order the
9530    /// threads happened to run in, which is exactly what this is here to catch.
9531    #[test]
9532    fn two_writes_of_the_same_rows_give_the_same_bytes() {
9533        fn written(path: &PathBuf) {
9534            let fields = (0..40)
9535                .map(|column| {
9536                    let ty =
9537                        if column % 4 == 0 { LogicalType::Varchar } else { LogicalType::BigInt };
9538                    Field::new(format!("c{column}"), ty)
9539                })
9540                .collect::<Vec<_>>();
9541            let mut writer = Writer::create(path, "wide", fields).expect("new file");
9542            for part in 0..70_u64 {
9543                let columns = (0..40)
9544                    .map(|column| {
9545                        let values = (0..64_u64)
9546                            .map(|row| {
9547                                let seed = part.wrapping_mul(31).wrapping_add(row);
9548                                if column % 4 == 0 {
9549                                    Value::Varchar(format!("v{}", seed % 17))
9550                                } else {
9551                                    Value::BigInt(i64::try_from(seed % 97).expect("small"))
9552                                }
9553                            })
9554                            .collect::<Vec<_>>();
9555                        let ty = if column % 4 == 0 {
9556                            LogicalType::Varchar
9557                        } else {
9558                            LogicalType::BigInt
9559                        };
9560                        Vector::from_values(ty, &values).expect("a column")
9561                    })
9562                    .collect::<Vec<_>>();
9563                writer.append(&Chunk::new(columns).expect("forty columns")).expect("a part");
9564            }
9565            writer.finish().expect("commit");
9566        }
9567
9568        let first = path("repeatable-one");
9569        let second = path("repeatable-two");
9570        written(&first);
9571        written(&second);
9572        let left = fs::read(&first).expect("the first file");
9573        let right = fs::read(&second).expect("the second file");
9574        assert_eq!(left.len(), right.len(), "two writes of the same rows differ in length");
9575        assert!(left == right, "two writes of the same rows differ in their bytes");
9576
9577        // And the rows are still there, since a pair of identically wrong files would pass the
9578        // comparison above on its own.
9579        let reader = Reader::open(&first).expect("valid directory");
9580        assert_eq!(reader.table().rows(), 70 * 64);
9581        let read = reader.read(0, &[0, 1]).expect("the first part back");
9582        assert_eq!(read.value_at(0, 0), Value::Varchar("v0".to_owned()));
9583        assert_eq!(read.value_at(0, 1), Value::BigInt(0));
9584        fs::remove_file(first).expect("remove scratch file");
9585        fs::remove_file(second).expect("remove scratch file");
9586    }
9587
9588    /// Three tables of different shapes in one file, read back by name.
9589    fn three_tables(path: &PathBuf) {
9590        let writer = Writer::create(
9591            path,
9592            "region",
9593            vec![
9594                Field::new("r_key", LogicalType::Integer),
9595                Field::new("r_name", LogicalType::Varchar),
9596            ],
9597        )
9598        .expect("new file");
9599        let mut writer = writer;
9600        writer
9601            .append(
9602                &Chunk::new(vec![
9603                    Vector::from_values(
9604                        LogicalType::Integer,
9605                        &[Value::Integer(0), Value::Integer(1)],
9606                    )
9607                    .expect("keys"),
9608                    Vector::from_values(
9609                        LogicalType::Varchar,
9610                        &[Value::Varchar("AFRICA".to_owned()), Value::Varchar("ASIA".to_owned())],
9611                    )
9612                    .expect("names"),
9613                ])
9614                .expect("two columns"),
9615            )
9616            .expect("a part");
9617        let mut writer = writer
9618            .next("empty", vec![Field::new("nothing", LogicalType::BigInt)])
9619            .expect("a second table");
9620        writer
9621            .append(
9622                &Chunk::new(vec![
9623                    Vector::from_values(LogicalType::BigInt, &[Value::BigInt(7)]).expect("a row"),
9624                ])
9625                .expect("one column"),
9626            )
9627            .expect("a part");
9628        let mut writer =
9629            writer.next("wide", vec![Field::new("n", LogicalType::BigInt)]).expect("a third table");
9630        for part in 0..70_i64 {
9631            let values = (0..64).map(|row| Value::BigInt(part * 64 + row)).collect::<Vec<_>>();
9632            writer
9633                .append(
9634                    &Chunk::new(vec![
9635                        Vector::from_values(LogicalType::BigInt, &values).expect("a column"),
9636                    ])
9637                    .expect("one column"),
9638                )
9639                .expect("a part");
9640        }
9641        writer.finish().expect("commit");
9642    }
9643
9644    #[test]
9645    fn three_tables_in_one_file_read_back_by_name() {
9646        let file = path("three-tables");
9647        three_tables(&file);
9648        let catalog = Catalog::open(&file).expect("a committed catalog");
9649        assert_eq!(catalog.names().collect::<Vec<_>>(), ["region", "empty", "wide"]);
9650
9651        let region = catalog.table("region").expect("the first table");
9652        assert_eq!(region.table().rows(), 2);
9653        assert_eq!(
9654            region.read(0, &[1]).expect("names").value_at(1, 0),
9655            Value::Varchar("ASIA".to_owned())
9656        );
9657
9658        let wide = catalog.table("wide").expect("the third table");
9659        assert_eq!(wide.table().rows(), 70 * 64);
9660        assert_eq!(wide.read(0, &[0]).expect("the first part").value_at(0, 0), Value::BigInt(0));
9661
9662        // The middle table is reached without the one after it having been touched, which is what
9663        // a directory per table buys over one directory of everything.
9664        let empty = catalog.table("empty").expect("the second table");
9665        assert_eq!(empty.table().rows(), 1);
9666        assert_eq!(empty.read(0, &[0]).expect("the row").value_at(0, 0), Value::BigInt(7));
9667
9668        fs::remove_file(file).expect("remove scratch file");
9669    }
9670
9671    #[test]
9672    fn a_name_the_file_does_not_hold_is_an_error_rather_than_the_first_table() {
9673        let file = path("three-tables-missing");
9674        three_tables(&file);
9675        let catalog = Catalog::open(&file).expect("a committed catalog");
9676        let error = catalog.table("nation").expect_err("no such table");
9677        assert!(error.message().contains("nation"), "{}", error.message());
9678        fs::remove_file(file).expect("remove scratch file");
9679    }
9680
9681    #[test]
9682    fn a_file_of_three_tables_will_not_open_as_one() {
9683        let file = path("three-tables-unnamed");
9684        three_tables(&file);
9685        let error = Reader::open(&file).expect_err("more than one table");
9686        assert!(error.message().contains("more than one table"), "{}", error.message());
9687        fs::remove_file(file).expect("remove scratch file");
9688    }
9689
9690    /// One column per storage width, because the width is what decides how many bytes a row costs.
9691    #[test]
9692    fn decimals_of_every_storage_width_round_trip() {
9693        let file = path("decimals");
9694        let widths = [(4_u8, 2_u8), (9, 2), (18, 4), (38, 6)];
9695        let fields = widths
9696            .iter()
9697            .enumerate()
9698            .map(|(index, (width, scale))| {
9699                Field::new(
9700                    format!("d{index}"),
9701                    LogicalType::decimal(*width, *scale).expect("a decimal type"),
9702                )
9703            })
9704            .collect::<Vec<_>>();
9705        let mut writer = Writer::create(&file, "money", fields).expect("new file");
9706        let rows: [i128; 3] = [-1234, 0, 999];
9707        let columns = widths
9708            .iter()
9709            .map(|(width, scale)| {
9710                let values = rows
9711                    .iter()
9712                    .map(|unscaled| Value::Decimal {
9713                        unscaled: *unscaled,
9714                        width: *width,
9715                        scale: *scale,
9716                    })
9717                    .collect::<Vec<_>>();
9718                Vector::from_values(
9719                    LogicalType::decimal(*width, *scale).expect("a decimal type"),
9720                    &values,
9721                )
9722                .expect("a decimal column")
9723            })
9724            .collect::<Vec<_>>();
9725        writer.append(&Chunk::new(columns).expect("four columns")).expect("a part");
9726        writer.finish().expect("commit");
9727
9728        let reader = Reader::open(&file).expect("a committed file");
9729        for (index, (width, scale)) in widths.iter().enumerate() {
9730            assert_eq!(
9731                reader.table().fields()[index].ty,
9732                LogicalType::decimal(*width, *scale).expect("a decimal type"),
9733                "column {index} came back as another type"
9734            );
9735            let column = reader.read(0, &[index]).expect("the column");
9736            for (row, unscaled) in rows.iter().enumerate() {
9737                assert_eq!(
9738                    column.value_at(row, 0),
9739                    Value::Decimal { unscaled: *unscaled, width: *width, scale: *scale },
9740                    "column {index} row {row}"
9741                );
9742            }
9743        }
9744        fs::remove_file(file).expect("remove scratch file");
9745    }
9746
9747    #[test]
9748    fn two_tables_of_one_name_are_refused_before_anything_is_committed() {
9749        let file = path("two-of-a-name");
9750        let writer = Writer::create(&file, "t", vec![Field::new("a", LogicalType::BigInt)])
9751            .expect("new file");
9752        let error = writer
9753            .next("t", vec![Field::new("a", LogicalType::BigInt)])
9754            .expect_err("the same name twice");
9755        assert!(error.message().contains("same name"), "{}", error.message());
9756        fs::remove_file(file).expect("remove scratch file");
9757    }
9758
9759    #[test]
9760    fn opening_the_catalog_reads_no_table_directory() {
9761        let file = path("catalog-only");
9762        three_tables(&file);
9763        let catalog = Catalog::open(&file).expect("a committed catalog");
9764        // The header and one slot, and nothing under it. The third table's directory covers seventy
9765        // stripes and reading it here would be the whole point of the two levels thrown away.
9766        assert_eq!(catalog.opening.reads, 2, "opening the catalog read more than the slot");
9767        assert_eq!(catalog.names().len(), 3);
9768        fs::remove_file(file).expect("remove scratch file");
9769    }
9770
9771    /// The checksum answers what it has always answered, at every length its branches split on.
9772    ///
9773    /// This is a compatibility test rather than a correctness one. Nothing about the hash has to be
9774    /// any particular function, but a file already on disk carries the answers the version that
9775    /// wrote it gave, so a change here is a change that makes every stored file fail to verify. The
9776    /// lengths are the ones the code makes decisions about: nothing, under a block, a block exactly,
9777    /// a block and a word, a word and a half word, and a half word and a byte.
9778    ///
9779    /// The empty answer is the published xxHash64 vector for an empty input at seed zero, which is
9780    /// also a check that this is the function it says it is.
9781    #[test]
9782    fn the_checksum_answers_what_it_has_always_answered() {
9783        let bytes: Vec<u8> =
9784            (0..1000_u32).map(|at| (at.wrapping_mul(31).wrapping_add(7) % 251) as u8).collect();
9785        for (length, expected) in [
9786            (0, 0xef46_db37_51d8_e999),
9787            (1, 0xa96c_7f0c_e858_bbb7),
9788            (3, 0x56e6_9576_32a4_87f9),
9789            (4, 0xc60d_15b1_e3ff_8f04),
9790            (5, 0x8088_1585_8624_dd4e),
9791            (7, 0xafbe_fc3d_6c6f_9a8e),
9792            (8, 0x3da5_c7aa_2696_83e0),
9793            (9, 0x465e_c429_b13c_3892),
9794            (15, 0xdee8_9d8a_065a_6233),
9795            (16, 0x1330_489a_7767_9c80),
9796            (31, 0x3391_303d_485e_846e),
9797            (32, 0x40b7_aff7_5d45_bbc8),
9798            (33, 0x4997_cae4_951c_17a5),
9799            (39, 0x5807_28fd_5c14_5739),
9800            (40, 0xf95c_f6f5_c08a_3d3b),
9801            (63, 0x2944_b4da_fc69_b206),
9802            (64, 0xbb76_f6ef_19bd_5a1b),
9803            (65, 0x814e_0c65_4a9f_d640),
9804            (127, 0x00de_aab1_31cf_f89b),
9805            (1000, 0x9e33_00c1_cde3_c58d),
9806        ] {
9807            assert_eq!(checksum(&bytes[..length]), expected, "the checksum of {length} bytes");
9808        }
9809        assert_eq!(checksum(b"the quick brown fox jumps over the lazy dog"), 0xed71_4233_c5a9_a792);
9810    }
9811    /// A declared order survives the file, and a table that declared none stays as it was.
9812    ///
9813    /// The second half is the one worth a test. The clustering section is written only when there
9814    /// is a declaration, so a file of two tables where one is clustered exercises both the present
9815    /// and the absent branch of the decoder in one directory, which is where a length bug would
9816    /// show up as one table reading the other's bytes.
9817    #[test]
9818    fn a_declared_order_comes_back_out_of_the_file() {
9819        let path = path("clustered");
9820        let shipped = vec![
9821            Field::new("key", LogicalType::BigInt),
9822            Field::new("line", LogicalType::Integer),
9823            Field::new("shipdate", LogicalType::Date),
9824        ];
9825        let plain = vec![Field::new("a", LogicalType::Integer)];
9826        let stage_zero = Clustering::new(vec![2, 0, 1], Width::Month, &shipped).expect("valid");
9827
9828        let mut writer = Writer::create(&path, "lineitem", shipped)
9829            .expect("new file")
9830            .declare(stage_zero.clone())
9831            .expect("the columns are the table's");
9832        let column = |ty: LogicalType, values: &[Value]| {
9833            Vector::from_values(ty, values).expect("the values match the type")
9834        };
9835        writer
9836            .append(
9837                &Chunk::new(vec![
9838                    column(
9839                        LogicalType::BigInt,
9840                        &[Value::BigInt(0), Value::BigInt(1), Value::BigInt(2), Value::BigInt(3)],
9841                    ),
9842                    column(
9843                        LogicalType::Integer,
9844                        &[
9845                            Value::Integer(1),
9846                            Value::Integer(1),
9847                            Value::Integer(1),
9848                            Value::Integer(1),
9849                        ],
9850                    ),
9851                    column(
9852                        LogicalType::Date,
9853                        &[Value::Date(0), Value::Date(1), Value::Date(2), Value::Date(3)],
9854                    ),
9855                ])
9856                .expect("three columns"),
9857            )
9858            .expect("four rows");
9859        let mut writer = writer.next("nation", plain).expect("a second table");
9860        writer
9861            .append(
9862                &Chunk::new(vec![column(LogicalType::Integer, &[Value::Integer(7)])])
9863                    .expect("one column"),
9864            )
9865            .expect("one row");
9866        writer.finish().expect("commit");
9867
9868        let catalog = Catalog::open(&path).expect("reopen");
9869        let lineitem = catalog.table("lineitem").expect("the clustered table");
9870        assert_eq!(lineitem.table().clustering(), Some(&stage_zero));
9871        let nation = catalog.table("nation").expect("the plain table");
9872        assert_eq!(nation.table().clustering(), None, "nobody declared one here");
9873
9874        // And the rows are still the rows, because the section goes on the end of the directory
9875        // and the easy way to break that is to leave the cursor somewhere the next read trusts.
9876        assert_eq!(lineitem.table().rows(), 4);
9877        assert_eq!(nation.table().rows(), 1);
9878        fs::remove_file(&path).ok();
9879    }
9880
9881    /// A declaration naming a column the table does not have is refused where it is made.
9882    #[test]
9883    fn a_declaration_off_the_end_of_the_table_never_reaches_the_file() {
9884        let path = path("clustered-bad");
9885        let writer = Writer::create(&path, "items", vec![Field::new("a", LogicalType::Integer)])
9886            .expect("new file");
9887        let four =
9888            (0..4).map(|at| Field::new(format!("c{at}"), LogicalType::Integer)).collect::<Vec<_>>();
9889        let wrong = Clustering::new(vec![3], Width::Exact, &four).expect("valid against four");
9890        assert!(writer.declare(wrong).is_err(), "the table has one column, not four");
9891        fs::remove_file(&path).ok();
9892    }
9893}