1#![forbid(unsafe_code)]
34
35use std::borrow::Cow;
36use std::cmp::Ordering;
37use std::collections::{HashMap, VecDeque};
38use std::fs::{File, OpenOptions};
39use std::io::{Read, Seek, SeekFrom};
40use std::mem::{size_of, size_of_val};
41use std::path::Path;
42use std::slice;
43use std::sync::atomic::{AtomicBool, AtomicUsize, Ordering as Atomic};
44use std::sync::{Arc, Mutex, OnceLock, Weak};
45
46use rudb_common::bounds::{self, Bound, Op, scaled_as};
47use rudb_common::{Clustering, Error, Field, LogicalType, PhysicalType, Result, Value, Width};
48use rudb_encoding::{bitpack, chooser, integer, string};
49use rudb_metrics::{LoadProfile, Stage};
50use rudb_storage::sieve::Sieve;
51use rudb_storage::{Probe, Range, Zone};
52use rudb_vector::string::StringColumn;
53use rudb_vector::validity::Validity;
54use rudb_vector::{Buffer, Chunk, Data, Packed, TextSource, Vector, search_below};
55
56mod distinct;
57pub mod graph;
58pub mod host;
59mod prepare;
60pub mod section;
61pub mod stats;
62mod zones;
63
64pub use prepare::{Merged, Paged, Prepared, Preparer};
65pub use section::Section;
66pub use zones::{Common, Stripes, ascending, distincts};
67
68const MAGIC: &[u8; 8] = b"RUDBNV10";
69const DIRECTORY: &[u8; 8] = b"RUDBDI10";
70const CATALOG: &[u8; 8] = b"RUDBCA10";
71const NONZERO_COUNTS: &[u8; 8] = b"RUDBNZ10";
72const AGGREGATE_SUMS: &[u8; 8] = b"RUDBAG10";
73const FORMAT: u32 = 28;
74
75const READABLE: &[u32] = &[22, 23, 24, 25, 26, 27, FORMAT];
104
105const HEADER: u64 = 80;
106const SLOT_BYTES: usize = 28;
107const MAX_PAGE: usize = 256 * 1024 * 1024;
108const MAX_DIRECTORY: usize = 128 * 1024 * 1024;
109const FREQUENCIES_V2: &[u8; 8] = b"RUDBFQ2\0";
110const FREQUENCIES: &[u8; 8] = b"RUDBFQ3\0";
111const FREQUENCY_TEXTS: &[u8; 8] = b"RUDBFT1\0";
119const HOST_GROUPS: &[u8; 8] = b"RUDBHG1\0";
121const PAIR_FREQUENCIES: &[u8; 8] = b"RUDBPF1\0";
127const CLUSTERING: &[u8; 8] = b"RUDBCL1\0";
142const SECTIONS: &[u8; 8] = b"RUDBSE1\0";
150const DICTIONARY_PAYLOADS: &[u8; 8] = b"RUDBDP1\0";
158
159const MAX_SECTIONS: usize = 4096;
166const FREQUENCY_CANDIDATES: usize = 32_768;
167const FREQUENCY_ENTRIES: usize = 512;
168const FREQUENCY_BUILD_RANK: usize = 10;
169const FREQUENCY_ORDINALS: usize = 131_072;
170const MAX_PAIR_FREQUENCIES: usize = 1024;
171const FREQUENCY_TEXT_BUDGET: usize = 1024 * 1024;
176const MAX_FREQUENCY_WORKERS: usize = 32;
183
184fn close_workers() -> usize {
186 std::thread::available_parallelism().map_or(1, usize::from).min(MAX_FREQUENCY_WORKERS)
187}
188
189const MAX_ENCODE_WORKERS: usize = 32;
196
197const SIEVE_BUDGET: usize = 8 * 1024;
205
206const PART_BOUND_BYTES: usize = 24;
215
216fn io(error: std::io::Error) -> Error {
217 Error::io(error.to_string())
218}
219
220fn invalid(message: &str) -> Error {
221 Error::invalid_input(format!("invalid rudb native file: {message}"))
222}
223
224fn sum(counts: impl Iterator<Item = u64>) -> u64 {
226 counts.fold(0, u64::saturating_add)
227}
228
229fn span_bytes(spans: &[Span], at: usize) -> u64 {
231 spans.get(at).map_or(0, |span| u64::from(span.length))
232}
233
234fn page_bytes(pages: &[Option<Page>], at: usize) -> u64 {
236 pages.get(at).and_then(Option::as_ref).map_or(0, Page::bytes)
237}
238
239fn dictionary_bytes(table: &Table, at: usize) -> u64 {
241 page_bytes(&table.dictionaries, at)
242 .saturating_add(table.dictionary_payloads.get(at).copied().unwrap_or(0))
243}
244
245fn checksum(bytes: &[u8]) -> u64 {
255 seeded_checksum(bytes, 0)
256}
257
258#[must_use]
265pub fn content_name(bytes: &[u8]) -> u128 {
266 let seed = u64::from(FORMAT);
267 u128::from(seeded_checksum(bytes, seed)) << 64 | u128::from(seeded_checksum(bytes, !seed))
268}
269
270#[derive(Debug, Clone)]
276pub struct ContentNamer {
277 seeds: [u64; 2],
278 lanes: [[u64; 4]; 2],
279 held: [u8; 32],
280 filled: usize,
281 length: u64,
282}
283
284impl Default for ContentNamer {
285 fn default() -> Self {
286 let seed = u64::from(FORMAT);
287 let seeds = [seed, !seed];
288 let lanes = seeds.map(|seed| {
289 [
290 seed.wrapping_add(XXH_P1).wrapping_add(XXH_P2),
291 seed.wrapping_add(XXH_P2),
292 seed,
293 seed.wrapping_sub(XXH_P1),
294 ]
295 });
296 Self { seeds, lanes, held: [0; 32], filled: 0, length: 0 }
297 }
298}
299
300impl ContentNamer {
301 pub fn update(&mut self, mut bytes: &[u8]) {
303 self.length += bytes.len() as u64;
304 if self.filled > 0 {
305 let take = (32 - self.filled).min(bytes.len());
306 self.held[self.filled..self.filled + take].copy_from_slice(&bytes[..take]);
307 self.filled += take;
308 bytes = &bytes[take..];
309 if self.filled < 32 {
310 return;
311 }
312 let block = self.held;
313 self.lanes.iter_mut().for_each(|lanes| checksum_block(lanes, &block));
314 self.filled = 0;
315 }
316 let mut blocks = bytes.chunks_exact(32);
317 for block in blocks.by_ref() {
318 self.lanes.iter_mut().for_each(|lanes| checksum_block(lanes, block));
319 }
320 let rest = blocks.remainder();
321 self.held[..rest.len()].copy_from_slice(rest);
322 self.filled = rest.len();
323 }
324
325 #[must_use]
327 pub fn finish(&self) -> u128 {
328 let rest = &self.held[..self.filled];
329 let [first, second] = [0, 1].map(|at| {
330 if self.length < 32 {
331 checksum_tail(self.seeds[at].wrapping_add(XXH_P5).wrapping_add(self.length), rest)
332 } else {
333 finish_checksum(self.lanes[at], rest, self.length)
334 }
335 });
336 u128::from(first) << 64 | u128::from(second)
337 }
338}
339
340fn seeded_checksum(bytes: &[u8], seed: u64) -> u64 {
349 let mut blocks = bytes.chunks_exact(32);
352 let rest = blocks.remainder();
353 if bytes.len() < 32 {
354 return checksum_tail(seed.wrapping_add(XXH_P5).wrapping_add(bytes.len() as u64), rest);
355 }
356 let mut lanes = [
357 seed.wrapping_add(XXH_P1).wrapping_add(XXH_P2),
358 seed.wrapping_add(XXH_P2),
359 seed,
360 seed.wrapping_sub(XXH_P1),
361 ];
362 for block in blocks.by_ref() {
363 checksum_block(&mut lanes, block);
364 }
365 finish_checksum(lanes, rest, bytes.len() as u64)
366}
367
368const XXH_P1: u64 = 11_400_714_785_074_694_791;
369const XXH_P2: u64 = 14_029_467_366_897_019_727;
370const XXH_P3: u64 = 1_609_587_929_392_839_161;
371const XXH_P4: u64 = 9_650_029_242_287_828_579;
372const XXH_P5: u64 = 2_870_177_450_012_600_261;
373
374fn checksum_round(state: u64, word: u64) -> u64 {
375 state.wrapping_add(word.wrapping_mul(XXH_P2)).rotate_left(31).wrapping_mul(XXH_P1)
376}
377
378fn checksum_word(chunk: &[u8]) -> u64 {
379 u64::from_le_bytes(chunk.try_into().expect("eight checksum bytes"))
380}
381
382fn checksum_block(lanes: &mut [u64; 4], block: &[u8]) {
384 for (lane, chunk) in lanes.iter_mut().zip(block.chunks_exact(8)) {
385 *lane = checksum_round(*lane, checksum_word(chunk));
386 }
387}
388
389fn finish_checksum(lanes: [u64; 4], rest: &[u8], length: u64) -> u64 {
391 let merge = |state: u64, lane: u64| {
392 (state ^ checksum_round(0, lane)).wrapping_mul(XXH_P1).wrapping_add(XXH_P4)
393 };
394 let [one, two, three, four] = lanes;
395 let combined = one
396 .rotate_left(1)
397 .wrapping_add(two.rotate_left(7))
398 .wrapping_add(three.rotate_left(12))
399 .wrapping_add(four.rotate_left(18));
400 let hash = merge(merge(merge(merge(combined, one), two), three), four);
401 checksum_tail(hash.wrapping_add(length), rest)
402}
403
404fn checksum_tail(mut hash: u64, mut rest: &[u8]) -> u64 {
406 let mut words = rest.chunks_exact(8);
407 for chunk in words.by_ref() {
408 hash ^= checksum_round(0, checksum_word(chunk));
409 hash = hash.rotate_left(27).wrapping_mul(XXH_P1).wrapping_add(XXH_P4);
410 }
411 rest = words.remainder();
412 if rest.len() >= 4 {
413 let (head, tail) = rest.split_at(4);
414 let quarter = u32::from_le_bytes(head.try_into().expect("four checksum bytes"));
415 hash ^= u64::from(quarter).wrapping_mul(XXH_P1);
416 hash = hash.rotate_left(23).wrapping_mul(XXH_P2).wrapping_add(XXH_P3);
417 rest = tail;
418 }
419 for &byte in rest {
420 hash ^= u64::from(byte).wrapping_mul(XXH_P5);
421 hash = hash.rotate_left(11).wrapping_mul(XXH_P1);
422 }
423 hash ^= hash >> 33;
424 hash = hash.wrapping_mul(XXH_P2);
425 hash ^= hash >> 29;
426 hash = hash.wrapping_mul(XXH_P3);
427 hash ^ (hash >> 32)
428}
429
430fn file_checksum(file: &File, offset: u64, length: usize) -> Result<u64> {
436 if length < 32 {
437 let mut bytes = vec![0; length];
438 read_at(file, offset, &mut bytes)?;
439 return Ok(checksum(&bytes));
440 }
441 let mut lanes = [XXH_P1.wrapping_add(XXH_P2), XXH_P2, 0, 0_u64.wrapping_sub(XXH_P1)];
442 let mut buffer = vec![0; DIRECTORY_WINDOW.min(length)];
443 let mut kept = 0;
444 let mut read = 0;
445 while read < length {
446 let want = (buffer.len() - kept).min(length - read);
447 read_at(file, offset + read as u64, &mut buffer[kept..kept + want])?;
448 read += want;
449 let filled = kept + want;
450 let whole = filled / 32 * 32;
451 for block in buffer[..whole].chunks_exact(32) {
452 checksum_block(&mut lanes, block);
453 }
454 buffer.copy_within(whole..filled, 0);
455 kept = filled - whole;
456 }
457 Ok(finish_checksum(lanes, &buffer[..kept], length as u64))
458}
459
460#[derive(Debug, Clone, Copy)]
461struct Slot {
462 offset: u64,
463 length: u32,
464 generation: u64,
465 hash: u64,
466}
467
468impl Slot {
469 fn bytes(self) -> [u8; SLOT_BYTES] {
470 let mut result = [0; SLOT_BYTES];
471 result[..8].copy_from_slice(&self.offset.to_le_bytes());
472 result[8..12].copy_from_slice(&self.length.to_le_bytes());
473 result[12..20].copy_from_slice(&self.generation.to_le_bytes());
474 result[20..28].copy_from_slice(&self.hash.to_le_bytes());
475 result
476 }
477
478 fn read(bytes: &[u8]) -> Self {
479 Self {
480 offset: u64::from_le_bytes(bytes[..8].try_into().expect("eight bytes")),
481 length: u32::from_le_bytes(bytes[8..12].try_into().expect("four bytes")),
482 generation: u64::from_le_bytes(bytes[12..20].try_into().expect("eight bytes")),
483 hash: u64::from_le_bytes(bytes[20..28].try_into().expect("eight bytes")),
484 }
485 }
486}
487
488#[derive(Debug, Clone, Copy)]
489struct Page {
490 offset: u64,
491 length: u32,
492 hash: u64,
493}
494
495impl Page {
496 fn bytes(&self) -> u64 {
498 u64::from(self.length)
499 }
500}
501
502#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
503enum FrequencyValue {
504 Null,
505 Integer(i128),
506 Code(u32),
507}
508
509type FrequencyMap<V> = HashMap<u64, V, Spread>;
515
516#[derive(Debug, Default, Clone, Copy)]
518struct Spread;
519
520impl std::hash::BuildHasher for Spread {
521 type Hasher = SpreadHasher;
522
523 fn build_hasher(&self) -> SpreadHasher {
524 SpreadHasher(0)
525 }
526}
527
528#[derive(Debug)]
535struct SpreadHasher(u64);
536
537impl SpreadHasher {
538 fn mix(&mut self, word: u64) {
539 let product = u128::from(self.0 ^ word) * 0x9E37_79B9_7F4A_7C15_u128;
540 self.0 = (product as u64) ^ ((product >> 64) as u64);
541 }
542}
543
544impl std::hash::Hasher for SpreadHasher {
545 fn write(&mut self, bytes: &[u8]) {
546 for part in bytes.chunks(8) {
547 let mut word = [0; 8];
548 word[..part.len()].copy_from_slice(part);
549 self.mix(u64::from_le_bytes(word));
550 }
551 }
552
553 fn write_u32(&mut self, value: u32) {
554 self.mix(u64::from(value));
555 }
556
557 fn write_u64(&mut self, value: u64) {
558 self.mix(value);
559 }
560
561 fn write_i128(&mut self, value: i128) {
562 self.mix(value as u64);
563 self.mix((value >> 64) as u64);
564 }
565
566 fn write_isize(&mut self, value: isize) {
567 self.mix(value as u64);
568 }
569
570 fn finish(&self) -> u64 {
571 self.0
572 }
573}
574
575#[derive(Debug, Clone)]
576struct FrequencyEntry {
577 value: FrequencyValue,
578 count: u64,
579}
580
581#[derive(Debug, Clone)]
586struct FrequencySummary {
587 entries: Vec<FrequencyEntry>,
588 omitted_max: u64,
589 ordinals: Vec<u64>,
590 ordinal_entries: Vec<u16>,
591}
592
593#[derive(Debug, Clone)]
594struct PairFrequencyEntry {
595 first_entry: u16,
596 second: Option<u32>,
597 count: u64,
598}
599
600#[derive(Debug, Clone)]
606struct PairFrequencySummary {
607 first: u16,
608 second: u16,
609 entries: Vec<PairFrequencyEntry>,
610 omitted_max: u64,
611}
612
613#[derive(Debug, Clone)]
621enum Frequencies {
622 Held(FrequencySummary),
623 Stored {
626 span: Span,
627 values: bool,
628 },
629}
630
631#[derive(Debug, Clone)]
636pub struct FrequencyPrefix {
637 pub entries: Vec<(Value, u64)>,
639 pub omitted_max: u64,
641}
642
643#[derive(Debug, Clone, PartialEq)]
645pub struct FrequencyOccurrences {
646 pub omitted_max: u64,
648 pub ordinals: Vec<u64>,
650 pub anchors: Vec<Value>,
652 pub anchor_indices: Vec<u16>,
654}
655
656pub type PairFrequencyCounts = Vec<(Vec<Value>, u64)>;
658
659#[derive(Debug, Clone, Copy, Default)]
666struct Span {
667 offset: u64,
668 length: u32,
669}
670
671#[derive(Debug, Clone, Default)]
679struct Pages {
680 columns: usize,
681 held: Box<[StripePage]>,
682}
683
684#[derive(Debug, Clone, Copy)]
686struct StripePage {
687 offset: u64,
688 hash: u64,
689 length: u32,
690 column: u32,
691}
692
693impl Pages {
694 fn from_slots(slots: Vec<Option<Page>>) -> Result<Self> {
696 let mut held = Vec::with_capacity(slots.iter().flatten().count());
697 for (column, page) in slots.iter().enumerate() {
698 if let Some(page) = page {
699 let column =
700 u32::try_from(column).map_err(|_| invalid("too many columns for a page"))?;
701 held.push(StripePage {
702 offset: page.offset,
703 hash: page.hash,
704 length: page.length,
705 column,
706 });
707 }
708 }
709 Ok(Self { columns: slots.len(), held: held.into_boxed_slice() })
710 }
711
712 fn get(&self, column: usize) -> Option<Page> {
714 let at = self.held.binary_search_by_key(&column, |placed| placed.column as usize).ok()?;
715 let placed = self.held[at];
716 Some(Page { offset: placed.offset, length: placed.length, hash: placed.hash })
717 }
718
719 fn slots(&self) -> impl Iterator<Item = Option<Page>> + '_ {
721 (0..self.columns).map(|column| self.get(column))
722 }
723
724 fn bytes(&self, column: usize) -> u64 {
726 self.get(column).map_or(0, |page| page.bytes())
727 }
728}
729
730#[derive(Debug, Clone)]
732pub struct Stripe {
733 rows: usize,
734 parts: Vec<u32>,
737 index: Span,
741 pages: Vec<Span>,
742 memberships: Pages,
743 sieves: Pages,
746 part_ranges: Pages,
757 zone: Zone,
758}
759
760impl Stripe {
761 #[must_use]
763 pub fn rows(&self) -> usize {
764 self.rows
765 }
766
767 #[must_use]
769 pub fn parts(&self) -> usize {
770 self.parts.len()
771 }
772
773 #[must_use]
779 pub fn zone(&self) -> &Zone {
780 &self.zone
781 }
782}
783
784#[derive(Debug, Clone)]
786pub struct Table {
787 name: String,
788 fields: Vec<Field>,
789 stripes: Vec<Stripe>,
790 rows: usize,
791 dictionaries: Vec<Option<Page>>,
792 dictionary_payloads: Vec<u64>,
798 frequencies: Vec<Option<Frequencies>>,
799 pair_frequencies: Vec<PairFrequencySummary>,
800 frequency_texts: Vec<Vec<Option<Vec<u8>>>>,
805 host_groups: Option<host::HostSummary>,
807 distincts: Vec<Option<u64>>,
817 clustering: Option<Clustering>,
825 generation: u64,
839 sections: Vec<Section>,
846}
847
848impl Table {
849 #[must_use]
851 pub fn name(&self) -> &str {
852 &self.name
853 }
854
855 #[must_use]
857 pub fn fields(&self) -> &[Field] {
858 &self.fields
859 }
860
861 #[must_use]
863 pub fn rows(&self) -> usize {
864 self.rows
865 }
866
867 #[must_use]
869 pub fn stripes(&self) -> &[Stripe] {
870 &self.stripes
871 }
872
873 #[must_use]
875 pub fn clustering(&self) -> Option<&Clustering> {
876 self.clustering.as_ref()
877 }
878
879 #[must_use]
884 pub fn generation(&self) -> u64 {
885 self.generation
886 }
887
888 #[must_use]
895 pub fn sections(&self) -> &[Section] {
896 &self.sections
897 }
898}
899
900#[derive(Debug, Clone)]
912struct Entry {
913 name: String,
914 fields: Vec<Field>,
915 rows: usize,
916 directory: Page,
918 nonzero: Vec<Option<u64>>,
920 aggregates: Vec<Option<(i128, u64)>>,
922}
923
924#[derive(Debug, Clone, PartialEq, Eq)]
937pub struct ViewEntry {
938 pub name: String,
940 pub sql: String,
942 pub statement: String,
944 pub aliases: Vec<String>,
946 pub columns: Vec<Field>,
948}
949
950#[derive(Debug, Clone)]
952pub struct ColumnLayout {
953 pub name: String,
955 pub kind: String,
957 pub pages: u64,
959 pub memberships: u64,
961 pub sieves: u64,
963 pub part_ranges: u64,
965 pub dictionary: u64,
967}
968
969impl ColumnLayout {
970 #[must_use]
972 pub fn total(&self) -> u64 {
973 self.pages
974 .saturating_add(self.memberships)
975 .saturating_add(self.sieves)
976 .saturating_add(self.part_ranges)
977 .saturating_add(self.dictionary)
978 }
979}
980
981#[derive(Debug, Clone)]
992pub struct Layout {
993 pub file: u64,
995 pub rows: usize,
997 pub stripes: usize,
999 pub parts: usize,
1001 pub columns: Vec<ColumnLayout>,
1003 pub indexes: u64,
1006 pub directory: u64,
1008 pub header: u64,
1010}
1011
1012impl Layout {
1013 #[must_use]
1015 pub fn columns_total(&self) -> u64 {
1016 self.columns.iter().map(ColumnLayout::total).fold(0, u64::saturating_add)
1017 }
1018
1019 #[must_use]
1025 pub fn unaccounted(&self) -> u64 {
1026 self.file
1027 .saturating_sub(self.columns_total())
1028 .saturating_sub(self.indexes)
1029 .saturating_sub(self.directory)
1030 .saturating_sub(self.header)
1031 }
1032}
1033
1034#[derive(Debug, Clone)]
1045pub struct StoredPart {
1046 pub stripe: usize,
1048 pub part: usize,
1050 pub row: usize,
1052 pub rows: usize,
1054 pub encoding: String,
1056 pub bytes: u64,
1058 pub page: u64,
1060 pub offset: u64,
1062 pub low: Option<Value>,
1064 pub high: Option<Value>,
1066 pub nulls: Option<usize>,
1068}
1069
1070const DICTIONARY_CHECK_SEED: u64 = 11_400_714_819_323_198_485;
1077
1078#[derive(Debug)]
1102struct GlobalDictionary {
1103 primary: HashMap<u64, u32>,
1104 collisions: HashMap<u64, Vec<u32>>,
1105 checks: Vec<u64>,
1107 ends: Vec<u32>,
1109 counts: Vec<u64>,
1110 nulls: u64,
1111 filling: Vec<u8>,
1113 grams: Vec<[u8; TEXT_GRAM_BYTES]>,
1115 waiting: Vec<(usize, Vec<u8>)>,
1120 sample: Vec<(usize, Vec<u8>)>,
1126 stride: usize,
1128 shape: Option<chooser::Settled>,
1130 settled: usize,
1132 blocks: Vec<Vec<u8>>,
1138 placed: Vec<Placed>,
1140}
1141
1142#[derive(Debug, Clone, Copy)]
1144struct Placed {
1145 start: u64,
1146 length: u64,
1147 hash: u64,
1148}
1149
1150type RankedDictionary = (Vec<(u64, u32)>, Vec<u8>, Vec<u64>);
1152
1153impl GlobalDictionary {
1154 fn new() -> Self {
1155 Self {
1156 primary: HashMap::new(),
1157 collisions: HashMap::new(),
1158 checks: Vec::new(),
1159 ends: Vec::new(),
1160 counts: Vec::new(),
1161 nulls: 0,
1162 filling: Vec::new(),
1163 grams: Vec::new(),
1164 waiting: Vec::new(),
1165 sample: Vec::new(),
1166 stride: 1,
1167 shape: None,
1168 settled: 0,
1169 blocks: Vec::new(),
1170 placed: Vec::new(),
1171 }
1172 }
1173
1174 fn values(&self) -> usize {
1176 self.ends.len()
1177 }
1178
1179 fn encoded(&self) -> usize {
1181 self.placed.len() + self.blocks.len()
1182 }
1183
1184 #[cfg(test)]
1185 fn code(&mut self, text: &str) -> Result<u32> {
1186 let bytes = text.as_bytes();
1187 self.code_hashed(bytes, checksum(bytes), seeded_checksum(bytes, DICTIONARY_CHECK_SEED))
1188 }
1189
1190 fn code_hashed(&mut self, text: &[u8], hash: u64, check: u64) -> Result<u32> {
1196 if let Some(&code) = self.primary.get(&hash) {
1197 if self.checks.get(code as usize) == Some(&check) {
1198 return Ok(code);
1199 }
1200 if let Some(codes) = self.collisions.get(&hash) {
1201 if let Some(code) =
1202 codes.iter().copied().find(|&code| self.checks[code as usize] == check)
1203 {
1204 return Ok(code);
1205 }
1206 }
1207 let code = self.insert(text, check)?;
1208 self.collisions.entry(hash).or_default().push(code);
1209 return Ok(code);
1210 }
1211 let code = self.insert(text, check)?;
1212 self.primary.insert(hash, code);
1213 Ok(code)
1214 }
1215
1216 fn insert(&mut self, text: &[u8], check: u64) -> Result<u32> {
1217 let code = u32::try_from(self.ends.len())
1218 .map_err(|_| invalid("global dictionary has too many values"))?;
1219 self.filling.extend_from_slice(text);
1220 self.ends.push(
1221 u32::try_from(self.filling.len())
1222 .map_err(|_| invalid("a global dictionary value exceeds 4 GiB"))?,
1223 );
1224 self.checks.push(check);
1225 self.counts.push(0);
1226 if self.ends.len() % TEXT_PAYLOAD_VALUES == 0 {
1227 self.seal();
1228 }
1229 Ok(code)
1230 }
1231
1232 fn seal(&mut self) {
1238 let at = self.ends.len().div_ceil(TEXT_PAYLOAD_VALUES) - 1;
1239 let bytes = std::mem::take(&mut self.filling);
1240 let mut grams = [0_u8; TEXT_GRAM_BYTES];
1241 for value in self.slices(at, &bytes) {
1242 for gram in value.windows(4) {
1243 for bit in gram_bits(gram) {
1244 grams[bit / 8] |= 1 << (bit % 8);
1245 }
1246 }
1247 }
1248 self.grams.push(grams);
1249 if at % self.stride == 0 {
1250 self.sample.push((at, bytes.clone()));
1251 if self.sample.len() > PAYLOAD_SAMPLE_BLOCKS {
1252 self.stride *= 2;
1253 let stride = self.stride;
1254 self.sample.retain(|(at, _)| at % stride == 0);
1255 }
1256 }
1257 self.waiting.push((at, bytes));
1258 }
1259
1260 fn slices<'a>(&self, at: usize, bytes: &'a [u8]) -> Vec<&'a [u8]> {
1262 let first = at * TEXT_PAYLOAD_VALUES;
1263 let last = (first + TEXT_PAYLOAD_VALUES).min(self.ends.len());
1264 let mut out = Vec::with_capacity(last.saturating_sub(first));
1265 let mut from = 0;
1266 for value in first..last {
1267 let to = self.ends[value] as usize;
1268 out.push(&bytes[from..to]);
1269 from = to;
1270 }
1271 out
1272 }
1273
1274 fn settle(&mut self) -> Result<()> {
1282 if self.sample.len() < PAYLOAD_SAMPLE_BLOCKS {
1283 return Ok(());
1284 }
1285 let complete = self.ends.len() / TEXT_PAYLOAD_VALUES;
1286 if self.shape.is_some() && complete < self.settled.saturating_mul(4) {
1287 return Ok(());
1288 }
1289 let sample =
1290 self.sample.iter().map(|(at, bytes)| self.slices(*at, bytes)).collect::<Vec<_>>();
1291 self.shape = Some(settle_shape(&sample)?);
1292 self.settled = complete;
1293 Ok(())
1294 }
1295
1296 fn seal_rest(&mut self) {
1298 if self.ends.len() % TEXT_PAYLOAD_VALUES != 0 {
1301 self.seal();
1302 }
1303 }
1304
1305 fn encode_waiting(&self, at: usize) -> Result<Vec<u8>> {
1308 let (block, bytes) = &self.waiting[at];
1309 let values = self.slices(*block, bytes);
1310 match &self.shape {
1311 Some(shape) => string::encode_with(&values, shape),
1312 None => string::encode(&values),
1313 }
1314 }
1315
1316 #[cfg(test)]
1318 fn finish_blocks(&mut self) -> Result<()> {
1319 self.seal_rest();
1320 let made = (0..self.waiting.len())
1321 .map(|at| self.encode_waiting(at))
1322 .collect::<Result<Vec<_>>>()?;
1323 for ((at, _), bytes) in std::mem::take(&mut self.waiting).into_iter().zip(made) {
1324 if self.encoded() != at {
1325 return Err(Error::internal("a dictionary block was encoded out of order"));
1326 }
1327 self.blocks.push(bytes);
1328 }
1329 Ok(())
1330 }
1331
1332 fn decoded(&self, file: Option<&File>) -> Result<(Vec<u8>, Vec<u64>)> {
1350 let count = self.placed.len() + self.blocks.len();
1351 if count != self.values().div_ceil(TEXT_PAYLOAD_VALUES) {
1352 return Err(invalid("global dictionary blocks do not cover its values"));
1353 }
1354 let mut bases = Vec::with_capacity(count);
1355 let mut total = 0_usize;
1356 for block in 0..count {
1357 bases.push(total as u64);
1358 let last = ((block + 1) * TEXT_PAYLOAD_VALUES).min(self.values()) - 1;
1359 total = total
1360 .checked_add(self.ends[last] as usize)
1361 .ok_or_else(|| invalid("global dictionary does not fit in memory"))?;
1362 }
1363 let mut flat = vec![0_u8; total];
1364 let mut outs = Vec::with_capacity(count);
1365 let mut rest = flat.as_mut_slice();
1366 for block in 0..count {
1367 let end = bases.get(block + 1).map_or(total, |&base| base as usize);
1368 let (out, after) = rest.split_at_mut(end - bases[block] as usize);
1369 outs.push((block, out));
1370 rest = after;
1371 }
1372 let one = |run: &mut [(usize, &mut [u8])]| -> Result<()> {
1373 let mut stored = Vec::new();
1374 for (block, out) in run {
1375 let encoded = match self.placed.get(*block) {
1376 Some(place) => {
1377 let file = file.ok_or_else(|| {
1378 Error::internal("a written dictionary block has no file")
1379 })?;
1380 let length = usize::try_from(place.length).map_err(|_| {
1381 invalid("global dictionary block does not fit in memory")
1382 })?;
1383 stored.resize(length, 0);
1384 read_at(file, place.start, &mut stored)?;
1385 if checksum(&stored) != place.hash {
1386 return Err(invalid(
1387 "a global dictionary block did not read back as written",
1388 ));
1389 }
1390 stored.as_slice()
1391 }
1392 None => &self.blocks[*block - self.placed.len()],
1393 };
1394 let decoded = string::decode_flat(encoded)?;
1395 if decoded.bytes().len() != out.len() {
1396 return Err(invalid(
1397 "a global dictionary block is not the length its ends say",
1398 ));
1399 }
1400 out.copy_from_slice(decoded.bytes());
1401 }
1402 Ok(())
1403 };
1404 let workers = close_workers().min(count / 16).max(1);
1407 if workers <= 1 {
1408 one(&mut outs)?;
1409 } else {
1410 let per = count.div_ceil(workers);
1411 std::thread::scope(|scope| {
1412 outs.chunks_mut(per)
1413 .map(|run| scope.spawn(|| one(run)))
1414 .collect::<Vec<_>>()
1415 .into_iter()
1416 .try_for_each(|handle| {
1417 handle.join().map_err(|_| {
1418 Error::internal("a global dictionary decode worker panicked")
1419 })?
1420 })
1421 })?;
1422 }
1423 drop(outs);
1424 Ok((flat, bases))
1425 }
1426
1427 fn value_span(ends: &[u32], bases: &[u64], code: usize) -> (usize, usize) {
1432 let Some(&base) = bases.get(code / TEXT_PAYLOAD_VALUES) else { return (0, 0) };
1433 let Some(&end) = ends.get(code) else { return (0, 0) };
1434 let base = base as usize;
1435 let from = if code % TEXT_PAYLOAD_VALUES == 0 { 0 } else { ends[code - 1] as usize };
1436 (base + from, base + end as usize)
1437 }
1438
1439 fn ranked_with_values(&self, file: Option<&File>) -> Result<RankedDictionary> {
1459 let (flat, bases) = self.decoded(file)?;
1460 let value = |code: u32| {
1461 let (from, to) = Self::value_span(&self.ends, &bases, code as usize);
1462 flat.get(from..to).unwrap_or_default()
1463 };
1464 let mut codes = (0..self.values() as u32).collect::<Vec<_>>();
1465 sort_by_value_across(&mut codes, value, close_workers());
1466 let order = codes.into_iter().map(|code| (head(value(code)), code)).collect();
1467 Ok((order, flat, bases))
1468 }
1469
1470 #[cfg(test)]
1471 fn ranked(&self, file: Option<&File>) -> Result<Vec<(u64, u32)>> {
1472 self.ranked_with_values(file).map(|(order, _, _)| order)
1473 }
1474}
1475
1476#[derive(Debug)]
1484pub struct Writer {
1485 file: File,
1486 at: u64,
1494 table: Table,
1495 generation: u64,
1496 order: Vec<((u64, u64), (u64, u64))>,
1499 next_order: u64,
1500 dictionaries: Vec<Option<GlobalDictionary>>,
1501 coded: Arc<[AtomicBool]>,
1504 gathers: Vec<Option<stats::Gather>>,
1510 pending: Vec<PendingChunk>,
1511 closed: Vec<Entry>,
1513 views: Vec<ViewEntry>,
1518 profile: Option<Arc<LoadProfile>>,
1524}
1525
1526#[derive(Debug)]
1534struct PendingChunk {
1535 order: (u64, u64),
1536 chunk: Chunk,
1537}
1538
1539#[derive(Debug, Clone, Copy)]
1545struct Part {
1546 order: (u64, u64),
1547 rows: usize,
1548 footprint: usize,
1549}
1550
1551impl Part {
1552 fn of(pending: &PendingChunk) -> Self {
1553 Self {
1554 order: pending.order,
1555 rows: pending.chunk.len(),
1556 footprint: pending.chunk.footprint(),
1557 }
1558 }
1559}
1560
1561#[derive(Debug)]
1567struct ColumnStripe {
1568 pages: Vec<Vec<u8>>,
1569 codes: Vec<Option<Vec<u32>>>,
1570 sieves: Vec<Option<Sieve>>,
1571 ranges: Vec<Range>,
1572}
1573
1574fn weight(ty: &LogicalType) -> usize {
1582 match ty {
1583 LogicalType::Varchar | LogicalType::Blob | LogicalType::Bit => 64,
1584 LogicalType::HugeInt
1585 | LogicalType::UHugeInt
1586 | LogicalType::Uuid
1587 | LogicalType::Interval => 16,
1588 LogicalType::BigInt
1589 | LogicalType::UBigInt
1590 | LogicalType::Timestamp
1591 | LogicalType::Time
1592 | LogicalType::TimeTz
1593 | LogicalType::TimestampTz
1594 | LogicalType::TimestampS
1595 | LogicalType::TimestampMs
1596 | LogicalType::TimestampNs
1597 | LogicalType::Double
1598 | LogicalType::Decimal { .. } => 8,
1599 LogicalType::Integer | LogicalType::UInteger | LogicalType::Date | LogicalType::Float => 4,
1600 LogicalType::SmallInt | LogicalType::USmallInt => 2,
1601 _ => 1,
1602 }
1603}
1604
1605pub const STRIPE_PARTS: usize = 64;
1612
1613const DICTIONARY_DECIDE_ROWS: usize = 4_096;
1621
1622const DICTIONARY_DISTINCT_IN_TEN: usize = 9;
1638
1639const INDEX_ENTRY: usize = size_of::<u32>() + size_of::<u64>();
1641
1642fn index_section(parts: usize) -> Result<usize> {
1644 parts
1645 .checked_mul(INDEX_ENTRY)
1646 .and_then(|bytes| bytes.checked_add(size_of::<u64>()))
1647 .ok_or_else(|| invalid("index page length overflow"))
1648}
1649
1650impl Writer {
1651 pub fn open(
1670 path: impl AsRef<Path>,
1671 name: impl Into<String>,
1672 fields: Vec<Field>,
1673 ) -> Result<Self> {
1674 for field in &fields {
1675 type_tag(&field.ty)?;
1676 }
1677 let name = name.into();
1678 let path = path.as_ref();
1679 let (_, size, slot, bytes, _) = slot_bytes(path)?;
1680 let (mut closed, views) = decode_catalog(&bytes, size)?;
1681 if let Some(at) = closed.iter().position(|held| held.name == name) {
1692 if closed[at].rows > 0 {
1693 return Err(invalid("two tables in one native file have the same name"));
1694 }
1695 closed.remove(at);
1696 }
1697 let generation = slot
1702 .generation
1703 .checked_add(1)
1704 .ok_or_else(|| invalid("native file generation overflow"))?;
1705 let file = OpenOptions::new().write(true).read(true).open(path).map_err(io)?;
1706 Ok(Self {
1707 file,
1708 at: size,
1711 dictionaries: fields
1712 .iter()
1713 .map(|field| (field.ty == LogicalType::Varchar).then(GlobalDictionary::new))
1714 .collect(),
1715 coded: fields
1716 .iter()
1717 .map(|field| AtomicBool::new(field.ty == LogicalType::Varchar))
1718 .collect(),
1719 gathers: fields.iter().map(|field| stats::Gather::new(&field.ty, generation)).collect(),
1720 table: Table {
1721 name,
1722 dictionaries: vec![None; fields.len()],
1723 dictionary_payloads: Vec::new(),
1724 distincts: vec![None; fields.len()],
1725 fields,
1726 stripes: Vec::new(),
1727 rows: 0,
1728 frequencies: Vec::new(),
1729 pair_frequencies: Vec::new(),
1730 frequency_texts: Vec::new(),
1731 host_groups: None,
1732 clustering: None,
1733 generation,
1734 sections: Vec::new(),
1735 },
1736 generation,
1737 order: Vec::new(),
1738 next_order: 0,
1739 pending: Vec::with_capacity(STRIPE_PARTS),
1740 closed,
1741 views,
1742 profile: None,
1743 })
1744 }
1745
1746 pub fn create(
1752 path: impl AsRef<Path>,
1753 name: impl Into<String>,
1754 fields: Vec<Field>,
1755 ) -> Result<Self> {
1756 for field in &fields {
1757 type_tag(&field.ty)?;
1758 }
1759 let file =
1760 OpenOptions::new().write(true).read(true).create_new(true).open(path).map_err(io)?;
1761 let mut header = [0; HEADER as usize];
1762 header[..8].copy_from_slice(MAGIC);
1763 header[8..12].copy_from_slice(&FORMAT.to_le_bytes());
1764 write_at(&file, 0, &header)?;
1765 Ok(Self {
1766 file,
1767 at: HEADER,
1768 dictionaries: fields
1769 .iter()
1770 .map(|field| (field.ty == LogicalType::Varchar).then(GlobalDictionary::new))
1771 .collect(),
1772 coded: fields
1773 .iter()
1774 .map(|field| AtomicBool::new(field.ty == LogicalType::Varchar))
1775 .collect(),
1776 gathers: fields.iter().map(|field| stats::Gather::new(&field.ty, 1)).collect(),
1777 table: Table {
1778 name: name.into(),
1779 dictionaries: vec![None; fields.len()],
1780 dictionary_payloads: Vec::new(),
1781 distincts: vec![None; fields.len()],
1782 fields,
1783 stripes: Vec::new(),
1784 rows: 0,
1785 frequencies: Vec::new(),
1786 pair_frequencies: Vec::new(),
1787 frequency_texts: Vec::new(),
1788 host_groups: None,
1789 clustering: None,
1790 generation: 1,
1791 sections: Vec::new(),
1792 },
1793 generation: 1,
1794 order: Vec::new(),
1795 next_order: 0,
1796 pending: Vec::with_capacity(STRIPE_PARTS),
1797 closed: Vec::new(),
1798 views: Vec::new(),
1799 profile: None,
1800 })
1801 }
1802
1803 pub fn empty(path: impl AsRef<Path>, views: &[ViewEntry]) -> Result<()> {
1825 let file =
1826 OpenOptions::new().write(true).read(true).create_new(true).open(path).map_err(io)?;
1827 let mut header = [0; HEADER as usize];
1828 header[..8].copy_from_slice(MAGIC);
1829 header[8..12].copy_from_slice(&FORMAT.to_le_bytes());
1830 write_at(&file, 0, &header)?;
1831 let catalog = encode_catalog(&[], views)?;
1832 write_at(&file, HEADER, &catalog)?;
1833 file.sync_all().map_err(io)?;
1837 let slot = Slot {
1838 offset: HEADER,
1839 length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
1840 generation: 1,
1841 hash: checksum(&catalog),
1842 };
1843 write_at(&file, slot_offset(1), &slot.bytes())?;
1844 file.sync_all().map_err(io)?;
1845 Ok(())
1846 }
1847
1848 pub fn next(mut self, name: impl Into<String>, fields: Vec<Field>) -> Result<Self> {
1859 for field in &fields {
1860 type_tag(&field.ty)?;
1861 }
1862 let name = name.into();
1863 let entry = self.close()?;
1864 if self.closed.iter().chain(std::iter::once(&entry)).any(|held| held.name == name) {
1865 return Err(invalid("two tables in one native file have the same name"));
1866 }
1867 let Self { file, at, generation, mut closed, views, .. } = self;
1868 closed.push(entry);
1869 Ok(Self {
1870 file,
1871 at,
1872 generation,
1873 closed,
1874 views,
1875 profile: None,
1876 dictionaries: fields
1877 .iter()
1878 .map(|field| (field.ty == LogicalType::Varchar).then(GlobalDictionary::new))
1879 .collect(),
1880 coded: fields
1881 .iter()
1882 .map(|field| AtomicBool::new(field.ty == LogicalType::Varchar))
1883 .collect(),
1884 gathers: fields.iter().map(|field| stats::Gather::new(&field.ty, generation)).collect(),
1885 table: Table {
1886 name,
1887 dictionaries: vec![None; fields.len()],
1888 dictionary_payloads: Vec::new(),
1889 distincts: vec![None; fields.len()],
1890 fields,
1891 stripes: Vec::new(),
1892 rows: 0,
1893 frequencies: Vec::new(),
1894 pair_frequencies: Vec::new(),
1895 frequency_texts: Vec::new(),
1896 host_groups: None,
1897 clustering: None,
1898 generation,
1899 sections: Vec::new(),
1900 },
1901 order: Vec::new(),
1902 next_order: 0,
1903 pending: Vec::with_capacity(STRIPE_PARTS),
1904 })
1905 }
1906
1907 #[must_use]
1917 pub fn with_views(mut self, views: Vec<ViewEntry>) -> Self {
1918 self.views = views;
1919 self
1920 }
1921
1922 #[must_use]
1928 pub fn with_profile(mut self, profile: Arc<LoadProfile>) -> Self {
1929 self.profile = Some(profile);
1930 self
1931 }
1932
1933 pub fn declare(mut self, clustering: Clustering) -> Result<Self> {
1948 self.table.clustering = Some(Clustering::new(
1951 clustering.columns().to_vec(),
1952 clustering.width(),
1953 &self.table.fields,
1954 )?);
1955 Ok(self)
1956 }
1957
1958 fn put(&mut self, bytes: &[u8]) -> Result<()> {
1963 write_at(&self.file, self.at, bytes)?;
1964 self.at = self
1965 .at
1966 .checked_add(bytes.len() as u64)
1967 .ok_or_else(|| invalid("native file length overflow"))?;
1968 Ok(())
1969 }
1970
1971 pub fn append(&mut self, chunk: &Chunk) -> Result<()> {
1977 let order = (self.next_order, 0);
1978 self.next_order = self.next_order.saturating_add(1);
1979 self.append_at(order, chunk)
1980 }
1981
1982 pub fn append_at(&mut self, order: (u64, u64), chunk: &Chunk) -> Result<()> {
1993 if chunk.is_empty() {
1994 return Ok(());
1995 }
1996 self.admit(chunk)?;
1997 if self.pending.last().is_some_and(|last| last.order > order) {
1998 self.flush_pending()?;
1999 }
2000 self.pending.push(PendingChunk { order, chunk: chunk.clone() });
2005 if self.pending.len() == STRIPE_PARTS {
2006 self.flush_pending()?;
2007 }
2008 Ok(())
2009 }
2010
2011 pub fn append_stripe(&mut self, parts: Vec<((u64, u64), Chunk)>) -> Result<()> {
2027 if parts.len() > STRIPE_PARTS {
2028 return Err(invalid("a stripe was handed more parts than it holds"));
2029 }
2030 self.flush_pending()?;
2033 for (order, chunk) in parts {
2034 if chunk.is_empty() {
2035 continue;
2036 }
2037 self.admit(&chunk)?;
2038 self.pending.push(PendingChunk { order, chunk });
2039 }
2040 self.flush_pending()
2041 }
2042
2043 fn admit(&mut self, chunk: &Chunk) -> Result<()> {
2045 if chunk.width() != self.table.fields.len() {
2046 return Err(invalid("chunk width differs from table schema"));
2047 }
2048 for (index, field) in self.table.fields.iter().enumerate() {
2049 if chunk.column(index)?.logical_type() != &field.ty {
2050 return Err(invalid("chunk type differs from table schema"));
2051 }
2052 }
2053 self.table.rows = self
2054 .table
2055 .rows
2056 .checked_add(chunk.len())
2057 .ok_or_else(|| invalid("row count overflow"))?;
2058 Ok(())
2059 }
2060
2061 fn encode_pages(columns: &[&Vector]) -> Result<ColumnStripe> {
2063 let mut stripe = ColumnStripe {
2064 pages: Vec::with_capacity(columns.len()),
2065 codes: Vec::with_capacity(columns.len()),
2066 sieves: Vec::with_capacity(columns.len()),
2067 ranges: Vec::with_capacity(columns.len()),
2068 };
2069 let mut settling = Settling::default();
2070 for &column in columns {
2071 let bytes = encode(column, &mut settling)?;
2072 if bytes.len() > MAX_PAGE {
2073 return Err(invalid("column page exceeds the configured bound"));
2074 }
2075 let range = Range::of(column);
2078 let sieve =
2089 Sieve::of(column, &range, SIEVE_BUDGET).filter(|sieve| sieve.len() < bytes.len());
2090 stripe.pages.push(bytes);
2091 stripe.codes.push(None);
2092 stripe.sieves.push(sieve);
2093 stripe.ranges.push(range);
2094 }
2095 Ok(stripe)
2096 }
2097
2098 fn place_blocks(&mut self) -> Result<()> {
2103 let mut dictionaries = std::mem::take(&mut self.dictionaries);
2104 let placed = dictionaries.iter_mut().flatten().try_for_each(|dictionary| {
2105 for block in std::mem::take(&mut dictionary.blocks) {
2106 let start = self.at;
2107 self.put(&block)?;
2108 dictionary.placed.push(Placed {
2109 start,
2110 length: block.len() as u64,
2111 hash: checksum(&block),
2112 });
2113 }
2114 Ok(())
2115 });
2116 self.dictionaries = dictionaries;
2117 placed
2118 }
2119
2120 fn flush_pending(&mut self) -> Result<()> {
2125 if self.pending.is_empty() {
2126 return Ok(());
2127 }
2128 let held = std::mem::take(&mut self.pending);
2129 let prepared = self.preparer().prepare_held(held)?;
2130 let merged = self.merge_held(prepared)?;
2131 let paged = merged.pages()?;
2132 self.write_paged(paged)
2133 }
2134
2135 fn write_stripe(&mut self, held: &[Part], encoded: Vec<ColumnStripe>) -> Result<()> {
2137 let width = self.table.fields.len();
2138 let parts = held.len();
2139 if encoded.len() != width {
2140 return Err(Error::internal("a stripe came to the writer with the wrong columns"));
2141 }
2142 let profile = self.profile.clone();
2143 if let Some(profile) = &profile {
2144 let rows = held.iter().map(|part| part.rows as u64).sum();
2145 let raw = held.iter().map(|part| part.footprint as u64).sum();
2146 let pages =
2147 encoded.iter().flat_map(|stripe| &stripe.pages).map(|page| page.len() as u64).sum();
2148 profile.moved(Stage::Pages, raw, pages, rows);
2149 }
2150 let timing = profile.as_deref().map(|profile| profile.span(Stage::Dictionary));
2153 let before = self.at;
2154 encode_ready(&mut self.dictionaries)?;
2155 self.place_blocks()?;
2156 drop(timing);
2157 if let Some(profile) = &profile {
2158 profile.moved(Stage::Dictionary, 0, self.at - before, 0);
2159 }
2160 let timing = profile.as_deref().map(|profile| profile.span(Stage::Write));
2161 let before = self.at;
2162 let mut pages = Vec::with_capacity(width);
2163 let mut memberships = vec![None; width];
2164 let mut ranges = Vec::with_capacity(width);
2165 let mut index = Vec::with_capacity(width.saturating_mul(index_section(parts)?));
2166 for stripe in &encoded {
2167 let offset = self.at;
2168 let section = index.len();
2169 let mut length = 0_usize;
2170 for bytes in &stripe.pages {
2171 write_at(&self.file, self.at + length as u64, bytes)?;
2172 put_u32(
2173 &mut index,
2174 u32::try_from(bytes.len()).map_err(|_| invalid("part length overflow"))?,
2175 );
2176 put_u64(&mut index, checksum(bytes));
2177 length = length
2178 .checked_add(bytes.len())
2179 .ok_or_else(|| invalid("column page length overflow"))?;
2180 }
2181 let hash = checksum(&index[section..]);
2182 put_u64(&mut index, hash);
2183 if length > MAX_PAGE {
2184 return Err(invalid("column page exceeds the configured bound"));
2185 }
2186 self.at = self
2187 .at
2188 .checked_add(length as u64)
2189 .ok_or_else(|| invalid("native file length overflow"))?;
2190 pages.push(Span {
2191 offset,
2192 length: u32::try_from(length).map_err(|_| invalid("page length overflow"))?,
2193 });
2194 ranges.push(merged_range(stripe.ranges.iter().cloned()));
2195 }
2196 for (membership, stripe) in memberships.iter_mut().zip(&encoded) {
2197 if stripe.codes.iter().all(Option::is_none) {
2198 continue;
2199 }
2200 let lists = stripe
2201 .codes
2202 .iter()
2203 .map(|codes| codes.clone().unwrap_or_default())
2204 .collect::<Vec<_>>();
2205 let bytes = encode_membership(&merged_codes(lists));
2206 let offset = self.at;
2207 self.put(&bytes)?;
2208 *membership = Some(Page {
2209 offset,
2210 length: u32::try_from(bytes.len())
2211 .map_err(|_| invalid("membership page length overflow"))?,
2212 hash: checksum(&bytes),
2213 });
2214 }
2215 let mut sieves = vec![None; width];
2216 for (page, stripe) in sieves.iter_mut().zip(&encoded) {
2217 if stripe.sieves.iter().all(Option::is_none) {
2218 continue;
2219 }
2220 let bytes = encode_sieves(stripe.sieves.iter())?;
2221 let offset = self.at;
2222 self.put(&bytes)?;
2223 *page = Some(Page {
2224 offset,
2225 length: u32::try_from(bytes.len())
2226 .map_err(|_| invalid("sieve page length overflow"))?,
2227 hash: checksum(&bytes),
2228 });
2229 }
2230 let mut part_ranges = vec![None; width];
2236 if parts > 1 {
2237 for ((page, stripe), span) in part_ranges.iter_mut().zip(&encoded).zip(&pages) {
2238 let bytes = encode_part_ranges(&stripe.ranges)?;
2239 if bytes.len() >= span.length as usize {
2240 continue;
2241 }
2242 let offset = self.at;
2243 self.put(&bytes)?;
2244 *page = Some(Page {
2245 offset,
2246 length: u32::try_from(bytes.len())
2247 .map_err(|_| invalid("part range page length overflow"))?,
2248 hash: checksum(&bytes),
2249 });
2250 }
2251 }
2252 let offset = self.at;
2253 self.put(&index)?;
2254 let index = Span {
2255 offset,
2256 length: u32::try_from(index.len())
2257 .map_err(|_| invalid("index page length overflow"))?,
2258 };
2259 let mut rows = 0_usize;
2260 let mut lengths = Vec::with_capacity(parts);
2261 let mut span = None;
2262 for part in held {
2263 rows = rows.checked_add(part.rows).ok_or_else(|| invalid("row count overflow"))?;
2264 lengths.push(u32::try_from(part.rows).map_err(|_| invalid("part row count overflow"))?);
2265 span = Some(span.map_or((part.order, part.order), |(first, _)| (first, part.order)));
2266 }
2267 self.order.push(span.ok_or_else(|| invalid("a stripe was flushed with no parts"))?);
2268 self.table.stripes.push(Stripe {
2269 rows,
2270 parts: lengths,
2271 index,
2272 pages,
2273 memberships: Pages::from_slots(memberships)?,
2274 sieves: Pages::from_slots(sieves)?,
2275 part_ranges: Pages::from_slots(part_ranges)?,
2276 zone: Zone::from_ranges(ranges),
2277 });
2278 drop(timing);
2279 if let Some(profile) = &profile {
2280 profile.moved(Stage::Write, 0, self.at - before, rows as u64);
2281 }
2282 Ok(())
2283 }
2284
2285 fn numeric_frequency(&self, column: usize) -> Result<(Option<FrequencySummary>, Option<u64>)> {
2301 let signed = match self.table.fields[column].ty {
2302 LogicalType::TinyInt
2303 | LogicalType::SmallInt
2304 | LogicalType::Integer
2305 | LogicalType::BigInt
2306 | LogicalType::Date
2307 | LogicalType::Timestamp => true,
2308 LogicalType::UTinyInt
2309 | LogicalType::USmallInt
2310 | LogicalType::UInteger
2311 | LogicalType::UBigInt => false,
2312 _ => return Ok((None, None)),
2313 };
2314 let value_of = |bits: Option<u64>| match bits {
2315 None => FrequencyValue::Null,
2316 Some(bits) if signed => FrequencyValue::Integer(i128::from(bits as i64)),
2317 Some(bits) => FrequencyValue::Integer(i128::from(bits)),
2318 };
2319 let mut candidates: FrequencyMap<u32> = FrequencyMap::default();
2320 let mut nulls = 0_u32;
2321 let mut decrements = 0_u64;
2322 let mut distinct = distinct::ExactDistinct::new();
2323 self.visit_numeric(column, signed, |_, bits| {
2324 let held = match bits {
2325 Some(bits) => {
2326 distinct.insert(bits);
2327 candidates.get_mut(&bits)
2328 }
2329 None if nulls != 0 => Some(&mut nulls),
2330 None => None,
2331 };
2332 if let Some(count) = held {
2333 *count = count.saturating_add(1);
2334 } else if candidates.len() + usize::from(nulls != 0) < FREQUENCY_CANDIDATES {
2335 match bits {
2336 Some(bits) => {
2337 candidates.insert(bits, 1);
2338 }
2339 None => nulls = 1,
2340 }
2341 } else {
2342 candidates.retain(|_, count| {
2343 *count -= 1;
2344 *count != 0
2345 });
2346 nulls = nulls.saturating_sub(1);
2347 decrements = decrements.saturating_add(1);
2348 }
2349 })?;
2350 let (exact, null_count) = if decrements == 0 {
2351 let exact = candidates
2352 .into_iter()
2353 .map(|(bits, count)| (bits, u64::from(count)))
2354 .collect::<FrequencyMap<_>>();
2355 (exact, (nulls != 0).then_some(u64::from(nulls)))
2356 } else {
2357 let mut lower = candidates.values().copied().collect::<Vec<_>>();
2358 if nulls != 0 {
2359 lower.push(nulls);
2360 }
2361 lower.sort_unstable_by(|left, right| right.cmp(left));
2362 if lower.len() < FREQUENCY_BUILD_RANK
2363 || u64::from(lower[FREQUENCY_BUILD_RANK - 1]) <= decrements
2364 {
2365 return Ok((None, distinct.count()));
2366 }
2367 let mut exact =
2368 candidates.into_keys().map(|bits| (bits, 0_u64)).collect::<FrequencyMap<_>>();
2369 let mut null_count = (nulls != 0).then_some(0_u64);
2370 self.visit_numeric(column, signed, |_, bits| {
2371 let held = match bits {
2372 Some(bits) => exact.get_mut(&bits),
2373 None => null_count.as_mut(),
2374 };
2375 if let Some(count) = held {
2376 *count = count.saturating_add(1);
2377 }
2378 })?;
2379 (exact, null_count)
2380 };
2381 let mut entries = exact
2382 .into_iter()
2383 .map(|(bits, count)| FrequencyEntry { value: value_of(Some(bits)), count })
2384 .chain(null_count.map(|count| FrequencyEntry { value: FrequencyValue::Null, count }))
2385 .collect::<Vec<_>>();
2386 let omitted_max = keep_most_frequent(&mut entries).max(decrements);
2387 let kept_rows = entries.iter().try_fold(0_u64, |total, entry| {
2388 total.checked_add(entry.count).filter(|&total| total <= FREQUENCY_ORDINALS as u64)
2389 });
2390 let mut ordinals = Vec::new();
2391 let mut ordinal_entries = Vec::new();
2392 if let Some(kept_rows) = kept_rows {
2393 let mut kept = FrequencyMap::default();
2394 let mut null_kept = None;
2395 for (at, entry) in entries.iter().enumerate() {
2396 let at = u16::try_from(at)
2397 .map_err(|_| invalid("too many retained frequency entries"))?;
2398 match entry.value {
2399 FrequencyValue::Integer(value) => {
2400 kept.insert(value as u64, at);
2401 }
2402 FrequencyValue::Null => null_kept = Some(at),
2403 FrequencyValue::Code(_) => {}
2404 }
2405 }
2406 ordinals.reserve(usize::try_from(kept_rows).unwrap_or(FREQUENCY_ORDINALS));
2407 ordinal_entries.reserve(usize::try_from(kept_rows).unwrap_or(FREQUENCY_ORDINALS));
2408 self.visit_numeric(column, signed, |ordinal, bits| {
2409 let held = match bits {
2410 Some(bits) => kept.get(&bits).copied(),
2411 None => null_kept,
2412 };
2413 if let Some(entry) = held {
2414 ordinals.push(ordinal);
2415 ordinal_entries.push(entry);
2416 }
2417 })?;
2418 }
2419 Ok((
2420 Some(FrequencySummary { entries, omitted_max, ordinals, ordinal_entries }),
2421 distinct.count(),
2422 ))
2423 }
2424
2425 fn visit_numeric(
2432 &self,
2433 column: usize,
2434 signed: bool,
2435 mut visit: impl FnMut(u64, Option<u64>),
2436 ) -> Result<()> {
2437 let ty = &self.table.fields[column].ty;
2438 let mut start = 0_u64;
2439 let mut block = Vec::new();
2440 for stripe in &self.table.stripes {
2441 let spans = read_index(&self.file, stripe, column)?;
2442 let page = stripe.pages[column];
2443 let mut bytes = vec![0; page.length as usize];
2444 read_at(&self.file, page.offset, &mut bytes)?;
2445 for (span, &rows) in spans.iter().zip(&stripe.parts) {
2446 let part = part_bytes(&bytes, *span)?;
2447 if checksum(part) != span.hash {
2448 return Err(invalid("column page checksum differs while building frequencies"));
2449 }
2450 let rows = rows as usize;
2451 let vector = decode(ty, rows, part, None)?;
2452 if signed && vector.signed_block(&mut block) && block.len() == rows {
2456 if vector.none_null() {
2457 for (row, &value) in block.iter().enumerate() {
2458 visit(start.saturating_add(row as u64), Some(value as u64));
2459 }
2460 } else {
2461 for (row, &value) in block.iter().enumerate() {
2462 let bits = (!vector.is_null_at(row)).then_some(value as u64);
2463 visit(start.saturating_add(row as u64), bits);
2464 }
2465 }
2466 start = start.saturating_add(rows as u64);
2467 continue;
2468 }
2469 for row in 0..rows {
2471 let bits = if vector.is_null_at(row) {
2472 None
2473 } else {
2474 let widened = match vector.signed_at(row) {
2478 Some(value) => Some(value as u64),
2479 None => match vector.value_at(row) {
2480 Value::UTinyInt(value) => Some(u64::from(value)),
2481 Value::USmallInt(value) => Some(u64::from(value)),
2482 Value::UInteger(value) => Some(u64::from(value)),
2483 Value::UBigInt(value) => Some(value),
2484 _ => None,
2485 },
2486 };
2487 Some(widened.ok_or_else(|| {
2488 invalid("numeric frequency page did not contain an integer value")
2489 })?)
2490 };
2491 visit(start.saturating_add(row as u64), bits);
2492 }
2493 start = start.saturating_add(rows as u64);
2494 }
2495 }
2496 Ok(())
2497 }
2498
2499 fn numeric_frequencies(&self) -> Result<Vec<(Option<FrequencySummary>, Option<u64>)>> {
2507 let mut columns = self
2508 .table
2509 .fields
2510 .iter()
2511 .enumerate()
2512 .filter_map(|(column, field)| {
2513 matches!(
2514 field.ty,
2515 LogicalType::TinyInt
2516 | LogicalType::SmallInt
2517 | LogicalType::Integer
2518 | LogicalType::BigInt
2519 | LogicalType::UTinyInt
2520 | LogicalType::USmallInt
2521 | LogicalType::UInteger
2522 | LogicalType::UBigInt
2523 | LogicalType::Date
2524 | LogicalType::Timestamp
2525 )
2526 .then_some(column)
2527 })
2528 .collect::<Vec<_>>();
2529 let workers = std::thread::available_parallelism()
2530 .map_or(1, usize::from)
2531 .min(MAX_FREQUENCY_WORKERS)
2532 .min(columns.len());
2533 let profile = self.profile.as_deref();
2534 if workers <= 1 {
2535 let _timing = profile.map(|profile| profile.span(Stage::Publish));
2536 let mut frequencies = vec![(None, None); self.table.fields.len()];
2537 for column in columns {
2538 frequencies[column] = self.numeric_frequency(column)?;
2539 }
2540 return Ok(frequencies);
2541 }
2542 columns.sort_by_key(|&column| weight(&self.table.fields[column].ty));
2545 let queue = Mutex::new(columns);
2546 let pieces = std::thread::scope(|scope| {
2547 (0..workers)
2548 .map(|_| {
2549 scope.spawn(|| {
2550 let _timing = profile.map(|profile| profile.span(Stage::Publish));
2551 let mut mine = Vec::new();
2552 loop {
2553 let taken = queue
2554 .lock()
2555 .map_err(|_| Error::internal("a native frequency worker panicked"))?
2556 .pop();
2557 let Some(column) = taken else { break };
2558 mine.push((column, self.numeric_frequency(column)?));
2559 }
2560 Ok(mine)
2561 })
2562 })
2563 .collect::<Vec<_>>()
2564 .into_iter()
2565 .map(|handle| {
2566 handle
2567 .join()
2568 .map_err(|_| Error::internal("a native frequency worker panicked"))?
2569 })
2570 .collect::<Result<Vec<_>>>()
2571 })?;
2572 let mut frequencies = vec![(None, None); self.table.fields.len()];
2573 for piece in pieces {
2574 for (column, summary) in piece {
2575 frequencies[column] = summary;
2576 }
2577 }
2578 Ok(frequencies)
2579 }
2580
2581 fn stable_codes_at(&self, column: usize, ordinals: &[u64]) -> Result<Option<Vec<Option<u32>>>> {
2583 if self.dictionaries.get(column).and_then(Option::as_ref).is_none() {
2584 return Ok(None);
2585 }
2586 if ordinals.windows(2).any(|pair| pair[0] >= pair[1]) {
2587 return Err(invalid("frequency ordinals are not sorted and unique"));
2588 }
2589 let mut out = Vec::with_capacity(ordinals.len());
2590 let mut wanted = 0;
2591 let mut stripe_start = 0_u64;
2592 for stripe in &self.table.stripes {
2593 let stripe_end = stripe_start.saturating_add(stripe.rows as u64);
2594 if wanted == ordinals.len() || ordinals[wanted] >= stripe_end {
2595 stripe_start = stripe_end;
2596 continue;
2597 }
2598 let spans = read_index(&self.file, stripe, column)?;
2599 let page = stripe.pages[column];
2600 let mut bytes = vec![0; page.length as usize];
2601 read_at(&self.file, page.offset, &mut bytes)?;
2602 let mut part_start = stripe_start;
2603 for (span, &rows) in spans.iter().zip(&stripe.parts) {
2604 let part_end = part_start.saturating_add(u64::from(rows));
2605 if wanted < ordinals.len() && ordinals[wanted] < part_end {
2606 let part = part_bytes(&bytes, *span)?;
2607 if checksum(part) != span.hash {
2608 return Err(invalid(
2609 "column page checksum differs while building pair frequencies",
2610 ));
2611 }
2612 let upto = ordinals.partition_point(|&ordinal| ordinal < part_end);
2613 let positions = ordinals[wanted..upto]
2614 .iter()
2615 .map(|&ordinal| {
2616 usize::try_from(ordinal.saturating_sub(part_start))
2617 .map_err(|_| invalid("frequency row offset does not fit in memory"))
2618 })
2619 .collect::<Result<Vec<_>>>()?;
2620 if !decode_selected_stable_codes(rows as usize, part, &positions, &mut out)? {
2621 return Ok(None);
2622 }
2623 wanted = upto;
2624 }
2625 part_start = part_end;
2626 }
2627 stripe_start = stripe_end;
2628 }
2629 if wanted != ordinals.len() {
2630 return Err(invalid("frequency ordinal is outside the table"));
2631 }
2632 Ok(Some(out))
2633 }
2634
2635 fn pair_frequencies(&self) -> Result<Vec<PairFrequencySummary>> {
2637 let anchors = self
2638 .table
2639 .frequencies
2640 .iter()
2641 .enumerate()
2642 .filter_map(|(column, summary)| {
2643 match summary {
2645 Some(Frequencies::Held(summary)) => Some(summary),
2646 _ => None,
2647 }
2648 .filter(|summary| {
2649 !summary.ordinals.is_empty()
2650 && summary.ordinal_entries.len() == summary.ordinals.len()
2651 })
2652 .cloned()
2653 .map(|summary| (column, summary))
2654 })
2655 .collect::<Vec<_>>();
2656 let strings = self
2657 .dictionaries
2658 .iter()
2659 .enumerate()
2660 .filter_map(|(column, dictionary)| dictionary.as_ref().map(|_| column))
2661 .collect::<Vec<_>>();
2662 let mut summaries = Vec::new();
2663 for (first, anchors) in anchors {
2664 for &second in &strings {
2665 if summaries.len() == MAX_PAIR_FREQUENCIES {
2666 return Ok(summaries);
2667 }
2668 let Some(codes) = self.stable_codes_at(second, &anchors.ordinals)? else {
2669 continue;
2670 };
2671 if codes.len() != anchors.ordinal_entries.len() {
2672 return Err(invalid("pair frequency columns have different lengths"));
2673 }
2674 let mut counts = HashMap::<(u16, Option<u32>), u64>::new();
2675 for (&anchor, code) in anchors.ordinal_entries.iter().zip(codes) {
2676 *counts.entry((anchor, code)).or_default() += 1;
2677 }
2678 let mut entries = counts
2679 .into_iter()
2680 .map(|((first_entry, second), count)| PairFrequencyEntry {
2681 first_entry,
2682 second,
2683 count,
2684 })
2685 .collect::<Vec<_>>();
2686 entries.sort_unstable_by(|left, right| {
2687 right
2688 .count
2689 .cmp(&left.count)
2690 .then_with(|| left.first_entry.cmp(&right.first_entry))
2691 .then_with(|| left.second.cmp(&right.second))
2692 });
2693 let pair_omitted = entries.get(FREQUENCY_ENTRIES).map_or(0, |entry| entry.count);
2694 entries.truncate(FREQUENCY_ENTRIES);
2695 summaries.push(PairFrequencySummary {
2696 first: u16::try_from(first)
2697 .map_err(|_| invalid("pair frequency column index overflows"))?,
2698 second: u16::try_from(second)
2699 .map_err(|_| invalid("pair frequency column index overflows"))?,
2700 entries,
2701 omitted_max: anchors.omitted_max.max(pair_omitted),
2702 });
2703 }
2704 }
2705 Ok(summaries)
2706 }
2707
2708 fn close(&mut self) -> Result<Entry> {
2719 self.flush_pending()?;
2720 let profile = self.profile.clone();
2724 let timing = profile.as_deref().map(|profile| profile.span(Stage::Publish));
2725 let before = self.at;
2726 let mut stripes = std::mem::take(&mut self.order)
2727 .into_iter()
2728 .zip(std::mem::take(&mut self.table.stripes))
2729 .collect::<Vec<_>>();
2730 stripes.sort_by_key(|(order, _)| order.0);
2731 let mut previous: Option<(u64, u64)> = None;
2732 for ((first, last), _) in &stripes {
2733 if previous.is_some_and(|previous| previous >= *first) {
2734 return Err(invalid("chunks did not arrive in source order"));
2735 }
2736 previous = Some(*last);
2737 }
2738 self.table.stripes = stripes.into_iter().map(|(_, stripe)| stripe).collect();
2739 drop(timing);
2743 let (frequencies, distincts): (Vec<Option<FrequencySummary>>, _) =
2744 self.numeric_frequencies()?.into_iter().unzip();
2745 self.table.frequencies =
2746 frequencies.into_iter().map(|held| held.map(Frequencies::Held)).collect();
2747 self.table.distincts = distincts;
2748 let timing = profile.as_deref().map(|profile| profile.span(Stage::Dictionary));
2749 let placing = self.at;
2750 finish_dictionaries(&mut self.dictionaries)?;
2751 self.place_blocks()?;
2752 self.table.pair_frequencies = self.pair_frequencies()?;
2753 let dictionaries = std::mem::take(&mut self.dictionaries);
2754 self.table.dictionary_payloads = vec![0; self.table.fields.len()];
2755 self.table.frequency_texts = vec![Vec::new(); self.table.fields.len()];
2756 self.table.host_groups = None;
2757 for (index, dictionary) in dictionaries.into_iter().enumerate() {
2762 let Some(dictionary) = dictionary else { continue };
2763 let (order, flat, bases) = dictionary.ranked_with_values(Some(&self.file))?;
2764 self.table.distincts[index] =
2768 Some(dictionary.counts.iter().filter(|count| **count != 0).count() as u64);
2769 let (frequencies, texts) = code_frequency(&dictionary, &flat, &bases)?;
2770 self.table.frequencies[index] = Some(Frequencies::Held(frequencies));
2771 self.table.frequency_texts[index] = texts;
2772 if self.table.fields[index].name.eq_ignore_ascii_case("Referer") {
2773 self.table.host_groups = host::build(index, &dictionary, &flat, &bases)?;
2774 }
2775 drop(flat);
2776 drop(bases);
2777 let encoded = encode_global_dictionary(&dictionary, &order, &dictionary.placed, true)?;
2778 drop(order);
2779 let offset = self.at;
2780 self.put(&encoded.index)?;
2781 self.put(&encoded.ranks)?;
2782 self.put(&encoded.grams)?;
2783 self.table.dictionary_payloads[index] = dictionary
2784 .placed
2785 .iter()
2786 .try_fold(0_u64, |sum, place| sum.checked_add(place.length))
2787 .ok_or_else(|| invalid("global dictionary payload overflow"))?;
2788 let length = encoded
2789 .index
2790 .len()
2791 .checked_add(encoded.ranks.len())
2792 .and_then(|len| len.checked_add(encoded.grams.len()))
2793 .ok_or_else(|| invalid("dictionary page length overflow"))?;
2794 self.table.dictionaries[index] = Some(Page {
2795 offset,
2796 length: u32::try_from(length)
2797 .map_err(|_| invalid("dictionary page length overflow"))?,
2798 hash: checksum(&encoded.index),
2799 });
2800 }
2801 drop(timing);
2802 let timing = profile.as_deref().map(|profile| profile.span(Stage::Publish));
2803 let placed = self.at - placing;
2804 self.write_stats()?;
2805 let directory = encode_directory(&self.table)?;
2806 if directory.len() > MAX_DIRECTORY {
2807 return Err(invalid("directory exceeds the configured bound"));
2808 }
2809 let offset = self.at;
2810 self.put(&directory)?;
2811 drop(timing);
2812 if let Some(profile) = &profile {
2813 profile.moved(Stage::Dictionary, 0, placed, 0);
2814 profile.moved(Stage::Publish, 0, self.at - before - placed, 0);
2815 }
2816 Ok(Entry {
2817 name: self.table.name.clone(),
2818 fields: self.table.fields.clone(),
2819 rows: self.table.rows,
2820 nonzero: table_nonzero_counts(&self.table),
2821 aggregates: table_aggregate_sums(&self.table),
2822 directory: Page {
2823 offset,
2824 length: u32::try_from(directory.len())
2825 .map_err(|_| invalid("directory length overflow"))?,
2826 hash: checksum(&directory),
2827 },
2828 })
2829 }
2830
2831 fn write_stats(&mut self) -> Result<()> {
2843 let gathers = std::mem::take(&mut self.gathers);
2844 let rows = self.table.rows as u64;
2845 let mut payloads = Vec::new();
2846 for (column, gather) in gathers.into_iter().enumerate() {
2847 let Some(gather) = gather else { continue };
2848 if gather.rows() != rows {
2854 continue;
2855 }
2856 let Some(stats) = gather.finish() else { continue };
2857 let mut summary = Vec::new();
2858 stats.summary.encode(&mut summary)?;
2859 let mut sketches = Vec::new();
2860 stats.sketches.encode(&mut sketches)?;
2861 payloads.push((column, summary, sketches));
2862 }
2863 if payloads.is_empty() {
2864 return Ok(());
2865 }
2866 let costs = payloads
2867 .iter()
2868 .map(|(_, summary, sketches)| summary.len() + sketches.len())
2869 .collect::<Vec<_>>();
2870 let allowance = stats::allowance(stats::column_bytes(&self.table), stats::BUDGET_SHARE);
2871 let keep = stats::within(&costs, allowance, 0);
2874 for ((column, summary, sketches), _) in
2875 payloads.iter().zip(&keep).filter(|&(_, &keep)| keep)
2876 {
2877 let id = u64::try_from(*column).map_err(|_| invalid("column index overflow"))?;
2878 for (kind, bytes, header_bytes) in [
2879 (*section::SUMMARY, summary, summary.len() as u32),
2882 (*section::SKETCHES, sketches, rudb_stats::sketches::HEADER_BYTES),
2883 ] {
2884 let written = write_section(
2885 &self.file,
2886 &mut self.at,
2887 §ion::Attachment { kind, id, flags: 0, header_bytes, bytes },
2888 self.generation,
2889 )?;
2890 self.table.sections.push(written);
2891 }
2892 }
2893 if self.table.sections.len() > MAX_SECTIONS {
2894 return Err(invalid("the table would name more sections than the bound allows"));
2895 }
2896 Ok(())
2897 }
2898
2899 pub fn finish(mut self) -> Result<Table> {
2909 let entry = self.close()?;
2910 let profile = self.profile.take();
2911 let _timing = profile.as_deref().map(|profile| profile.span(Stage::Publish));
2912 let mut tables = std::mem::take(&mut self.closed);
2913 tables.push(entry);
2914 let catalog = encode_catalog(&tables, &self.views)?;
2915 if catalog.len() > MAX_DIRECTORY {
2916 return Err(invalid("catalog exceeds the configured bound"));
2917 }
2918 let offset = self.at;
2919 self.put(&catalog)?;
2920 if let Some(profile) = &profile {
2921 profile.moved(Stage::Publish, 0, catalog.len() as u64, 0);
2922 }
2923 synced(&self.file, profile.as_deref())?;
2927 let slot = Slot {
2928 offset,
2929 length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
2930 generation: self.generation,
2931 hash: checksum(&catalog),
2932 };
2933 write_at(&self.file, slot_offset(self.generation), &slot.bytes())?;
2938 synced(&self.file, profile.as_deref())?;
2939 Ok(self.table)
2940 }
2941
2942 pub fn restate(path: impl AsRef<Path>, views: &[ViewEntry]) -> Result<()> {
2959 let path = path.as_ref();
2960 let (_, size, slot, bytes, _) = slot_bytes(path)?;
2961 let (closed, _) = decode_catalog(&bytes, size)?;
2962 let generation = slot
2963 .generation
2964 .checked_add(1)
2965 .ok_or_else(|| invalid("native file generation overflow"))?;
2966 let catalog = encode_catalog(&closed, views)?;
2967 if catalog.len() > MAX_DIRECTORY {
2968 return Err(invalid("catalog exceeds the configured bound"));
2969 }
2970 let file = OpenOptions::new().write(true).read(true).open(path).map_err(io)?;
2971 write_at(&file, size, &catalog)?;
2972 file.sync_all().map_err(io)?;
2973 let slot = Slot {
2974 offset: size,
2975 length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
2976 generation,
2977 hash: checksum(&catalog),
2978 };
2979 write_at(&file, slot_offset(generation), &slot.bytes())?;
2980 file.sync_all().map_err(io)?;
2981 Ok(())
2982 }
2983
2984 pub fn certify_summaries(path: impl AsRef<Path>) -> Result<()> {
2987 let path = path.as_ref();
2988 let (_, size, slot, bytes, _) = slot_bytes(path)?;
2989 let (mut entries, views) = decode_catalog(&bytes, size)?;
2990 let native = Catalog::open(path)?;
2991 for entry in &mut entries {
2992 let reader = native.table(&entry.name)?;
2993 entry.nonzero = reader_nonzero_counts(&reader)?;
2994 entry.aggregates = reader_aggregate_sums(&reader)?;
2995 }
2996 let generation = slot
2997 .generation
2998 .checked_add(1)
2999 .ok_or_else(|| invalid("native file generation overflow"))?;
3000 let catalog = encode_catalog(&entries, &views)?;
3001 if catalog.len() > MAX_DIRECTORY {
3002 return Err(invalid("catalog exceeds the configured bound"));
3003 }
3004 let file = OpenOptions::new().write(true).read(true).open(path).map_err(io)?;
3005 write_at(&file, size, &catalog)?;
3006 file.sync_all().map_err(io)?;
3007 let slot = Slot {
3008 offset: size,
3009 length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
3010 generation,
3011 hash: checksum(&catalog),
3012 };
3013 write_at(&file, slot_offset(generation), &slot.bytes())?;
3014 file.sync_all().map_err(io)?;
3015 Ok(())
3016 }
3017
3018 pub fn certify_counts(path: impl AsRef<Path>) -> Result<()> {
3020 Self::certify_summaries(path)
3021 }
3022}
3023
3024fn append(file: &File, at: &mut u64, bytes: &[u8]) -> Result<u64> {
3030 let offset = *at;
3031 write_at(file, offset, bytes)?;
3032 *at =
3033 at.checked_add(bytes.len() as u64).ok_or_else(|| invalid("native file length overflow"))?;
3034 Ok(offset)
3035}
3036
3037fn write_section(
3043 file: &File,
3044 at: &mut u64,
3045 one: §ion::Attachment<'_>,
3046 generation: u64,
3047) -> Result<Section> {
3048 if !one.bytes.is_empty() && one.header_bytes as usize > one.bytes.len() {
3052 return Err(invalid("a section's header is longer than its payload"));
3053 }
3054 let mut extents = Vec::new();
3055 let mut first = 0_u64;
3056 for chunk in one.bytes.chunks(section::MAX_EXTENT as usize) {
3057 let offset = append(file, at, chunk)?;
3058 extents.push(section::Extent {
3059 offset,
3060 length: u32::try_from(chunk.len()).map_err(|_| invalid("extent length overflow"))?,
3061 hash: checksum(chunk),
3062 first,
3063 });
3064 first += chunk.len() as u64;
3065 }
3066 let mut table = Vec::with_capacity(extents.len() * section::EXTENT_BYTES);
3067 section::encode_extents(&extents, &mut table)?;
3068 let extent_page = if table.is_empty() { 0 } else { append(file, at, &table)? };
3072 Ok(Section {
3073 kind: one.kind,
3074 id: one.id,
3075 generation,
3076 extents: u32::try_from(extents.len()).map_err(|_| invalid("too many extents"))?,
3077 extent_page,
3078 extent_bytes: u32::try_from(table.len()).map_err(|_| invalid("extent table overflow"))?,
3079 hash: checksum(&table),
3080 flags: one.flags,
3081 header_bytes: one.header_bytes,
3082 })
3083}
3084
3085pub fn attach(
3109 path: impl AsRef<Path>,
3110 table: &str,
3111 attachments: &[section::Attachment<'_>],
3112) -> Result<Table> {
3113 let path = path.as_ref();
3114 let (_, size, slot, bytes, _) = slot_bytes(path)?;
3115 let (mut entries, views) = decode_catalog(&bytes, size)?;
3116 let at = entries
3117 .iter()
3118 .position(|entry| entry.name == table)
3119 .ok_or_else(|| invalid(&format!("the file holds no table called {table}")))?;
3120 let file = OpenOptions::new().write(true).read(true).open(path).map_err(io)?;
3121 let mut version = [0; 4];
3122 read_at(&file, 8, &mut version)?;
3123 let version = u32::from_le_bytes(version);
3124 if version != FORMAT {
3130 return Err(invalid(&format!(
3131 "the file is format {version} and a graph section needs format {FORMAT}, so it has \
3132 to be written again"
3133 )));
3134 }
3135 let mut directory = vec![0; entries[at].directory.length as usize];
3136 read_at(&file, entries[at].directory.offset, &mut directory)?;
3137 if checksum(&directory) != entries[at].directory.hash {
3138 return Err(invalid(&format!("the directory of table {table} does not checksum")));
3139 }
3140 let mut held = decode_directory(&directory, size)?;
3141 let mut cursor = size;
3142 for one in attachments {
3143 let written = write_section(&file, &mut cursor, one, held.generation)?;
3144 held.sections.retain(|old| !(old.kind == one.kind && old.id == one.id));
3145 held.sections.push(written);
3146 }
3147 if held.sections.len() > MAX_SECTIONS {
3148 return Err(invalid("the table would name more sections than the bound allows"));
3149 }
3150 let encoded = encode_directory(&held)?;
3151 if encoded.len() > MAX_DIRECTORY {
3152 return Err(invalid("directory exceeds the configured bound"));
3153 }
3154 let offset = append(&file, &mut cursor, &encoded)?;
3155 entries[at].directory = Page {
3156 offset,
3157 length: u32::try_from(encoded.len()).map_err(|_| invalid("directory length overflow"))?,
3158 hash: checksum(&encoded),
3159 };
3160 let catalog = encode_catalog(&entries, &views)?;
3163 if catalog.len() > MAX_DIRECTORY {
3164 return Err(invalid("catalog exceeds the configured bound"));
3165 }
3166 let offset = append(&file, &mut cursor, &catalog)?;
3167 file.sync_all().map_err(io)?;
3168 let generation =
3169 slot.generation.checked_add(1).ok_or_else(|| invalid("native file generation overflow"))?;
3170 let committed = Slot {
3171 offset,
3172 length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
3173 generation,
3174 hash: checksum(&catalog),
3175 };
3176 write_at(&file, slot_offset(generation), &committed.bytes())?;
3177 file.sync_all().map_err(io)?;
3178 Ok(held)
3179}
3180
3181type Synopsis = Arc<Vec<(Value, u64)>>;
3184
3185#[derive(Debug, Clone)]
3187pub struct Reader {
3188 file: Arc<File>,
3189 table: Arc<Table>,
3190 dictionaries: Arc<Vec<OnceLock<Arc<Vector>>>>,
3191 loading: Arc<Vec<Mutex<()>>>,
3200 frequency_values: Arc<Vec<OnceLock<Synopsis>>>,
3203 frequency_summaries: Arc<Vec<OnceLock<Arc<FrequencySummary>>>>,
3207 opened: Arc<AtomicUsize>,
3211 sieves: Arc<Vec<Vec<SieveSlot>>>,
3215 part_ranges: Arc<Vec<Vec<RangeSlot>>>,
3218 places: Arc<Vec<Place>>,
3220 cache: Arc<Shelf>,
3221 pool: PagePool,
3223 pages: Arc<AtomicUsize>,
3226 indexes: Arc<AtomicUsize>,
3229 size: u64,
3231 directory: u64,
3233 opening: Opening,
3235}
3236
3237#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
3249pub struct Opening {
3250 pub reads: u32,
3253 pub bytes: u64,
3255}
3256
3257#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
3259pub struct Reads {
3260 pub opening: Opening,
3262 pub pages: usize,
3264 pub indexes: usize,
3266 pub dictionaries: usize,
3269}
3270
3271#[derive(Debug, Clone, Copy)]
3273struct Place {
3274 stripe: u32,
3275 part: u32,
3276 rows: u32,
3277}
3278
3279#[derive(Debug, Clone, Copy)]
3281struct PartSpan {
3282 start: usize,
3283 length: usize,
3284 hash: u64,
3285}
3286
3287#[derive(Debug, Clone)]
3293struct CachedColumn {
3294 stripe: usize,
3295 index: Arc<Vec<PartSpan>>,
3296 page: Option<Arc<Vec<u8>>>,
3297}
3298
3299#[derive(Debug, Default)]
3319struct Cached {
3320 pages: Vec<Option<Resident>>,
3321 loading: Vec<usize>,
3322 index: Vec<Option<Arc<Vec<PartSpan>>>>,
3323}
3324
3325#[derive(Debug, Clone)]
3327struct Resident {
3328 page: Arc<Vec<u8>>,
3329 used: Arc<AtomicBool>,
3330}
3331
3332#[derive(Debug)]
3334struct Shelf {
3335 columns: Vec<Mutex<Cached>>,
3336 held: Vec<AtomicUsize>,
3339 kept: AtomicUsize,
3342}
3343
3344#[derive(Debug, Clone, Default)]
3363pub struct PagePool {
3364 ring: Arc<Mutex<Ring>>,
3365 budget: Arc<AtomicUsize>,
3366}
3367
3368#[derive(Debug, Default)]
3369struct Ring {
3370 held: VecDeque<Held>,
3371 bytes: usize,
3372}
3373
3374#[derive(Debug)]
3379struct Held {
3380 shelf: Weak<Shelf>,
3381 column: usize,
3382 stripe: usize,
3383 bytes: usize,
3384 used: Arc<AtomicBool>,
3385}
3386
3387impl PagePool {
3388 #[must_use]
3390 pub fn new(budget: usize) -> Self {
3391 let pool = Self::default();
3392 pool.budget.store(budget, Atomic::Relaxed);
3393 pool
3394 }
3395
3396 #[must_use]
3402 pub fn bytes(&self) -> usize {
3403 self.ring.lock().map_or(0, |ring| ring.bytes)
3404 }
3405
3406 fn admit(&self, held: Held) {
3412 let budget = self.budget.load(Atomic::Relaxed);
3413 let mut gone = Vec::new();
3414 {
3415 let Ok(mut ring) = self.ring.lock() else { return };
3416 ring.bytes += held.bytes;
3417 ring.held.push_back(held);
3418 let mut looked = 0;
3421 let limit = ring.held.len();
3422 while ring.bytes > budget && looked < limit {
3423 looked += 1;
3424 let Some(entry) = ring.held.pop_front() else { break };
3425 let Some(shelf) = entry.shelf.upgrade() else {
3426 ring.bytes -= entry.bytes;
3427 continue;
3428 };
3429 if entry.used.swap(false, Atomic::Relaxed) {
3430 ring.held.push_back(entry);
3431 continue;
3432 }
3433 let count = &shelf.held[entry.column];
3434 if count.load(Atomic::Relaxed) <= shelf.kept.load(Atomic::Relaxed).max(1) {
3435 ring.held.push_back(entry);
3436 continue;
3437 }
3438 count.fetch_sub(1, Atomic::Relaxed);
3439 ring.bytes -= entry.bytes;
3440 gone.push((shelf, entry));
3441 }
3442 while ring.held.front().is_some_and(|entry| entry.shelf.strong_count() == 0) {
3445 if let Some(entry) = ring.held.pop_front() {
3446 ring.bytes -= entry.bytes;
3447 }
3448 }
3449 }
3450 for (shelf, entry) in gone {
3451 let Ok(mut cached) = shelf.columns[entry.column].lock() else { continue };
3452 if let Some(slot) = cached.pages.get_mut(entry.stripe) {
3453 if slot.as_ref().is_some_and(|slot| Arc::ptr_eq(&slot.used, &entry.used)) {
3454 *slot = None;
3455 }
3456 }
3457 }
3458 }
3459}
3460
3461const CACHED_STRIPES_PER_COLUMN: usize = 4;
3473
3474type SieveSlot = OnceLock<Arc<Vec<Option<Sieve>>>>;
3476
3477type RangeSlot = OnceLock<Arc<Vec<Range>>>;
3478
3479#[derive(Debug)]
3480struct NativeText {
3481 file: Arc<File>,
3482 values: usize,
3484 offsets: Vec<u8>,
3493 offset_bits: usize,
3496 value_ends: OnceLock<Option<Vec<u32>>>,
3509 value_lens: OnceLock<Option<Vec<u32>>>,
3519 ends_asked: AtomicUsize,
3525 ranks: usize,
3527 rank_at: u64,
3531 rank_ends: Vec<u64>,
3535 rank_hashes: Vec<u64>,
3536 rank_blocks: Vec<OnceLock<Result<Vec<u8>>>>,
3537 code_bits: usize,
3540 code_ranks: OnceLock<Option<Vec<u32>>>,
3547 starts: Vec<u64>,
3554 lengths: Vec<u64>,
3555 hashes: Vec<u64>,
3556 grams: Option<NativeGrams>,
3558 blocks: Vec<OnceLock<Result<Vec<u8>>>>,
3560 keep_budget: usize,
3563 payload_kept: AtomicUsize,
3571 swept: Vec<AtomicBool>,
3579 searched: Mutex<HashMap<Vec<u8>, (usize, bool)>>,
3596}
3597
3598#[derive(Debug)]
3599struct NativeGrams {
3600 start: u64,
3601 length: usize,
3602 hash: u64,
3603 loaded: OnceLock<Result<Vec<u8>>>,
3604}
3605
3606const TEXT_SEARCH_MEMO: usize = 64;
3611
3612const TEXT_PAYLOAD_VALUES: usize = 1024;
3628
3629const TEXT_GRAM_BYTES: usize = 2048;
3632
3633fn gram_bits(bytes: &[u8]) -> [usize; 2] {
3635 let original = u32::from_le_bytes(bytes.try_into().expect("a four-byte gram"));
3636 let mut first = original ^ (original >> 16);
3637 first = first.wrapping_mul(0x7feb_352d);
3638 first ^= first >> 15;
3639 let mut second = original ^ (original >> 17);
3640 second = second.wrapping_mul(0x846c_a68b);
3641 second ^= second >> 16;
3642 let mask = TEXT_GRAM_BYTES * 8 - 1;
3643 [(first as usize) & mask, (second as usize) & mask]
3644}
3645
3646const TEXT_KEEP_BUDGET: usize = 256 * 1024 * 1024;
3667
3668fn lengths_of(ends: &[u32]) -> Option<Vec<u32>> {
3674 let mut lens = Vec::with_capacity(ends.len());
3675 for block in ends.chunks(TEXT_PAYLOAD_VALUES) {
3676 let mut start = 0;
3677 for &end in block {
3678 lens.push(end.checked_sub(start)?);
3679 start = end;
3680 }
3681 }
3682 Some(lens)
3683}
3684
3685const TEXT_OFFSET_RUN: usize = 512;
3692
3693const DICTIONARY_HEADER: usize = 16;
3696
3697const DICTIONARY_SCATTERED: u32 = 1 << 31;
3711const DICTIONARY_GRAMS: u32 = 1 << 30;
3713
3714const TEXT_RANK_BLOCK: usize = 512;
3725
3726const RANK_BLOCK_HEADER: usize = size_of::<u64>() + 1;
3740
3741impl NativeText {
3742 fn payload_block(&self, block: usize) -> Result<Option<&[u8]>> {
3749 let Some(slot) = self.blocks.get(block) else { return Ok(None) };
3750 let bytes = slot.get_or_init(|| self.decode_block(block)).as_ref().map_err(Clone::clone)?;
3751 Ok(Some(bytes.as_slice()))
3752 }
3753
3754 fn decode_block(&self, block: usize) -> Result<Vec<u8>> {
3759 let len = self.lengths[block];
3760 let mut stored = vec![
3761 0;
3762 usize::try_from(len).map_err(|_| invalid(
3763 "global dictionary block does not fit in memory"
3764 ))?
3765 ];
3766 read_at(&self.file, self.starts[block], &mut stored)?;
3767 if checksum(&stored) != self.hashes[block] {
3768 return Err(invalid("global dictionary payload checksum differs"));
3769 }
3770 let first = block * TEXT_PAYLOAD_VALUES;
3771 let last = (first + TEXT_PAYLOAD_VALUES).min(self.values);
3772 let want = self.end_within(last - 1)? as usize;
3773 let values = string::decode_flat(&stored)?;
3774 if values.len() != last - first {
3775 return Err(invalid("global dictionary block holds the wrong value count"));
3776 }
3777 let bytes = values.into_bytes();
3778 if bytes.len() != want {
3779 return Err(invalid("global dictionary block decodes to the wrong length"));
3780 }
3781 Ok(bytes)
3782 }
3783
3784 fn ends_worth_unpacking(&self) -> usize {
3801 self.values.max(TEXT_PAYLOAD_VALUES)
3802 }
3803
3804 fn value_ends(&self) -> Option<&[u32]> {
3806 if let Some(built) = self.value_ends.get() {
3807 return built.as_deref();
3808 }
3809 if self.ends_asked.fetch_add(1, Atomic::Relaxed) < self.ends_worth_unpacking() {
3810 return None;
3811 }
3812 self.value_ends.get_or_init(|| self.unpack_ends()).as_deref()
3813 }
3814
3815 fn unpack_ends(&self) -> Option<Vec<u32>> {
3821 let mut ends = vec![0u32; self.values];
3822 for (run, into) in ends.chunks_mut(TEXT_OFFSET_RUN).enumerate() {
3823 let bytes = self.offsets.get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)?;
3824 bitpack::unpack_tail_into(bytes, self.offset_bits, into, |bits| {
3825 u32::try_from(bits).unwrap_or(u32::MAX)
3826 })
3827 .ok()?;
3828 }
3829 if ends.contains(&u32::MAX) { None } else { Some(ends) }
3832 }
3833
3834 fn end_within(&self, index: usize) -> Result<u32> {
3836 if let Some(ends) = self.value_ends() {
3837 return ends
3838 .get(index)
3839 .copied()
3840 .ok_or_else(|| invalid("global dictionary offsets are short"));
3841 }
3842 let run = index / TEXT_OFFSET_RUN;
3843 let bytes = self
3844 .offsets
3845 .get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)
3846 .ok_or_else(|| invalid("global dictionary offsets are short"))?;
3847 let end = bitpack::tail_at(bytes, self.offset_bits, index % TEXT_OFFSET_RUN)
3848 .map_err(|_| invalid("global dictionary offsets are short"))?;
3849 u32::try_from(end).map_err(|_| invalid("global dictionary offset is past the payload"))
3850 }
3851
3852 fn ends_within(&self, first: usize, last: usize) -> Result<Vec<u64>> {
3870 let mut ends = vec![0u64; last.saturating_sub(first)];
3871 let mut scratch = Vec::new();
3872 let mut at = first;
3873 while at < last {
3874 let run = at / TEXT_OFFSET_RUN;
3875 let stop = ((run + 1) * TEXT_OFFSET_RUN).min(last);
3876 let held = self.values.saturating_sub(run * TEXT_OFFSET_RUN).min(TEXT_OFFSET_RUN);
3877 let bytes = self
3878 .offsets
3879 .get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)
3880 .ok_or_else(|| invalid("global dictionary offsets are short"))?;
3881 let from = at % TEXT_OFFSET_RUN;
3882 let upto = stop - run * TEXT_OFFSET_RUN;
3883 if upto > held || bytes.len() < bitpack::tail_len(held, self.offset_bits) {
3884 return Err(invalid("global dictionary offsets are short"));
3885 }
3886 let into = &mut ends[at - first..stop - first];
3887 if from == 0 {
3888 bitpack::unpack_tail_into(bytes, self.offset_bits, into, |bits| bits)
3889 .map_err(|_| invalid("global dictionary offsets are short"))?;
3890 } else {
3891 scratch.resize(held, 0);
3892 bitpack::unpack_tail_into(bytes, self.offset_bits, &mut scratch, |bits| bits)
3893 .map_err(|_| invalid("global dictionary offsets are short"))?;
3894 into.copy_from_slice(&scratch[from..upto]);
3895 }
3896 at = stop;
3897 }
3898 Ok(ends)
3899 }
3900
3901 fn start_within(&self, index: usize) -> Result<u32> {
3904 if index % TEXT_PAYLOAD_VALUES == 0 { Ok(0) } else { self.end_within(index - 1) }
3905 }
3906
3907 fn span_within(&self, index: usize) -> Result<(u32, u32)> {
3915 if let Some(ends) = self.value_ends() {
3916 let end =
3917 *ends.get(index).ok_or_else(|| invalid("global dictionary offsets are short"))?;
3918 let start = if index % TEXT_PAYLOAD_VALUES == 0 { 0 } else { ends[index - 1] };
3921 if start > end {
3922 return Err(invalid("global dictionary value ends before it starts"));
3923 }
3924 return Ok((start, end));
3925 }
3926 let within = index % TEXT_OFFSET_RUN;
3927 let (start, end) = if within == 0 {
3928 (self.start_within(index)?, self.end_within(index)?)
3929 } else {
3930 let run = index / TEXT_OFFSET_RUN;
3931 let bytes = self
3932 .offsets
3933 .get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)
3934 .ok_or_else(|| invalid("global dictionary offsets are short"))?;
3935 let (start, end) = bitpack::tail_pair(bytes, self.offset_bits, within)
3936 .map_err(|_| invalid("global dictionary offsets are short"))?;
3937 let ends = u32::try_from(end)
3938 .map_err(|_| invalid("global dictionary offset is past the payload"))?;
3939 let starts = u32::try_from(start)
3940 .map_err(|_| invalid("global dictionary offset is past the payload"))?;
3941 (starts, ends)
3942 };
3943 if start > end {
3944 return Err(invalid("global dictionary value ends before it starts"));
3945 }
3946 Ok((start, end))
3947 }
3948
3949 fn rank_parts(&self, rank: usize) -> Result<(&[u8], usize)> {
3956 let slot = self
3957 .rank_blocks
3958 .get(rank / TEXT_RANK_BLOCK)
3959 .ok_or_else(|| invalid("global dictionary rank is past the order"))?;
3960 let block = slot
3961 .get_or_init(|| {
3962 let which = rank / TEXT_RANK_BLOCK;
3963 let start = if which == 0 { 0 } else { self.rank_ends[which - 1] };
3964 let end = self.rank_ends[which];
3965 let mut bytes = vec![0; (end - start) as usize];
3966 read_at(&self.file, self.rank_at + start, &mut bytes)?;
3967 if checksum(&bytes)
3968 != *self
3969 .rank_hashes
3970 .get(rank / TEXT_RANK_BLOCK)
3971 .ok_or_else(|| invalid("global dictionary rank block has no checksum"))?
3972 {
3973 return Err(invalid("global dictionary rank checksum differs"));
3974 }
3975 Ok(bytes)
3976 })
3977 .as_ref()
3978 .map_err(Clone::clone)?;
3979 Ok((block.as_slice(), rank % TEXT_RANK_BLOCK))
3980 }
3981
3982 fn head_at(&self, rank: usize) -> Result<u64> {
3984 let (block, within) = self.rank_parts(rank)?;
3985 let (base, width, packed) = rank_heads(block)?;
3986 let above = bitpack::tail_at(packed, width, within)
3987 .map_err(|_| invalid("global dictionary rank block is short of heads"))?;
3988 Ok(base.wrapping_add(above))
3989 }
3990
3991 fn rank_codes<'block>(&self, block: &'block [u8], count: usize) -> Result<&'block [u8]> {
3993 let (_, width, packed) = rank_heads(block)?;
3994 packed
3995 .get(bitpack::tail_len(count, width)..)
3996 .ok_or_else(|| invalid("global dictionary rank block is short of codes"))
3997 }
3998
3999 fn rank_block_len(&self, rank: usize) -> usize {
4001 let first = rank / TEXT_RANK_BLOCK * TEXT_RANK_BLOCK;
4002 TEXT_RANK_BLOCK.min(self.ranks - first)
4003 }
4004}
4005
4006fn rank_heads(block: &[u8]) -> Result<(u64, usize, &[u8])> {
4008 let header = block
4009 .get(..RANK_BLOCK_HEADER)
4010 .ok_or_else(|| invalid("global dictionary rank block is short"))?;
4011 let base = u64::from_le_bytes(header[..8].try_into().expect("eight bytes"));
4012 let width = header[8] as usize;
4013 if width > 64 {
4014 return Err(invalid("global dictionary rank block packs heads past a word"));
4015 }
4016 Ok((base, width, &block[RANK_BLOCK_HEADER..]))
4017}
4018
4019fn offset_width(ends: &[u32]) -> usize {
4026 let span = ends.iter().copied().max().unwrap_or(0);
4030 (u32::BITS - span.leading_zeros()) as usize
4031}
4032
4033fn offset_bytes(values: usize, bits: usize) -> usize {
4036 let full = values / TEXT_OFFSET_RUN;
4037 let rest = values % TEXT_OFFSET_RUN;
4038 full * TEXT_OFFSET_RUN / 8 * bits + bitpack::tail_len(rest, bits)
4039}
4040
4041fn encode_offsets(ends: &[u32], bits: usize, out: &mut Vec<u8>) -> Result<()> {
4045 let mut run = Vec::with_capacity(TEXT_OFFSET_RUN);
4046 for chunk in ends.chunks(TEXT_OFFSET_RUN) {
4047 run.clear();
4048 run.extend(chunk.iter().map(|&end| u64::from(end)));
4049 bitpack::pack_tail(&run, bits, out)
4050 .map_err(|_| invalid("global dictionary offsets do not pack"))?;
4051 }
4052 Ok(())
4053}
4054
4055fn code_width(values: usize) -> usize {
4057 match u64::try_from(values).unwrap_or(u64::MAX) {
4058 0 | 1 => 0,
4059 last => (u64::BITS - (last - 1).leading_zeros()) as usize,
4060 }
4061}
4062
4063impl TextSource for NativeText {
4064 fn len(&self) -> usize {
4065 self.values
4066 }
4067
4068 fn might_contain(&self, first: usize, literal: &[u8]) -> Result<bool> {
4069 let Some(grams) = &self.grams else { return Ok(true) };
4070 if literal.len() < 4 || first >= self.values {
4071 return Ok(true);
4072 }
4073 let bytes = grams
4074 .loaded
4075 .get_or_init(|| {
4076 let mut bytes = vec![0; grams.length];
4077 read_at(&self.file, grams.start, &mut bytes)?;
4078 if checksum(&bytes) != grams.hash {
4079 return Err(invalid("global dictionary substring signatures checksum differs"));
4080 }
4081 Ok(bytes)
4082 })
4083 .as_ref()
4084 .map_err(Clone::clone)?;
4085 let block = first / TEXT_PAYLOAD_VALUES;
4086 let Some(bits) = bytes.get(block * TEXT_GRAM_BYTES..(block + 1) * TEXT_GRAM_BYTES) else {
4087 return Ok(true);
4088 };
4089 Ok(literal.windows(4).all(|gram| {
4090 gram_bits(gram).into_iter().all(|bit| bits[bit / 8] & (1 << (bit % 8)) != 0)
4091 }))
4092 }
4093
4094 fn bytes_at(&self, index: usize) -> Result<Option<&[u8]>> {
4095 if index >= self.values {
4096 return Ok(None);
4097 }
4098 let (start, end) = self.span_within(index)?;
4099 if start == end {
4100 return Ok(Some(&[]));
4101 }
4102 let block = index / TEXT_PAYLOAD_VALUES;
4105 let Some(bytes) = self.payload_block(block)? else { return Ok(None) };
4106 Ok(bytes.get(start as usize..end as usize))
4107 }
4108
4109 fn bytes_len_at(&self, index: usize) -> Result<Option<usize>> {
4110 if index >= self.values {
4111 return Ok(None);
4112 }
4113 let (start, end) = self.span_within(index)?;
4114 Ok(Some((end - start) as usize))
4115 }
4116
4117 fn bytes_lens_at(&self, indices: &[u32], into: &mut [i64]) -> Result<()> {
4124 self.ends_asked.fetch_add(indices.len(), Atomic::Relaxed);
4125 let Some(ends) = self.value_ends() else {
4126 for (slot, &index) in into.iter_mut().zip(indices) {
4127 *slot = self
4128 .bytes_len_at(index as usize)?
4129 .map_or(0, |len| i64::try_from(len).unwrap_or(i64::MAX));
4130 }
4131 return Ok(());
4132 };
4133 if let Some(lens) = self.value_lens.get_or_init(|| lengths_of(ends)) {
4134 for (slot, &index) in into.iter_mut().zip(indices) {
4135 *slot = lens.get(index as usize).map_or(0, |&len| i64::from(len));
4138 }
4139 return Ok(());
4140 }
4141 for (slot, &index) in into.iter_mut().zip(indices) {
4142 let index = index as usize;
4143 let Some(&end) = ends.get(index) else {
4145 *slot = 0;
4146 continue;
4147 };
4148 let start = if index % TEXT_PAYLOAD_VALUES == 0 { 0 } else { ends[index - 1] };
4149 if start > end {
4150 return Err(invalid("global dictionary value ends before it starts"));
4151 }
4152 *slot = i64::from(end - start);
4153 }
4154 Ok(())
4155 }
4156
4157 fn sweep(
4170 &self,
4171 first: usize,
4172 limit: usize,
4173 body: &mut dyn FnMut(usize, &[u8]) -> Result<()>,
4174 ) -> Result<usize> {
4175 let limit = limit.min(self.values);
4176 if first >= limit {
4177 return Ok(first);
4178 }
4179 let block = first / TEXT_PAYLOAD_VALUES;
4180 let last = ((block + 1) * TEXT_PAYLOAD_VALUES).min(limit);
4181 let decoded;
4182 let kept = self.blocks.get(block).and_then(OnceLock::get);
4183 let again = kept.is_none()
4184 && self.swept.get(block).is_some_and(|swept| swept.swap(true, Atomic::Relaxed));
4185 let bytes: &[u8] = match kept {
4186 Some(Ok(kept)) => kept,
4187 _ if again && self.payload_kept.load(Atomic::Relaxed) < self.keep_budget => {
4188 let kept = self
4189 .payload_block(block)?
4190 .ok_or_else(|| invalid("global dictionary block is past the payload"))?;
4191 self.payload_kept.fetch_add(kept.len(), Atomic::Relaxed);
4192 kept
4193 }
4194 _ => {
4195 decoded = self.decode_block(block)?;
4196 &decoded
4197 }
4198 };
4199 let ends = self.ends_within(first, last)?;
4200 if ends.len() != last - first {
4201 return Err(invalid("global dictionary offsets are short"));
4202 }
4203 let mut start = u64::from(self.start_within(first)?);
4204 for (index, &end) in (first..last).zip(&ends) {
4207 let value = usize::try_from(start)
4208 .ok()
4209 .zip(usize::try_from(end).ok())
4210 .and_then(|(from, to)| bytes.get(from..to))
4211 .ok_or_else(|| invalid("global dictionary value is past its block"))?;
4212 body(index, value)?;
4213 start = end;
4214 }
4215 Ok(last)
4216 }
4217
4218 fn visit(
4224 &self,
4225 indices: &[usize],
4226 body: &mut dyn FnMut(usize, &[u8]) -> Result<()>,
4227 ) -> Result<()> {
4228 let mut at = 0;
4229 while at < indices.len() {
4230 let block = indices[at] / TEXT_PAYLOAD_VALUES;
4231 let upto =
4232 at + indices[at..].partition_point(|&index| index / TEXT_PAYLOAD_VALUES == block);
4233 let wanted = &indices[at..upto];
4234 if wanted.iter().any(|&index| index >= self.values) {
4235 return Err(invalid("a visited value is past the global dictionary"));
4236 }
4237 let decoded;
4238 let bytes: &[u8] = match self.blocks.get(block).and_then(OnceLock::get) {
4239 Some(Ok(kept)) => kept,
4240 _ => {
4241 decoded = self.decode_block(block)?;
4242 &decoded
4243 }
4244 };
4245 for (offset, &index) in wanted.iter().enumerate() {
4246 let (start, end) = self.span_within(index)?;
4247 let value = bytes
4248 .get(start as usize..end as usize)
4249 .ok_or_else(|| invalid("global dictionary value is past its block"))?;
4250 body(at + offset, value)?;
4251 }
4252 at = upto;
4253 }
4254 Ok(())
4255 }
4256
4257 fn ranks(&self) -> Option<usize> {
4258 (self.ranks > 0).then_some(self.ranks)
4259 }
4260
4261 fn below(&self, ranks: usize, wanted: &[u8]) -> Result<(usize, bool)> {
4269 let mut memo = self.searched.lock().map_err(|_| invalid("a poisoned dictionary search"))?;
4270 if let Some(&answer) = memo.get(wanted) {
4271 return Ok(answer);
4272 }
4273 let answer = search_below(self, ranks, wanted)?;
4274 if memo.len() >= TEXT_SEARCH_MEMO {
4275 memo.clear();
4276 }
4277 memo.insert(wanted.to_vec(), answer);
4278 Ok(answer)
4279 }
4280
4281 fn compare_rank(&self, rank: usize, wanted: &[u8]) -> Result<Ordering> {
4282 let settled = self.head_at(rank)?.cmp(&head(wanted));
4286 if settled != Ordering::Equal {
4287 return Ok(settled);
4288 }
4289 let code = self.code_at_rank(rank)?;
4290 let bytes = self
4291 .bytes_at(code as usize)?
4292 .ok_or_else(|| invalid("global dictionary order names a code it does not have"))?;
4293 Ok(bytes.cmp(wanted))
4294 }
4295
4296 fn code_at_rank(&self, rank: usize) -> Result<u32> {
4297 let (block, within) = self.rank_parts(rank)?;
4298 let codes = self.rank_codes(block, self.rank_block_len(rank))?;
4299 let code = bitpack::tail_at(codes, self.code_bits, within)
4300 .map_err(|_| invalid("global dictionary rank block is short of codes"))?;
4301 let code = u32::try_from(code)
4302 .map_err(|_| invalid("global dictionary order names a code it does not have"))?;
4303 if code as usize >= self.len() {
4304 return Err(invalid("global dictionary order names a code it does not have"));
4305 }
4306 Ok(code)
4307 }
4308
4309 fn code_ranks(&self) -> Option<&[u32]> {
4310 if self.ranks == 0 || self.ranks != self.len() {
4314 return None;
4315 }
4316 self.code_ranks
4317 .get_or_init(|| {
4318 let mut ranks = vec![u32::MAX; self.ranks];
4319 for first in (0..self.ranks).step_by(TEXT_RANK_BLOCK) {
4322 let (block, _) = self.rank_parts(first).ok()?;
4323 let count = self.rank_block_len(first);
4324 let codes = self.rank_codes(block, count).ok()?;
4325 for (within, code) in bitpack::unpack_tail(codes, self.code_bits, count)
4326 .ok()?
4327 .into_iter()
4328 .enumerate()
4329 {
4330 let code = usize::try_from(code).ok()?;
4331 *ranks.get_mut(code)? = u32::try_from(first + within).ok()?;
4332 }
4333 }
4334 if ranks.contains(&u32::MAX) {
4335 return None;
4336 }
4337 Some(ranks)
4338 })
4339 .as_deref()
4340 }
4341
4342 fn footprint(&self) -> usize {
4343 self.offsets.capacity()
4344 + self
4345 .value_ends
4346 .get()
4347 .and_then(Option::as_ref)
4348 .map_or(0, |ends| ends.capacity() * size_of::<u32>())
4349 + self
4350 .value_lens
4351 .get()
4352 .and_then(Option::as_ref)
4353 .map_or(0, |lens| lens.capacity() * size_of::<u32>())
4354 + self
4355 .code_ranks
4356 .get()
4357 .and_then(Option::as_ref)
4358 .map_or(0, |ranks| ranks.capacity() * size_of::<u32>())
4359 + self.rank_hashes.capacity() * size_of::<u64>()
4360 + self.rank_ends.capacity() * size_of::<u64>()
4361 + self.rank_blocks.capacity() * size_of::<OnceLock<Result<Vec<u8>>>>()
4362 + self
4363 .rank_blocks
4364 .iter()
4365 .filter_map(OnceLock::get)
4366 .filter_map(|result| result.as_ref().ok())
4367 .map(Vec::capacity)
4368 .sum::<usize>()
4369 + self.blocks.capacity() * size_of::<OnceLock<Result<Vec<u8>>>>()
4370 + self.hashes.capacity() * size_of::<u64>()
4371 + self.starts.capacity() * size_of::<u64>()
4372 + self.lengths.capacity() * size_of::<u64>()
4373 + self
4374 .grams
4375 .as_ref()
4376 .and_then(|grams| grams.loaded.get())
4377 .and_then(|result| result.as_ref().ok())
4378 .map_or(0, Vec::capacity)
4379 + self
4380 .blocks
4381 .iter()
4382 .filter_map(OnceLock::get)
4383 .filter_map(|result| result.as_ref().ok())
4384 .map(Vec::capacity)
4385 .sum::<usize>()
4386 }
4387}
4388
4389fn places(table: &Table) -> Result<Vec<Place>> {
4391 let mut places = Vec::with_capacity(table.stripes.len().saturating_mul(STRIPE_PARTS));
4392 for (at, stripe) in table.stripes.iter().enumerate() {
4393 let index = u32::try_from(at).map_err(|_| invalid("too many stripes"))?;
4394 for (part, &rows) in stripe.parts.iter().enumerate() {
4395 places.push(Place {
4396 stripe: index,
4397 part: u32::try_from(part).map_err(|_| invalid("too many parts in a stripe"))?,
4398 rows,
4399 });
4400 }
4401 }
4402 Ok(places)
4403}
4404
4405fn read_index(file: &File, stripe: &Stripe, column: usize) -> Result<Vec<PartSpan>> {
4410 let parts = stripe.parts.len();
4411 let section = index_section(parts)?;
4412 let at = column.checked_mul(section).ok_or_else(|| invalid("index page offset overflow"))?;
4413 let end = at.checked_add(section).ok_or_else(|| invalid("index page offset overflow"))?;
4414 if end > stripe.index.length as usize {
4415 return Err(invalid("index page is shorter than its columns"));
4416 }
4417 let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
4418 let mut bytes = vec![0; section];
4419 let offset = stripe
4420 .index
4421 .offset
4422 .checked_add(at as u64)
4423 .ok_or_else(|| invalid("index page offset overflow"))?;
4424 read_at(file, offset, &mut bytes)?;
4425 let entries = section - size_of::<u64>();
4426 let stored = u64::from_le_bytes(bytes[entries..].try_into().expect("eight bytes"));
4427 if checksum(&bytes[..entries]) != stored {
4428 return Err(invalid(&format!(
4431 "index page section checksum differs, column {column} of {parts} parts at {offset}, \
4432 wanted {stored:016x} and got {:016x}",
4433 checksum(&bytes[..entries]),
4434 )));
4435 }
4436 let mut spans = Vec::with_capacity(parts);
4437 let mut start = 0_usize;
4438 for part in 0..parts {
4439 let at = part * INDEX_ENTRY;
4440 let length = u32::from_le_bytes(bytes[at..at + 4].try_into().expect("four bytes")) as usize;
4441 let hash = u64::from_le_bytes(bytes[at + 4..at + 12].try_into().expect("eight bytes"));
4442 spans.push(PartSpan { start, length, hash });
4443 start = start.checked_add(length).ok_or_else(|| invalid("column page length overflow"))?;
4444 }
4445 if start != page.length as usize {
4446 return Err(invalid("column page length differs from its index"));
4447 }
4448 Ok(spans)
4449}
4450
4451fn part_bytes(page: &[u8], span: PartSpan) -> Result<&[u8]> {
4453 let end = span.start.checked_add(span.length).ok_or_else(|| invalid("part range overflow"))?;
4454 page.get(span.start..end).ok_or_else(|| invalid("part exceeds its column page"))
4455}
4456
4457fn remember(cached: &mut Cached, held: &CachedColumn) -> Option<(usize, Arc<AtomicBool>)> {
4463 if let Some(slot) = cached.index.get_mut(held.stripe) {
4464 if slot.is_none() {
4465 *slot = Some(Arc::clone(&held.index));
4466 }
4467 }
4468 let page = held.page.clone()?;
4469 let slot = cached.pages.get_mut(held.stripe)?;
4470 if slot.is_some() {
4471 return None;
4472 }
4473 let bytes = page.len();
4474 let used = Arc::new(AtomicBool::new(true));
4477 *slot = Some(Resident { page, used: Arc::clone(&used) });
4478 Some((bytes, used))
4479}
4480
4481#[derive(Debug, Clone)]
4490pub struct Catalog {
4491 file: Arc<File>,
4492 size: u64,
4493 entries: Arc<Vec<Entry>>,
4494 views: Arc<Vec<ViewEntry>>,
4496 opening: Opening,
4497 pool: PagePool,
4499}
4500
4501#[derive(Debug, Clone, PartialEq, Eq)]
4503pub struct CertifiedSums {
4504 pub columns: Vec<(i128, u64)>,
4505 pub rows: u64,
4506}
4507
4508impl Catalog {
4509 pub fn open(path: impl AsRef<Path>) -> Result<Self> {
4518 Self::open_in(path, &PagePool::default())
4519 }
4520
4521 pub fn open_in(path: impl AsRef<Path>, pool: &PagePool) -> Result<Self> {
4527 let (file, size, _, bytes, opening) = slot_bytes(path)?;
4528 let (entries, views) = decode_catalog(&bytes, size)?;
4529 Ok(Self {
4530 file: Arc::new(file),
4531 size,
4532 entries: Arc::new(entries),
4533 views: Arc::new(views),
4534 opening,
4535 pool: pool.clone(),
4536 })
4537 }
4538
4539 pub fn names(&self) -> impl ExactSizeIterator<Item = &str> {
4541 self.entries.iter().map(|entry| entry.name.as_str())
4542 }
4543
4544 pub fn rows(&self) -> impl ExactSizeIterator<Item = (&str, usize)> {
4551 self.entries.iter().map(|entry| (entry.name.as_str(), entry.rows))
4552 }
4553
4554 pub fn views(&self) -> impl ExactSizeIterator<Item = &ViewEntry> {
4560 self.views.iter()
4561 }
4562
4563 #[must_use]
4565 pub fn len(&self) -> usize {
4566 self.entries.len()
4567 }
4568
4569 #[must_use]
4572 pub fn is_empty(&self) -> bool {
4573 self.entries.is_empty()
4574 }
4575
4576 pub fn table(&self, name: &str) -> Result<Reader> {
4582 let entry = self
4583 .entries
4584 .iter()
4585 .find(|entry| entry.name == name)
4586 .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
4587 let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
4591 if file_checksum(&self.file, offset, length)? != entry.directory.hash {
4592 return Err(invalid(&format!("the directory of table {name} does not checksum")));
4593 }
4594 let mut opening = self.opening;
4595 opening.reads += 1;
4596 opening.bytes += u64::from(entry.directory.length);
4597 Reader::build(
4598 Arc::clone(&self.file),
4599 self.size,
4600 read_directory(Cursor::over(&self.file, offset, length), self.size, Some(offset))?,
4601 u64::from(entry.directory.length),
4602 opening,
4603 self.pool.clone(),
4604 )
4605 }
4606
4607 pub fn nonzero_count(&self, name: &str, column: usize) -> Result<Option<u64>> {
4611 let entry = self
4612 .entries
4613 .iter()
4614 .find(|entry| entry.name == name)
4615 .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
4616 let Some(field) = entry.fields.get(column) else {
4617 return Err(invalid("frequency column index out of range"));
4618 };
4619 if !matches!(
4620 field.ty,
4621 LogicalType::TinyInt
4622 | LogicalType::SmallInt
4623 | LogicalType::Integer
4624 | LogicalType::BigInt
4625 | LogicalType::UTinyInt
4626 | LogicalType::USmallInt
4627 | LogicalType::UInteger
4628 | LogicalType::UBigInt
4629 ) {
4630 return Ok(None);
4631 }
4632 let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
4633 if file_checksum(&self.file, offset, length)? != entry.directory.hash {
4634 return Err(invalid(&format!("the directory of table {name} does not checksum")));
4635 }
4636 if let Some(count) = entry.nonzero.get(column).copied().flatten() {
4637 return Ok(Some(count));
4638 }
4639 quick_nonzero(
4640 Cursor::over(&self.file, offset, length),
4641 &entry.name,
4642 &entry.fields,
4643 entry.rows,
4644 column,
4645 )
4646 }
4647
4648 pub fn aggregate_sums(&self, name: &str, columns: &[usize]) -> Result<Option<CertifiedSums>> {
4651 let entry = self
4652 .entries
4653 .iter()
4654 .find(|entry| entry.name == name)
4655 .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
4656 let mut sums = Vec::with_capacity(columns.len());
4657 for &column in columns {
4658 let Some(field) = entry.fields.get(column) else {
4659 return Err(invalid("aggregate column index out of range"));
4660 };
4661 if !signed_integer(&field.ty) {
4662 return Ok(None);
4663 }
4664 let Some(sum) = entry.aggregates[column] else {
4665 return Ok(None);
4666 };
4667 sums.push(sum);
4668 }
4669 let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
4670 if file_checksum(&self.file, offset, length)? != entry.directory.hash {
4671 return Err(invalid(&format!("the directory of table {name} does not checksum")));
4672 }
4673 Ok(Some(CertifiedSums { columns: sums, rows: entry.rows as u64 }))
4674 }
4675
4676 pub fn table_fields(&self, name: &str) -> Option<&[Field]> {
4678 self.entries.iter().find(|entry| entry.name == name).map(|entry| entry.fields.as_slice())
4679 }
4680}
4681
4682fn slot_offset(generation: u64) -> u64 {
4687 16 + (generation - 1) % 2 * SLOT_BYTES as u64
4688}
4689
4690fn slot_bytes(path: impl AsRef<Path>) -> Result<(File, u64, Slot, Vec<u8>, Opening)> {
4695 let mut file = File::open(path).map_err(io)?;
4696 let size = file.metadata().map_err(io)?.len();
4697 if size < HEADER {
4698 return Err(invalid("file is shorter than its header"));
4699 }
4700 let mut header = [0; HEADER as usize];
4701 file.read_exact(&mut header).map_err(io)?;
4702 let mut opening = Opening { reads: 1, bytes: HEADER };
4703 let version = u32::from_le_bytes([header[8], header[9], header[10], header[11]]);
4704 if &header[..8] != MAGIC {
4709 return Err(invalid("the header does not begin with a rudb native magic"));
4710 }
4711 if !READABLE.contains(&version) {
4712 return Err(invalid(&format!(
4713 "the file is format {version} and this build reads format {FORMAT}, so it has to \
4714 be written again"
4715 )));
4716 }
4717 let mut selected = None;
4718 for start in [16, 16 + SLOT_BYTES] {
4719 let slot = Slot::read(&header[start..start + SLOT_BYTES]);
4720 if slot.generation == 0 || slot.length == 0 || slot.length as usize > MAX_DIRECTORY {
4721 continue;
4722 }
4723 let Some(end) = slot.offset.checked_add(u64::from(slot.length)) else { continue };
4724 if slot.offset < HEADER || end > size {
4725 continue;
4726 }
4727 let mut bytes = vec![0; slot.length as usize];
4728 file.seek(SeekFrom::Start(slot.offset)).map_err(io)?;
4729 file.read_exact(&mut bytes).map_err(io)?;
4730 opening.reads += 1;
4731 opening.bytes += u64::from(slot.length);
4732 if checksum(&bytes) == slot.hash
4733 && selected
4734 .as_ref()
4735 .is_none_or(|(old, _): &(Slot, Vec<u8>)| old.generation < slot.generation)
4736 {
4737 selected = Some((slot, bytes));
4738 }
4739 }
4740 let (slot, bytes) = selected.ok_or_else(|| invalid("no committed directory slot is valid"))?;
4741 Ok((file, size, slot, bytes, opening))
4742}
4743
4744impl Reader {
4745 pub fn open(path: impl AsRef<Path>) -> Result<Self> {
4752 let catalog = Catalog::open(path)?;
4753 let mut names = catalog.names();
4754 let name = names.next().ok_or_else(|| invalid("the file holds no table"))?.to_string();
4755 if names.next().is_some() {
4756 return Err(invalid(
4757 "the file holds more than one table, so it has to be opened by name",
4758 ));
4759 }
4760 catalog.table(&name)
4761 }
4762
4763 fn build(
4765 file: Arc<File>,
4766 size: u64,
4767 table: Table,
4768 directory: u64,
4769 opening: Opening,
4770 pool: PagePool,
4771 ) -> Result<Self> {
4772 let places = places(&table)?;
4773 let dictionaries = (0..table.fields.len()).map(|_| OnceLock::new()).collect();
4774 let table_fields = table.fields.len();
4775 let stripes = table.stripes.len();
4776 let columns = (0..table.fields.len())
4777 .map(|_| {
4778 Mutex::new(Cached {
4779 pages: (0..stripes).map(|_| None).collect(),
4780 index: (0..stripes).map(|_| None).collect(),
4781 ..Cached::default()
4782 })
4783 })
4784 .collect::<Vec<_>>();
4785 let cache = Shelf {
4786 columns,
4787 held: (0..table_fields).map(|_| AtomicUsize::new(0)).collect(),
4788 kept: AtomicUsize::new(CACHED_STRIPES_PER_COLUMN),
4789 };
4790 let sieves: Vec<Vec<SieveSlot>> = (0..table.fields.len())
4791 .map(|_| table.stripes.iter().map(|_| OnceLock::new()).collect())
4792 .collect();
4793 let part_ranges: Vec<Vec<RangeSlot>> = (0..table.fields.len())
4794 .map(|_| table.stripes.iter().map(|_| OnceLock::new()).collect())
4795 .collect();
4796 Ok(Self {
4797 file,
4798 table: Arc::new(table),
4799 dictionaries: Arc::new(dictionaries),
4800 loading: Arc::new((0..table_fields).map(|_| Mutex::new(())).collect()),
4801 frequency_values: Arc::new((0..table_fields).map(|_| OnceLock::new()).collect()),
4802 frequency_summaries: Arc::new((0..table_fields).map(|_| OnceLock::new()).collect()),
4803 opened: Arc::new(AtomicUsize::new(0)),
4804 sieves: Arc::new(sieves),
4805 part_ranges: Arc::new(part_ranges),
4806 places: Arc::new(places),
4807 cache: Arc::new(cache),
4808 pool,
4809 pages: Arc::new(AtomicUsize::new(0)),
4810 indexes: Arc::new(AtomicUsize::new(0)),
4811 size,
4812 directory,
4813 opening,
4814 })
4815 }
4816
4817 #[must_use]
4824 pub fn reads(&self) -> Reads {
4825 Reads {
4826 opening: self.opening,
4827 pages: self.pages.load(Atomic::Relaxed),
4828 indexes: self.indexes.load(Atomic::Relaxed),
4829 dictionaries: self.opened.load(Atomic::Relaxed),
4830 }
4831 }
4832
4833 #[must_use]
4838 pub fn layout(&self) -> Layout {
4839 let table = &self.table;
4840 let stripes = table.stripes.as_slice();
4841 let columns = table
4842 .fields
4843 .iter()
4844 .enumerate()
4845 .map(|(at, field)| ColumnLayout {
4846 name: field.name.clone(),
4847 kind: field.ty.to_string(),
4848 pages: sum(stripes.iter().map(|stripe| span_bytes(&stripe.pages, at))),
4849 memberships: sum(stripes.iter().map(|stripe| stripe.memberships.bytes(at))),
4850 sieves: sum(stripes.iter().map(|stripe| stripe.sieves.bytes(at))),
4851 part_ranges: sum(stripes.iter().map(|stripe| stripe.part_ranges.bytes(at))),
4852 dictionary: dictionary_bytes(table, at),
4853 })
4854 .collect();
4855 Layout {
4856 file: self.size,
4857 rows: table.rows,
4858 stripes: stripes.len(),
4859 parts: self.places.len(),
4860 columns,
4861 indexes: sum(stripes.iter().map(|stripe| u64::from(stripe.index.length))),
4862 directory: self.directory,
4863 header: HEADER,
4864 }
4865 }
4866
4867 pub fn stored(&self, column: usize) -> Result<Vec<StoredPart>> {
4884 let field = self
4885 .table
4886 .fields
4887 .get(column)
4888 .ok_or_else(|| invalid("stored column index out of range"))?;
4889 let mut stored = Vec::with_capacity(self.places.len());
4890 let mut row = 0;
4891 for (at, stripe) in self.table.stripes.iter().enumerate() {
4892 let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
4893 let index = read_index(&self.file, stripe, column)?;
4894 let mut bytes = vec![0; page.length as usize];
4895 read_at(&self.file, page.offset, &mut bytes)?;
4896 let ranges = self.stripe_part_ranges(at, column);
4897 for (part, &rows) in stripe.parts.iter().enumerate() {
4898 let span = *index.get(part).ok_or_else(|| invalid("part index out of range"))?;
4899 let held = part_bytes(&bytes, span)?;
4900 let range = ranges.and_then(|held| held.get(part));
4901 stored.push(StoredPart {
4902 stripe: at,
4903 part,
4904 row,
4905 rows: rows as usize,
4906 encoding: page_encoding(&field.ty, rows as usize, held),
4907 bytes: span.length as u64,
4908 page: page.offset,
4909 offset: span.start as u64,
4910 low: range
4911 .and_then(|range| range.low.clone())
4912 .and_then(|bound| bound.into_value(&field.ty)),
4913 high: range
4914 .and_then(|range| range.high.clone())
4915 .and_then(|bound| bound.into_value(&field.ty)),
4916 nulls: range.map(|range| range.nulls),
4917 });
4918 row += rows as usize;
4919 }
4920 }
4921 Ok(stored)
4922 }
4923
4924 #[must_use]
4926 pub fn parts(&self) -> usize {
4927 self.places.len()
4928 }
4929
4930 #[must_use]
4937 pub fn stripe_parts(&self) -> Vec<std::ops::Range<usize>> {
4938 let mut runs = Vec::with_capacity(self.table.stripes.len());
4939 let mut start = 0;
4940 for stripe in &self.table.stripes {
4941 let end = start + stripe.parts.len();
4942 runs.push(start..end);
4943 start = end;
4944 }
4945 runs
4946 }
4947
4948 #[must_use]
4953 pub fn stripe_rows(&self, stripe: usize) -> usize {
4954 self.table.stripes.get(stripe).map_or(0, |held| held.rows)
4955 }
4956
4957 pub fn keep_stripes(&self, stripes: usize) {
4964 self.cache.kept.fetch_max(stripes, Atomic::Relaxed);
4965 }
4966
4967 #[must_use]
4969 pub fn part_rows(&self, at: usize) -> usize {
4970 self.places.get(at).map_or(0, |place| place.rows as usize)
4971 }
4972
4973 #[must_use]
4975 pub fn table(&self) -> &Table {
4976 &self.table
4977 }
4978
4979 pub fn top_frequencies(&self, column: usize, top: usize) -> Result<Option<Vec<(Value, u64)>>> {
4988 let field = self
4989 .table
4990 .fields
4991 .get(column)
4992 .ok_or_else(|| invalid("frequency column index out of range"))?;
4993 let Some(summary) = self.frequency_summary(column)? else {
4994 return Ok(None);
4995 };
4996 if top == 0 || summary.entries.len() < top {
4997 return Ok(None);
4998 }
4999 let boundary = summary.entries[top - 1].count;
5000 if boundary <= summary.omitted_max {
5001 return Ok(None);
5002 }
5003 self.decode_frequencies(column, &field.ty, &summary.entries).map(Some)
5004 }
5005
5006 pub fn top_pair_frequencies(
5017 &self,
5018 first: usize,
5019 second: usize,
5020 top: usize,
5021 ) -> Result<Option<PairFrequencyCounts>> {
5022 if first >= self.table.fields.len() || second >= self.table.fields.len() {
5023 return Err(invalid("pair frequency column index out of range"));
5024 }
5025 let Some(summary) =
5026 self.table.pair_frequencies.iter().find(|summary| {
5027 summary.first as usize == first && summary.second as usize == second
5028 })
5029 else {
5030 return Ok(None);
5031 };
5032 if top == 0 || summary.entries.len() < top {
5033 return Ok(None);
5034 }
5035 let boundary = summary.entries[top - 1].count;
5036 if boundary <= summary.omitted_max {
5037 return Ok(None);
5038 }
5039 let first_summary = self
5040 .frequency_summary(first)?
5041 .ok_or_else(|| invalid("pair frequency first column has no synopsis"))?;
5042 let anchors = self
5043 .decode_frequencies(first, &self.table.fields[first].ty, &first_summary.entries)?
5044 .into_iter()
5045 .map(|(value, _)| value)
5046 .collect::<Vec<_>>();
5047 let dictionary = self
5048 .dictionary(second)?
5049 .ok_or_else(|| invalid("pair frequency second column has no dictionary"))?;
5050 let mut codes = summary.entries.iter().filter_map(|entry| entry.second).collect::<Vec<_>>();
5051 codes.sort_unstable();
5052 codes.dedup();
5053 let texts = dictionary
5054 .try_values_visited(&codes.iter().map(|&code| code as usize).collect::<Vec<_>>())?;
5055 let mut out = Vec::with_capacity(summary.entries.len());
5056 for entry in &summary.entries {
5057 if entry.count < boundary {
5058 break;
5059 }
5060 let first = anchors
5061 .get(entry.first_entry as usize)
5062 .cloned()
5063 .ok_or_else(|| invalid("pair frequency anchor is outside its values"))?;
5064 let second = match entry.second {
5065 None => Value::Null,
5066 Some(code) => {
5067 let at = codes
5068 .binary_search(&code)
5069 .map_err(|_| invalid("pair frequency code was not among the codes read"))?;
5070 texts[at].clone()
5071 }
5072 };
5073 out.push((vec![first, second], entry.count));
5074 }
5075 Ok(Some(out))
5076 }
5077
5078 pub fn exact_frequencies(&self, column: usize) -> Result<Option<Vec<(Value, u64)>>> {
5098 let Some(prefix) = self.frequency_prefix(column)? else {
5099 return Ok(None);
5100 };
5101 Ok((prefix.omitted_max == 0).then_some(prefix.entries))
5102 }
5103
5104 pub fn frequency_prefix(&self, column: usize) -> Result<Option<FrequencyPrefix>> {
5127 let field = self
5128 .table
5129 .fields
5130 .get(column)
5131 .ok_or_else(|| invalid("frequency column index out of range"))?;
5132 let Some(summary) = self.frequency_summary(column)? else {
5133 return Ok(None);
5134 };
5135 let entries = self.decode_frequencies(column, &field.ty, &summary.entries)?;
5136 Ok(Some(FrequencyPrefix { entries, omitted_max: summary.omitted_max }))
5137 }
5138
5139 fn frequency_summary(&self, column: usize) -> Result<Option<Cow<'_, FrequencySummary>>> {
5141 Ok(match self.table.frequencies.get(column) {
5142 None | Some(None) => None,
5143 Some(Some(Frequencies::Held(summary))) => Some(Cow::Borrowed(summary)),
5144 Some(Some(Frequencies::Stored { span, values })) => {
5145 let slot = self
5146 .frequency_summaries
5147 .get(column)
5148 .ok_or_else(|| invalid("frequency column index out of range"))?;
5149 if let Some(summary) = slot.get() {
5150 return Ok(Some(Cow::Borrowed(summary.as_ref())));
5151 }
5152 let field = self
5153 .table
5154 .fields
5155 .get(column)
5156 .ok_or_else(|| invalid("frequency column index out of range"))?;
5157 let mut bytes = vec![0; span.length as usize];
5158 read_at(&self.file, span.offset, &mut bytes)?;
5159 let summary =
5160 decode_summary(&mut Cursor::new(&bytes), field, self.table.rows, *values)?;
5161 let summary = summary.ok_or_else(|| invalid("a stored synopsis is missing"))?;
5162 let _ = slot.set(Arc::new(summary));
5163 Some(Cow::Borrowed(slot.get().expect("the decoded summary was stored").as_ref()))
5164 }
5165 })
5166 }
5167
5168 fn decode_frequencies(
5176 &self,
5177 column: usize,
5178 ty: &LogicalType,
5179 entries: &[FrequencyEntry],
5180 ) -> Result<Vec<(Value, u64)>> {
5181 if let Some(values) = self.frequency_values.get(column).and_then(OnceLock::get) {
5182 return Ok(values.as_ref().clone());
5183 }
5184 let values = self.decode_frequencies_once(column, ty, entries)?;
5185 if let Some(slot) = self.frequency_values.get(column) {
5186 let _ = slot.set(Arc::new(values.clone()));
5187 }
5188 Ok(values)
5189 }
5190
5191 fn decode_frequencies_once(
5192 &self,
5193 column: usize,
5194 ty: &LogicalType,
5195 entries: &[FrequencyEntry],
5196 ) -> Result<Vec<(Value, u64)>> {
5197 let stored_texts = self.table.frequency_texts.get(column).filter(|texts| !texts.is_empty());
5198 if stored_texts.is_some_and(|texts| texts.len() != entries.len()) {
5199 return Err(invalid("frequency text count differs from its synopsis"));
5200 }
5201 let dictionary = if *ty == LogicalType::Varchar && stored_texts.is_none() {
5202 self.dictionary(column)?
5203 } else {
5204 None
5205 };
5206 let mut codes = entries
5207 .iter()
5208 .filter_map(|entry| match entry.value {
5209 FrequencyValue::Code(code) => Some(code as usize),
5210 _ => None,
5211 })
5212 .collect::<Vec<_>>();
5213 codes.sort_unstable();
5214 codes.dedup();
5215 let texts = match &dictionary {
5216 Some(dictionary) if !codes.is_empty() => dictionary.try_values_visited(&codes)?,
5217 _ => Vec::new(),
5218 };
5219 let mut out = Vec::with_capacity(entries.len());
5220 for (entry_at, entry) in entries.iter().enumerate() {
5221 let value = match entry.value {
5222 FrequencyValue::Null => {
5223 if stored_texts.and_then(|texts| texts[entry_at].as_ref()).is_some() {
5224 return Err(invalid("a null frequency entry has text"));
5225 }
5226 Value::Null
5227 }
5228 FrequencyValue::Integer(value) => match *ty {
5229 LogicalType::TinyInt => Value::TinyInt(
5230 i8::try_from(value)
5231 .map_err(|_| invalid("frequency TINYINT is out of range"))?,
5232 ),
5233 LogicalType::UTinyInt => Value::UTinyInt(
5234 u8::try_from(value)
5235 .map_err(|_| invalid("frequency UTINYINT is out of range"))?,
5236 ),
5237 LogicalType::USmallInt => Value::USmallInt(
5238 u16::try_from(value)
5239 .map_err(|_| invalid("frequency USMALLINT is out of range"))?,
5240 ),
5241 LogicalType::UInteger => Value::UInteger(
5242 u32::try_from(value)
5243 .map_err(|_| invalid("frequency UINTEGER is out of range"))?,
5244 ),
5245 LogicalType::UBigInt => Value::UBigInt(
5246 u64::try_from(value)
5247 .map_err(|_| invalid("frequency UBIGINT is out of range"))?,
5248 ),
5249 LogicalType::SmallInt => Value::SmallInt(
5250 i16::try_from(value)
5251 .map_err(|_| invalid("frequency SMALLINT is out of range"))?,
5252 ),
5253 LogicalType::Integer => Value::Integer(
5254 i32::try_from(value)
5255 .map_err(|_| invalid("frequency INTEGER is out of range"))?,
5256 ),
5257 LogicalType::BigInt => Value::BigInt(
5258 i64::try_from(value)
5259 .map_err(|_| invalid("frequency BIGINT is out of range"))?,
5260 ),
5261 LogicalType::Date => Value::Date(
5262 i32::try_from(value)
5263 .map_err(|_| invalid("frequency DATE is out of range"))?,
5264 ),
5265 LogicalType::Timestamp => Value::Timestamp(
5266 i64::try_from(value)
5267 .map_err(|_| invalid("frequency TIMESTAMP is out of range"))?,
5268 ),
5269 _ => return Err(invalid("integer frequency belongs to another type")),
5270 },
5271 FrequencyValue::Code(code) => {
5272 if let Some(text) = stored_texts.and_then(|texts| texts[entry_at].as_ref()) {
5273 Value::Varchar(
5274 String::from_utf8(text.clone())
5275 .map_err(|_| invalid("frequency text is not UTF-8"))?,
5276 )
5277 } else {
5278 if dictionary.is_none() {
5279 return Err(invalid("frequency code has no dictionary or stored text"));
5280 }
5281 let at = codes
5282 .binary_search(&(code as usize))
5283 .map_err(|_| invalid("frequency code was not among the codes read"))?;
5284 texts[at].clone()
5285 }
5286 }
5287 };
5288 out.push((value, entry.count));
5289 }
5290 Ok(out)
5291 }
5292
5293 pub fn frequency_occurrences(&self, column: usize) -> Result<Option<FrequencyOccurrences>> {
5303 let field = self
5304 .table
5305 .fields
5306 .get(column)
5307 .ok_or_else(|| invalid("frequency column index out of range"))?;
5308 let Some(summary) = self.frequency_summary(column)? else {
5309 return Ok(None);
5310 };
5311 if summary.ordinals.is_empty() {
5312 return Ok(None);
5313 }
5314 let (anchors, anchor_indices) = if summary.ordinal_entries.len() == summary.ordinals.len() {
5315 let entries = self.decode_frequencies(column, &field.ty, &summary.entries)?;
5316 (entries.into_iter().map(|(value, _)| value).collect(), summary.ordinal_entries.clone())
5317 } else {
5318 (Vec::new(), Vec::new())
5319 };
5320 Ok(Some(FrequencyOccurrences {
5321 omitted_max: summary.omitted_max,
5322 ordinals: summary.ordinals.clone(),
5323 anchors,
5324 anchor_indices,
5325 }))
5326 }
5327
5328 pub fn distinct_values(&self, column: usize) -> Result<Option<u64>> {
5354 self.table
5355 .distincts
5356 .get(column)
5357 .copied()
5358 .ok_or_else(|| invalid("distinct column index out of range"))
5359 }
5360
5361 pub fn null_count(&self, column: usize) -> Result<u64> {
5372 if column >= self.table.fields.len() {
5373 return Err(invalid("null count column index out of range"));
5374 }
5375 let mut nulls = 0_u64;
5376 for stripe in &self.table.stripes {
5377 let range = stripe
5378 .zone
5379 .column(column)
5380 .ok_or_else(|| invalid("stripe zone is narrower than the schema"))?;
5381 nulls = nulls
5382 .checked_add(range.nulls as u64)
5383 .ok_or_else(|| invalid("null count overflow"))?;
5384 }
5385 Ok(nulls)
5386 }
5387
5388 pub fn text_extremes(&self, column: usize) -> Result<Option<(Value, Value)>> {
5403 if self.null_count(column)? > 0 {
5404 return Ok(None);
5405 }
5406 let Some(dictionary) = self.dictionary(column)? else { return Ok(None) };
5407 let Some(ranks) = dictionary.ranks() else { return Ok(None) };
5408 if ranks == 0 {
5409 return Ok(None);
5410 }
5411 let low = text_at_rank(&dictionary, 0)?;
5412 let high = text_at_rank(&dictionary, ranks - 1)?;
5413 Ok(Some((low, high)))
5414 }
5415
5416 pub fn exact_extremes(&self, column: usize) -> Result<Option<(Bound, Bound)>> {
5439 if column >= self.table.fields.len() {
5440 return Err(invalid("extremes column index out of range"));
5441 }
5442 let mut low: Option<Bound> = None;
5443 let mut high: Option<Bound> = None;
5444 for stripe in &self.table.stripes {
5445 let range = stripe
5446 .zone
5447 .column(column)
5448 .ok_or_else(|| invalid("stripe zone is narrower than the schema"))?;
5449 if !range.exact {
5450 return Ok(None);
5451 }
5452 let (Some(small), Some(large)) = (range.low.as_ref(), range.high.as_ref()) else {
5457 if stripe.rows > range.nulls {
5458 return Ok(None);
5459 }
5460 continue;
5461 };
5462 low = Some(low.map_or_else(|| small.clone(), |held| held.smaller(small.clone())));
5463 high = Some(high.map_or_else(|| large.clone(), |held| held.larger(large.clone())));
5464 }
5465 Ok(low.zip(high))
5466 }
5467
5468 pub fn exact_sum(&self, column: usize) -> Result<Option<(i128, u64)>> {
5481 if column >= self.table.fields.len() {
5482 return Err(invalid("sum column index out of range"));
5483 }
5484 let mut total = 0_i128;
5485 let mut rows = 0_u64;
5486 for stripe in &self.table.stripes {
5487 let range = stripe
5488 .zone
5489 .column(column)
5490 .ok_or_else(|| invalid("stripe zone is narrower than the schema"))?;
5491 let Some(part) = range.sum else { return Ok(None) };
5492 let Some(sum) = total.checked_add(part) else { return Ok(None) };
5493 total = sum;
5494 rows = rows.saturating_add(stripe.rows as u64 - range.nulls as u64);
5495 }
5496 Ok(Some((total, rows)))
5497 }
5498
5499 pub fn host_groups(
5502 &self,
5503 column: usize,
5504 minimum_count: u64,
5505 ) -> Result<Option<Vec<host::HostEntry>>> {
5506 if column >= self.table.fields.len() {
5507 return Err(invalid("host group column index out of range"));
5508 }
5509 let Some(summary) = &self.table.host_groups else { return Ok(None) };
5510 if summary.column != column || minimum_count <= summary.omitted_max {
5511 return Ok(None);
5512 }
5513 Ok(Some(summary.entries.clone()))
5514 }
5515
5516 fn dictionary(&self, column: usize) -> Result<Option<Arc<Vector>>> {
5525 let Some(page) = self.table.dictionaries[column] else { return Ok(None) };
5526 if let Some(dictionary) = self.dictionaries[column].get() {
5527 return Ok(Some(Arc::clone(dictionary)));
5528 }
5529 let _queued = self.loading[column].lock().map_err(|_| invalid("a poisoned dictionary"))?;
5530 if let Some(dictionary) = self.dictionaries[column].get() {
5531 return Ok(Some(Arc::clone(dictionary)));
5532 }
5533 self.opened.fetch_add(1, Atomic::Relaxed);
5534 let dictionary = Arc::new(open_global_dictionary(
5535 Arc::clone(&self.file),
5536 page,
5537 &self.table.fields[column].ty,
5538 TEXT_KEEP_BUDGET,
5539 )?);
5540 let _ = self.dictionaries[column].set(Arc::clone(&dictionary));
5541 Ok(Some(dictionary))
5542 }
5543
5544 pub fn extents(&self, of: &Section) -> Result<Vec<section::Extent>> {
5551 if of.extent_bytes == 0 {
5552 return Ok(Vec::new());
5553 }
5554 let mut bytes = vec![0; of.extent_bytes as usize];
5555 read_at(&self.file, of.extent_page, &mut bytes)?;
5556 if checksum(&bytes) != of.hash {
5557 return Err(invalid("a section's extent table does not checksum"));
5558 }
5559 let extents = section::decode_extents(&bytes)?;
5560 if extents.len() != of.extents as usize {
5561 return Err(invalid("a section's extent table is not the length the entry says"));
5562 }
5563 Ok(extents)
5564 }
5565
5566 pub fn extent(&self, of: §ion::Extent) -> Result<Vec<u8>> {
5576 let end = of
5577 .offset
5578 .checked_add(u64::from(of.length))
5579 .ok_or_else(|| invalid("an extent overflows the file"))?;
5580 if of.offset < HEADER || end > self.size {
5581 return Err(invalid("an extent is outside the file"));
5582 }
5583 let mut bytes = vec![0; of.length as usize];
5584 read_at(&self.file, of.offset, &mut bytes)?;
5585 if checksum(&bytes) != of.hash {
5586 return Err(invalid("an extent does not checksum"));
5587 }
5588 Ok(bytes)
5589 }
5590
5591 pub fn payload(&self, of: &Section) -> Result<Vec<u8>> {
5600 let extents = self.extents(of)?;
5601 let mut bytes =
5602 Vec::with_capacity(sum(extents.iter().map(|one| u64::from(one.length))) as usize);
5603 for one in &extents {
5604 if one.first != bytes.len() as u64 {
5605 return Err(invalid("a section's extents do not join up"));
5606 }
5607 bytes.extend_from_slice(&self.extent(one)?);
5608 }
5609 if !bytes.is_empty() && of.header_bytes as usize > bytes.len() {
5612 return Err(invalid("a section's header is longer than its payload"));
5613 }
5614 Ok(bytes)
5615 }
5616
5617 pub fn read(&self, part: usize, columns: &[usize]) -> Result<Chunk> {
5626 self.read_impl(part, columns, true)
5627 }
5628
5629 pub fn read_sparse(&self, part: usize, columns: &[usize]) -> Result<Chunk> {
5639 self.read_impl(part, columns, false)
5640 }
5641
5642 pub fn skips_codes(&self, part: usize, column: usize, candidates: &[u32]) -> Result<bool> {
5649 if candidates.is_empty() {
5650 return Ok(true);
5651 }
5652 if candidates.windows(2).any(|pair| pair[0] >= pair[1]) {
5653 return Err(Error::internal("native code candidates are not sorted and unique"));
5654 }
5655 let stripe = self.stripe_of(part)?;
5656 let Some(page) = stripe.memberships.get(column) else {
5657 return Ok(false);
5658 };
5659 let mut bytes = vec![0; page.length as usize];
5660 read_at(&self.file, page.offset, &mut bytes)?;
5661 if checksum(&bytes) != page.hash {
5662 return Err(invalid("membership page checksum differs"));
5663 }
5664 let codes = decode_membership(&bytes)?;
5665 let mut left = 0;
5666 let mut right = 0;
5667 while left < codes.len() && right < candidates.len() {
5668 match codes[left].cmp(&candidates[right]) {
5669 Ordering::Less => left += 1,
5670 Ordering::Greater => right += 1,
5671 Ordering::Equal => return Ok(false),
5672 }
5673 }
5674 Ok(true)
5675 }
5676
5677 fn stripe_of(&self, part: usize) -> Result<&Stripe> {
5678 let place = self.places.get(part).ok_or_else(|| invalid("part index out of range"))?;
5679 self.table
5680 .stripes
5681 .get(place.stripe as usize)
5682 .ok_or_else(|| invalid("stripe index out of range"))
5683 }
5684
5685 fn held(&self, at: usize, stripe: &Stripe, column: usize, whole: bool) -> Result<CachedColumn> {
5702 let cache =
5703 self.cache.columns.get(column).ok_or_else(|| invalid("column index out of range"))?;
5704 let mut cached = cache.lock().map_err(|_| invalid("column page cache is poisoned"))?;
5705 let known = cached.index.get(at).and_then(Clone::clone);
5706 let page = cached.pages.get(at).and_then(Option::as_ref).map(|slot| {
5707 slot.used.store(true, Atomic::Relaxed);
5708 Arc::clone(&slot.page)
5709 });
5710 if let Some(index) = known.clone() {
5711 if !whole || page.is_some() {
5712 return Ok(CachedColumn { stripe: at, index, page });
5713 }
5714 }
5715 if cached.loading.contains(&at) {
5716 drop(cached);
5717 if let Some(index) = known {
5721 return Ok(CachedColumn { stripe: at, index, page: None });
5722 }
5723 let held = self.page_of(stripe, column, at, false, None)?;
5724 let mut cached = cache.lock().map_err(|_| invalid("column page cache is poisoned"))?;
5725 remember(&mut cached, &held);
5726 return Ok(held);
5727 }
5728 cached.loading.push(at);
5729 drop(cached);
5730
5731 let read = self.page_of(stripe, column, at, whole, known);
5732
5733 let mut cached = cache.lock().map_err(|_| invalid("column page cache is poisoned"))?;
5737 if let Some(position) = cached.loading.iter().position(|loading| *loading == at) {
5738 cached.loading.remove(position);
5739 }
5740 let held = read?;
5741 let taken = remember(&mut cached, &held);
5742 drop(cached);
5743 if let Some((bytes, used)) = taken {
5744 self.cache.held[column].fetch_add(1, Atomic::Relaxed);
5745 self.pool.admit(Held {
5746 shelf: Arc::downgrade(&self.cache),
5747 column,
5748 stripe: at,
5749 bytes,
5750 used,
5751 });
5752 }
5753 Ok(held)
5754 }
5755
5756 fn page_of(
5762 &self,
5763 stripe: &Stripe,
5764 column: usize,
5765 at: usize,
5766 whole: bool,
5767 known: Option<Arc<Vec<PartSpan>>>,
5768 ) -> Result<CachedColumn> {
5769 let index = match known {
5770 Some(index) => index,
5771 None => {
5772 self.indexes.fetch_add(1, Atomic::Relaxed);
5773 Arc::new(read_index(&self.file, stripe, column)?)
5774 }
5775 };
5776 let page = if whole {
5777 self.pages.fetch_add(1, Atomic::Relaxed);
5778 let span = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
5779 let mut bytes = vec![0; span.length as usize];
5780 read_at(&self.file, span.offset, &mut bytes)?;
5781 Some(Arc::new(bytes))
5782 } else {
5783 None
5784 };
5785 Ok(CachedColumn { stripe: at, index, page })
5786 }
5787
5788 fn read_impl(&self, at: usize, columns: &[usize], whole: bool) -> Result<Chunk> {
5789 let place = *self.places.get(at).ok_or_else(|| invalid("part index out of range"))?;
5790 let index = place.stripe as usize;
5791 let stripe =
5792 self.table.stripes.get(index).ok_or_else(|| invalid("stripe index out of range"))?;
5793 let rows = place.rows as usize;
5794 let mut picked = Vec::with_capacity(columns.len());
5795 for &column in columns {
5796 let field = self
5797 .table
5798 .fields
5799 .get(column)
5800 .ok_or_else(|| invalid("column index out of range"))?;
5801 let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
5802 let held = self.held(index, stripe, column, whole)?;
5803 let span = *held
5804 .index
5805 .get(place.part as usize)
5806 .ok_or_else(|| invalid("part index out of range"))?;
5807 let owned;
5808 let bytes = match &held.page {
5809 Some(held) => part_bytes(held, span)?,
5810 None => {
5811 let offset = page
5812 .offset
5813 .checked_add(span.start as u64)
5814 .ok_or_else(|| invalid("part range overflow"))?;
5815 let mut bytes = vec![0; span.length];
5816 read_at(&self.file, offset, &mut bytes)?;
5817 owned = bytes;
5818 &owned
5819 }
5820 };
5821 if checksum(bytes) != span.hash {
5822 return Err(invalid(&format!(
5823 "column page checksum differs, column {column} part {} at {}+{} of {} bytes, \
5824 wanted {:016x} and got {:016x}",
5825 place.part,
5826 page.offset,
5827 span.start,
5828 span.length,
5829 span.hash,
5830 checksum(bytes),
5831 )));
5832 }
5833 let dictionary = self.dictionary(column)?;
5834 picked.push(decode(&field.ty, rows, bytes, dictionary)?.into_pages());
5840 }
5841 Chunk::with_rows(picked, rows)
5842 }
5843
5844 #[must_use]
5860 pub fn skips(&self, part: usize, probes: &[Probe]) -> bool {
5861 let Some(place) = self.places.get(part).copied() else { return false };
5862 let Some(stripe) = self.table.stripes.get(place.stripe as usize) else { return false };
5863 if stripe.zone.skips(probes) {
5864 return true;
5865 }
5866 probes.iter().any(|probe| self.outside(place, probe) || self.sifted(place, probe))
5867 }
5868
5869 fn outside(&self, place: Place, probe: &Probe) -> bool {
5875 match self.stripe_part_ranges(place.stripe as usize, probe.column) {
5876 Some(ranges) => ranges
5877 .get(place.part as usize)
5878 .is_some_and(|range| range.excludes(probe.op, &probe.value)),
5879 None => false,
5880 }
5881 }
5882
5883 fn stripe_part_ranges(&self, stripe: usize, column: usize) -> Option<&[Range]> {
5889 let slot = self.part_ranges.get(column)?.get(stripe)?;
5890 if let Some(held) = slot.get() {
5891 return Some(held);
5892 }
5893 let page = self.table.stripes.get(stripe)?.part_ranges.get(column)?;
5894 let mut bytes = vec![0; page.length as usize];
5895 read_at(&self.file, page.offset, &mut bytes).ok()?;
5896 if checksum(&bytes) != page.hash {
5897 return None;
5898 }
5899 let ranges = Arc::new(decode_part_ranges(&bytes).ok()?);
5900 let _ = slot.set(ranges);
5901 slot.get().map(|held| held.as_slice())
5902 }
5903
5904 #[must_use]
5921 pub fn certain(&self, part: usize, probes: &[Probe]) -> bool {
5922 let Some(place) = self.places.get(part).copied() else { return false };
5923 let Some(stripe) = self.table.stripes.get(place.stripe as usize) else { return false };
5924 if stripe.zone.certain(probes) {
5925 return true;
5926 }
5927 probes
5928 .iter()
5929 .all(|probe| stripe.zone.certain(slice::from_ref(probe)) || self.inside(place, probe))
5930 }
5931
5932 fn inside(&self, place: Place, probe: &Probe) -> bool {
5938 match self.stripe_part_ranges(place.stripe as usize, probe.column) {
5939 Some(ranges) => ranges
5940 .get(place.part as usize)
5941 .is_some_and(|range| range.certain(probe.op, &probe.value)),
5942 None => false,
5943 }
5944 }
5945
5946 #[must_use]
5957 pub fn stripe_skips(&self, stripe: usize, probes: &[Probe]) -> bool {
5958 self.table.stripes.get(stripe).is_some_and(|held| held.zone.skips(probes))
5959 }
5960
5961 fn sifted(&self, place: Place, probe: &Probe) -> bool {
5967 if probe.op != Op::Equal {
5968 return false;
5969 }
5970 match self.stripe_sieves(place.stripe as usize, probe.column) {
5971 Some(sieves) => sieves
5972 .get(place.part as usize)
5973 .and_then(Option::as_ref)
5974 .is_some_and(|sieve| sieve.excludes(&probe.value)),
5975 None => false,
5976 }
5977 }
5978
5979 fn stripe_sieves(&self, stripe: usize, column: usize) -> Option<&[Option<Sieve>]> {
5986 let slot = self.sieves.get(column)?.get(stripe)?;
5987 if let Some(held) = slot.get() {
5988 return Some(held);
5989 }
5990 let page = self.table.stripes.get(stripe)?.sieves.get(column)?;
5991 let mut bytes = vec![0; page.length as usize];
5992 read_at(&self.file, page.offset, &mut bytes).ok()?;
5993 if checksum(&bytes) != page.hash {
5994 return None;
5995 }
5996 let sieves = Arc::new(decode_sieves(&bytes).ok()?);
5997 let _ = slot.set(sieves);
5998 slot.get().map(|held| held.as_slice())
5999 }
6000}
6001
6002fn text_at_rank(dictionary: &Vector, rank: usize) -> Result<Value> {
6004 let code = dictionary.code_at_rank(rank)? as usize;
6005 let text = dictionary
6006 .try_text_at(code)?
6007 .ok_or_else(|| invalid("global dictionary order names a code it does not have"))?;
6008 Ok(Value::Varchar(text.into()))
6009}
6010
6011#[cfg(unix)]
6016fn write_at(file: &File, mut offset: u64, mut bytes: &[u8]) -> Result<()> {
6017 use std::os::unix::fs::FileExt;
6018 while !bytes.is_empty() {
6019 let written = file.write_at(bytes, offset).map_err(io)?;
6020 if written == 0 {
6021 return Err(invalid("a write to the native file wrote nothing"));
6022 }
6023 offset += written as u64;
6024 bytes = &bytes[written..];
6025 }
6026 Ok(())
6027}
6028
6029#[cfg(windows)]
6031fn write_at(file: &File, mut offset: u64, mut bytes: &[u8]) -> Result<()> {
6032 use std::os::windows::fs::FileExt;
6033 while !bytes.is_empty() {
6034 let written = file.seek_write(bytes, offset).map_err(io)?;
6035 if written == 0 {
6036 return Err(invalid("a write to the native file wrote nothing"));
6037 }
6038 offset += written as u64;
6039 bytes = &bytes[written..];
6040 }
6041 Ok(())
6042}
6043
6044#[cfg(not(any(unix, windows)))]
6046fn write_at(file: &File, offset: u64, bytes: &[u8]) -> Result<()> {
6047 use std::io::Write;
6048 let mut file = file.try_clone().map_err(io)?;
6049 file.seek(SeekFrom::Start(offset)).map_err(io)?;
6050 file.write_all(bytes).map_err(io)
6051}
6052
6053#[cfg(unix)]
6063fn read_at(file: &File, mut offset: u64, mut bytes: &mut [u8]) -> Result<()> {
6064 use std::os::unix::fs::FileExt;
6065 while !bytes.is_empty() {
6066 let read = file.read_at(bytes, offset).map_err(io)?;
6067 if read == 0 {
6068 return Err(invalid("column page ends before its declared length"));
6069 }
6070 offset += read as u64;
6071 bytes = &mut bytes[read..];
6072 }
6073 Ok(())
6074}
6075
6076#[cfg(windows)]
6082fn read_at(file: &File, mut offset: u64, mut bytes: &mut [u8]) -> Result<()> {
6083 use std::os::windows::fs::FileExt;
6084 while !bytes.is_empty() {
6085 let read = file.seek_read(bytes, offset).map_err(io)?;
6086 if read == 0 {
6087 return Err(invalid("column page ends before its declared length"));
6088 }
6089 offset += read as u64;
6090 bytes = &mut bytes[read..];
6091 }
6092 Ok(())
6093}
6094
6095#[cfg(not(any(unix, windows)))]
6100fn read_at(file: &File, offset: u64, bytes: &mut [u8]) -> Result<()> {
6101 let mut file = file.try_clone().map_err(io)?;
6102 file.seek(SeekFrom::Start(offset)).map_err(io)?;
6103 file.read_exact(bytes).map_err(io)
6104}
6105
6106fn type_tag(ty: &LogicalType) -> Result<u8> {
6113 match ty {
6114 LogicalType::SmallInt => Ok(1),
6115 LogicalType::Integer => Ok(2),
6116 LogicalType::BigInt => Ok(3),
6117 LogicalType::Varchar => Ok(4),
6118 LogicalType::Date => Ok(5),
6119 LogicalType::Timestamp => Ok(6),
6120 LogicalType::Boolean => Ok(7),
6121 LogicalType::TinyInt => Ok(8),
6122 LogicalType::UTinyInt => Ok(9),
6123 LogicalType::USmallInt => Ok(10),
6124 LogicalType::UInteger => Ok(11),
6125 LogicalType::UBigInt => Ok(12),
6126 LogicalType::Decimal { .. } => Ok(13),
6127 LogicalType::Float => Ok(14),
6128 LogicalType::Double => Ok(15),
6129 LogicalType::HugeInt => Ok(16),
6130 LogicalType::UHugeInt => Ok(17),
6131 LogicalType::Time => Ok(18),
6132 LogicalType::TimeTz => Ok(19),
6133 LogicalType::TimestampTz => Ok(20),
6134 LogicalType::Interval => Ok(21),
6135 LogicalType::Uuid => Ok(22),
6136 LogicalType::Blob => Ok(23),
6137 LogicalType::Bit => Ok(24),
6138 LogicalType::TimestampS => Ok(25),
6139 LogicalType::TimestampMs => Ok(26),
6140 LogicalType::TimestampNs => Ok(27),
6141 _ => Err(Error::not_implemented(format!("native storage for {ty}"))),
6142 }
6143}
6144
6145fn put_type(out: &mut Vec<u8>, ty: &LogicalType) -> Result<()> {
6151 out.push(type_tag(ty)?);
6152 if let LogicalType::Decimal { width, scale } = ty {
6153 out.push(*width);
6154 out.push(*scale);
6155 }
6156 Ok(())
6157}
6158
6159fn read_type(cur: &mut Cursor<'_>) -> Result<LogicalType> {
6161 let tag = cur.u8()?;
6162 if tag == 13 {
6163 let width = cur.u8()?;
6164 let scale = cur.u8()?;
6165 return LogicalType::decimal(width, scale)
6166 .map_err(|_| invalid("decimal column width and scale are not a decimal"));
6167 }
6168 tag_type(tag)
6169}
6170
6171fn tag_type(tag: u8) -> Result<LogicalType> {
6172 match tag {
6173 1 => Ok(LogicalType::SmallInt),
6174 2 => Ok(LogicalType::Integer),
6175 3 => Ok(LogicalType::BigInt),
6176 4 => Ok(LogicalType::Varchar),
6177 5 => Ok(LogicalType::Date),
6178 6 => Ok(LogicalType::Timestamp),
6179 7 => Ok(LogicalType::Boolean),
6180 8 => Ok(LogicalType::TinyInt),
6181 9 => Ok(LogicalType::UTinyInt),
6182 10 => Ok(LogicalType::USmallInt),
6183 11 => Ok(LogicalType::UInteger),
6184 12 => Ok(LogicalType::UBigInt),
6185 14 => Ok(LogicalType::Float),
6186 15 => Ok(LogicalType::Double),
6187 16 => Ok(LogicalType::HugeInt),
6188 17 => Ok(LogicalType::UHugeInt),
6189 18 => Ok(LogicalType::Time),
6190 19 => Ok(LogicalType::TimeTz),
6191 20 => Ok(LogicalType::TimestampTz),
6192 21 => Ok(LogicalType::Interval),
6193 22 => Ok(LogicalType::Uuid),
6194 23 => Ok(LogicalType::Blob),
6195 24 => Ok(LogicalType::Bit),
6196 25 => Ok(LogicalType::TimestampS),
6197 26 => Ok(LogicalType::TimestampMs),
6198 27 => Ok(LogicalType::TimestampNs),
6199 _ => Err(invalid("column type tag is unknown")),
6200 }
6201}
6202
6203fn put_u16(out: &mut Vec<u8>, value: u16) {
6204 out.extend_from_slice(&value.to_le_bytes());
6205}
6206fn put_u32(out: &mut Vec<u8>, value: u32) {
6207 out.extend_from_slice(&value.to_le_bytes());
6208}
6209fn put_u64(out: &mut Vec<u8>, value: u64) {
6210 out.extend_from_slice(&value.to_le_bytes());
6211}
6212fn put_var_u64(out: &mut Vec<u8>, mut value: u64) {
6213 while value >= 0x80 {
6214 out.push((value as u8 & 0x7f) | 0x80);
6215 value >>= 7;
6216 }
6217 out.push(value as u8);
6218}
6219
6220fn frequency_order(left: FrequencyValue, right: FrequencyValue) -> Ordering {
6221 match (left, right) {
6222 (FrequencyValue::Null, FrequencyValue::Null) => Ordering::Equal,
6223 (FrequencyValue::Null, _) => Ordering::Less,
6224 (_, FrequencyValue::Null) => Ordering::Greater,
6225 (FrequencyValue::Integer(left), FrequencyValue::Integer(right)) => left.cmp(&right),
6226 (FrequencyValue::Code(left), FrequencyValue::Code(right)) => left.cmp(&right),
6227 (FrequencyValue::Integer(_), FrequencyValue::Code(_)) => Ordering::Less,
6228 (FrequencyValue::Code(_), FrequencyValue::Integer(_)) => Ordering::Greater,
6229 }
6230}
6231
6232fn keep_most_frequent(entries: &mut Vec<FrequencyEntry>) -> u64 {
6245 let order = |left: &FrequencyEntry, right: &FrequencyEntry| {
6246 right.count.cmp(&left.count).then_with(|| frequency_order(left.value, right.value))
6247 };
6248 let omitted_max = if entries.len() > FREQUENCY_ENTRIES {
6249 let (_, next, _) = entries.select_nth_unstable_by(FREQUENCY_ENTRIES, order);
6250 let omitted_max = next.count;
6251 entries.truncate(FREQUENCY_ENTRIES);
6252 omitted_max
6253 } else {
6254 0
6255 };
6256 entries.sort_unstable_by(order);
6257 omitted_max
6258}
6259
6260fn code_frequency(
6261 dictionary: &GlobalDictionary,
6262 flat: &[u8],
6263 bases: &[u64],
6264) -> Result<(FrequencySummary, Vec<Option<Vec<u8>>>)> {
6265 let mut entries = dictionary
6266 .counts
6267 .iter()
6268 .enumerate()
6269 .filter(|(_, count)| **count != 0)
6270 .map(|(code, &count)| FrequencyEntry { value: FrequencyValue::Code(code as u32), count })
6271 .collect::<Vec<_>>();
6272 if dictionary.nulls != 0 {
6273 entries.push(FrequencyEntry { value: FrequencyValue::Null, count: dictionary.nulls });
6274 }
6275 let omitted_max = keep_most_frequent(&mut entries);
6276 let mut spans = Vec::with_capacity(entries.len());
6277 let mut text_bytes = 0_usize;
6278 for entry in &entries {
6279 let span = match entry.value {
6280 FrequencyValue::Code(code) => {
6281 let span = GlobalDictionary::value_span(&dictionary.ends, bases, code as usize);
6282 let bytes = flat
6283 .get(span.0..span.1)
6284 .ok_or_else(|| invalid("a frequency code is outside its dictionary"))?;
6285 text_bytes = text_bytes.saturating_add(bytes.len());
6286 Some(span)
6287 }
6288 FrequencyValue::Null | FrequencyValue::Integer(_) => None,
6289 };
6290 spans.push(span);
6291 }
6292 let texts = if text_bytes > FREQUENCY_TEXT_BUDGET {
6293 Vec::new()
6294 } else {
6295 spans.into_iter().map(|span| span.map(|(from, to)| flat[from..to].to_vec())).collect()
6296 };
6297 Ok((
6298 FrequencySummary {
6299 entries,
6300 omitted_max,
6301 ordinals: Vec::new(),
6302 ordinal_entries: Vec::new(),
6303 },
6304 texts,
6305 ))
6306}
6307
6308fn encode_directory(table: &Table) -> Result<Vec<u8>> {
6309 let mut out = DIRECTORY.to_vec();
6310 let name = table.name.as_bytes();
6311 put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("table name too long"))?);
6312 out.extend_from_slice(name);
6313 put_u16(&mut out, u16::try_from(table.fields.len()).map_err(|_| invalid("too many columns"))?);
6314 for field in &table.fields {
6315 let name = field.name.as_bytes();
6316 put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("column name too long"))?);
6317 out.extend_from_slice(name);
6318 put_type(&mut out, &field.ty)?;
6319 out.push(u8::from(field.not_null));
6320 }
6321 for dictionary in &table.dictionaries {
6322 match dictionary {
6323 None => out.push(0),
6324 Some(page) => {
6325 out.push(1);
6326 put_u64(&mut out, page.offset);
6327 put_u32(&mut out, page.length);
6328 put_u64(&mut out, page.hash);
6329 }
6330 }
6331 }
6332 for distinct in &table.distincts {
6333 match distinct {
6334 None => out.push(0),
6335 Some(count) => {
6336 out.push(1);
6337 put_u64(&mut out, *count);
6338 }
6339 }
6340 }
6341 put_u64(&mut out, u64::try_from(table.rows).map_err(|_| invalid("row count overflow"))?);
6342 put_u32(&mut out, u32::try_from(table.stripes.len()).map_err(|_| invalid("too many stripes"))?);
6343 for stripe in &table.stripes {
6344 put_u32(
6345 &mut out,
6346 u32::try_from(stripe.parts.len()).map_err(|_| invalid("too many parts in a stripe"))?,
6347 );
6348 for &rows in &stripe.parts {
6349 put_u32(&mut out, rows);
6350 }
6351 put_u64(&mut out, stripe.index.offset);
6352 put_u32(&mut out, stripe.index.length);
6353 for page in &stripe.pages {
6354 put_u64(&mut out, page.offset);
6355 put_u32(&mut out, page.length);
6356 }
6357 for ((field, dictionary), membership) in
6362 table.fields.iter().zip(&table.dictionaries).zip(stripe.memberships.slots())
6363 {
6364 if field.ty != LogicalType::Varchar || dictionary.is_none() {
6365 continue;
6366 }
6367 let page =
6368 membership.ok_or_else(|| invalid("string page has no code membership index"))?;
6369 put_u64(&mut out, page.offset);
6370 put_u32(&mut out, page.length);
6371 put_u64(&mut out, page.hash);
6372 }
6373 for sieve in stripe.sieves.slots() {
6374 match sieve {
6375 None => out.push(0),
6376 Some(page) => {
6377 out.push(1);
6378 put_u64(&mut out, page.offset);
6379 put_u32(&mut out, page.length);
6380 put_u64(&mut out, page.hash);
6381 }
6382 }
6383 }
6384 for held in stripe.part_ranges.slots() {
6385 match held {
6386 None => out.push(0),
6387 Some(page) => {
6388 out.push(1);
6389 put_u64(&mut out, page.offset);
6390 put_u32(&mut out, page.length);
6391 put_u64(&mut out, page.hash);
6392 }
6393 }
6394 }
6395 for range in stripe.zone.columns() {
6396 put_bound(&mut out, range.low.as_ref())?;
6397 put_bound(&mut out, range.high.as_ref())?;
6398 put_u32(
6399 &mut out,
6400 u32::try_from(range.nulls).map_err(|_| invalid("null count overflow"))?,
6401 );
6402 out.push(u8::from(range.exact));
6403 match range.sum {
6404 None => out.push(0),
6405 Some(total) => {
6406 out.push(1);
6407 out.extend_from_slice(&total.to_le_bytes());
6408 }
6409 }
6410 }
6411 }
6412 out.extend_from_slice(FREQUENCIES);
6413 put_u16(
6414 &mut out,
6415 u16::try_from(table.frequencies.len())
6416 .map_err(|_| invalid("too many frequency columns"))?,
6417 );
6418 for summary in &table.frequencies {
6419 let summary = match summary {
6420 None => {
6421 out.push(0);
6422 continue;
6423 }
6424 Some(Frequencies::Held(summary)) => summary,
6425 Some(Frequencies::Stored { .. }) => {
6427 return Err(invalid("a synopsis left in the file cannot be written back"));
6428 }
6429 };
6430 out.push(1);
6431 put_u64(&mut out, summary.omitted_max);
6432 put_u32(
6433 &mut out,
6434 u32::try_from(summary.entries.len())
6435 .map_err(|_| invalid("too many frequency entries"))?,
6436 );
6437 for entry in &summary.entries {
6438 match entry.value {
6439 FrequencyValue::Null => out.push(0),
6440 FrequencyValue::Integer(value) => {
6441 out.push(1);
6442 out.extend_from_slice(&value.to_le_bytes());
6443 }
6444 FrequencyValue::Code(value) => {
6445 out.push(2);
6446 put_u32(&mut out, value);
6447 }
6448 }
6449 put_u64(&mut out, entry.count);
6450 }
6451 put_u32(
6452 &mut out,
6453 u32::try_from(summary.ordinals.len())
6454 .map_err(|_| invalid("too many frequency ordinals"))?,
6455 );
6456 let mut previous = 0_u64;
6457 for (at, &ordinal) in summary.ordinals.iter().enumerate() {
6458 let delta = if at == 0 {
6459 ordinal
6460 } else {
6461 ordinal
6462 .checked_sub(previous)
6463 .ok_or_else(|| invalid("frequency ordinals are not ordered"))?
6464 };
6465 if at != 0 && delta == 0 {
6466 return Err(invalid("frequency ordinals are not unique"));
6467 }
6468 put_var_u64(&mut out, delta);
6469 previous = ordinal;
6470 }
6471 if summary.ordinal_entries.len() != summary.ordinals.len() {
6472 return Err(invalid("frequency ordinal values have a different length"));
6473 }
6474 for &entry in &summary.ordinal_entries {
6475 if entry as usize >= summary.entries.len() {
6476 return Err(invalid("frequency ordinal value is outside its entries"));
6477 }
6478 put_u16(&mut out, entry);
6479 }
6480 }
6481 if !table.pair_frequencies.is_empty() {
6482 out.extend_from_slice(PAIR_FREQUENCIES);
6483 put_u16(
6484 &mut out,
6485 u16::try_from(table.pair_frequencies.len())
6486 .map_err(|_| invalid("too many pair frequency summaries"))?,
6487 );
6488 for summary in &table.pair_frequencies {
6489 put_u16(&mut out, summary.first);
6490 put_u16(&mut out, summary.second);
6491 put_u64(&mut out, summary.omitted_max);
6492 put_u16(
6493 &mut out,
6494 u16::try_from(summary.entries.len())
6495 .map_err(|_| invalid("too many pair frequency entries"))?,
6496 );
6497 for entry in &summary.entries {
6498 put_u16(&mut out, entry.first_entry);
6499 match entry.second {
6500 None => out.push(0),
6501 Some(code) => {
6502 out.push(1);
6503 put_u32(&mut out, code);
6504 }
6505 }
6506 put_u64(&mut out, entry.count);
6507 }
6508 }
6509 }
6510 let text_columns = table.frequency_texts.iter().filter(|texts| !texts.is_empty()).count();
6511 if text_columns != 0 {
6512 out.extend_from_slice(FREQUENCY_TEXTS);
6513 put_u16(
6514 &mut out,
6515 u16::try_from(text_columns)
6516 .map_err(|_| invalid("too many string frequency columns"))?,
6517 );
6518 for (column, texts) in table.frequency_texts.iter().enumerate() {
6519 if texts.is_empty() {
6520 continue;
6521 }
6522 put_u16(
6523 &mut out,
6524 u16::try_from(column).map_err(|_| invalid("frequency text column overflows"))?,
6525 );
6526 put_u16(
6527 &mut out,
6528 u16::try_from(texts.len())
6529 .map_err(|_| invalid("too many frequency text entries"))?,
6530 );
6531 for text in texts {
6532 match text {
6533 None => out.push(0),
6534 Some(text) => {
6535 out.push(1);
6536 put_u32(
6537 &mut out,
6538 u32::try_from(text.len())
6539 .map_err(|_| invalid("frequency text is too long"))?,
6540 );
6541 out.extend_from_slice(text);
6542 }
6543 }
6544 }
6545 }
6546 }
6547 if let Some(summary) = &table.host_groups {
6548 out.extend_from_slice(HOST_GROUPS);
6549 put_u16(
6550 &mut out,
6551 u16::try_from(summary.column).map_err(|_| invalid("host column overflows"))?,
6552 );
6553 put_u64(&mut out, summary.omitted_max);
6554 put_u16(
6555 &mut out,
6556 u16::try_from(summary.entries.len()).map_err(|_| invalid("too many host groups"))?,
6557 );
6558 for entry in &summary.entries {
6559 put_u32(
6560 &mut out,
6561 u32::try_from(entry.host.len()).map_err(|_| invalid("host name is too long"))?,
6562 );
6563 out.extend_from_slice(entry.host.as_bytes());
6564 put_u64(&mut out, entry.count);
6565 out.extend_from_slice(&entry.bytes_sum.to_le_bytes());
6566 put_u32(
6567 &mut out,
6568 u32::try_from(entry.minimum.len())
6569 .map_err(|_| invalid("host minimum is too long"))?,
6570 );
6571 out.extend_from_slice(entry.minimum.as_bytes());
6572 }
6573 }
6574 if let Some(clustering) = &table.clustering {
6577 out.extend_from_slice(CLUSTERING);
6578 out.push(clustering.width().tag());
6579 put_u16(
6580 &mut out,
6581 u16::try_from(clustering.columns().len())
6582 .map_err(|_| invalid("too many clustering columns"))?,
6583 );
6584 for &column in clustering.columns() {
6585 put_u16(
6586 &mut out,
6587 u16::try_from(column).map_err(|_| invalid("clustering column index overflow"))?,
6588 );
6589 }
6590 }
6591 out.extend_from_slice(SECTIONS);
6597 put_u64(&mut out, table.generation);
6598 put_u16(
6599 &mut out,
6600 u16::try_from(table.sections.len()).map_err(|_| invalid("too many sections"))?,
6601 );
6602 for held in &table.sections {
6603 held.encode(&mut out)?;
6604 }
6605 if table.dictionary_payloads.iter().any(|&bytes| bytes != 0) {
6606 out.extend_from_slice(DICTIONARY_PAYLOADS);
6607 put_u16(
6608 &mut out,
6609 u16::try_from(table.fields.len()).map_err(|_| invalid("too many columns"))?,
6610 );
6611 for at in 0..table.fields.len() {
6612 put_u64(&mut out, table.dictionary_payloads.get(at).copied().unwrap_or(0));
6613 }
6614 }
6615 Ok(out)
6616}
6617
6618fn table_nonzero_counts(table: &Table) -> Vec<Option<u64>> {
6627 table
6628 .fields
6629 .iter()
6630 .enumerate()
6631 .map(|(column, field)| {
6632 if !matches!(
6633 field.ty,
6634 LogicalType::TinyInt
6635 | LogicalType::SmallInt
6636 | LogicalType::Integer
6637 | LogicalType::BigInt
6638 | LogicalType::UTinyInt
6639 | LogicalType::USmallInt
6640 | LogicalType::UInteger
6641 | LogicalType::UBigInt
6642 ) {
6643 return None;
6644 }
6645 let Some(Frequencies::Held(summary)) = &table.frequencies[column] else {
6646 return None;
6647 };
6648 let zero = summary
6649 .entries
6650 .iter()
6651 .find(|entry| entry.value == FrequencyValue::Integer(0))
6652 .map(|entry| entry.count)
6653 .or_else(|| (summary.omitted_max == 0).then_some(0))?;
6654 let nulls = table.stripes.iter().try_fold(0_u64, |count, stripe| {
6655 count.checked_add(stripe.zone.column(column)?.nulls as u64)
6656 })?;
6657 (table.rows as u64).checked_sub(nulls)?.checked_sub(zero)
6658 })
6659 .collect()
6660}
6661
6662fn signed_integer(ty: &LogicalType) -> bool {
6663 matches!(
6664 ty,
6665 LogicalType::TinyInt | LogicalType::SmallInt | LogicalType::Integer | LogicalType::BigInt
6666 )
6667}
6668
6669fn table_exact_sum(table: &Table, column: usize) -> Option<(i128, u64)> {
6670 table.stripes.iter().try_fold((0_i128, 0_u64), |(sum, count), stripe| {
6671 let range = stripe.zone.column(column)?;
6672 let sum = sum.checked_add(range.sum?)?;
6673 let nonnull = (stripe.rows as u64).checked_sub(range.nulls as u64)?;
6674 Some((sum, count.checked_add(nonnull)?))
6675 })
6676}
6677
6678fn table_aggregate_sums(table: &Table) -> Vec<Option<(i128, u64)>> {
6679 table
6680 .fields
6681 .iter()
6682 .enumerate()
6683 .map(|(column, field)| {
6684 signed_integer(&field.ty).then(|| table_exact_sum(table, column)).flatten()
6685 })
6686 .collect()
6687}
6688
6689fn reader_nonzero_counts(reader: &Reader) -> Result<Vec<Option<u64>>> {
6690 reader
6691 .table
6692 .fields
6693 .iter()
6694 .enumerate()
6695 .map(|(column, field)| {
6696 if !matches!(
6697 field.ty,
6698 LogicalType::TinyInt
6699 | LogicalType::SmallInt
6700 | LogicalType::Integer
6701 | LogicalType::BigInt
6702 | LogicalType::UTinyInt
6703 | LogicalType::USmallInt
6704 | LogicalType::UInteger
6705 | LogicalType::UBigInt
6706 ) {
6707 return Ok(None);
6708 }
6709 let Some(summary) = reader.frequency_summary(column)? else {
6710 return Ok(None);
6711 };
6712 let zero = summary
6713 .entries
6714 .iter()
6715 .find(|entry| entry.value == FrequencyValue::Integer(0))
6716 .map(|entry| entry.count)
6717 .or_else(|| (summary.omitted_max == 0).then_some(0));
6718 let Some(zero) = zero else { return Ok(None) };
6719 let nulls = reader.null_count(column)?;
6720 Ok((reader.table.rows as u64)
6721 .checked_sub(nulls)
6722 .and_then(|count| count.checked_sub(zero)))
6723 })
6724 .collect()
6725}
6726
6727fn reader_aggregate_sums(reader: &Reader) -> Result<Vec<Option<(i128, u64)>>> {
6728 reader
6729 .table
6730 .fields
6731 .iter()
6732 .enumerate()
6733 .map(
6734 |(column, field)| {
6735 if signed_integer(&field.ty) { reader.exact_sum(column) } else { Ok(None) }
6736 },
6737 )
6738 .collect()
6739}
6740
6741fn encode_catalog(entries: &[Entry], views: &[ViewEntry]) -> Result<Vec<u8>> {
6742 let mut out = CATALOG.to_vec();
6743 put_u32(&mut out, u32::try_from(entries.len()).map_err(|_| invalid("too many tables"))?);
6744 for entry in entries {
6745 let name = entry.name.as_bytes();
6746 put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("table name too long"))?);
6747 out.extend_from_slice(name);
6748 put_u64(&mut out, u64::try_from(entry.rows).map_err(|_| invalid("row count overflow"))?);
6749 put_u16(
6750 &mut out,
6751 u16::try_from(entry.fields.len()).map_err(|_| invalid("too many columns"))?,
6752 );
6753 for field in &entry.fields {
6754 let name = field.name.as_bytes();
6755 put_u16(
6756 &mut out,
6757 u16::try_from(name.len()).map_err(|_| invalid("column name too long"))?,
6758 );
6759 out.extend_from_slice(name);
6760 put_type(&mut out, &field.ty)?;
6761 out.push(u8::from(field.not_null));
6762 }
6763 put_u64(&mut out, entry.directory.offset);
6764 put_u32(&mut out, entry.directory.length);
6765 put_u64(&mut out, entry.directory.hash);
6766 }
6767 put_u32(&mut out, u32::try_from(views.len()).map_err(|_| invalid("too many views"))?);
6768 for view in views {
6769 let name = view.name.as_bytes();
6770 put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("view name too long"))?);
6771 out.extend_from_slice(name);
6772 put_long_text(&mut out, &view.sql, "view body")?;
6773 put_long_text(&mut out, &view.statement, "view statement")?;
6774 put_u16(
6775 &mut out,
6776 u16::try_from(view.aliases.len()).map_err(|_| invalid("too many aliases"))?,
6777 );
6778 for alias in &view.aliases {
6779 let alias = alias.as_bytes();
6780 put_u16(
6781 &mut out,
6782 u16::try_from(alias.len()).map_err(|_| invalid("alias name too long"))?,
6783 );
6784 out.extend_from_slice(alias);
6785 }
6786 put_u16(
6787 &mut out,
6788 u16::try_from(view.columns.len()).map_err(|_| invalid("too many columns"))?,
6789 );
6790 for field in &view.columns {
6791 let name = field.name.as_bytes();
6792 put_u16(
6793 &mut out,
6794 u16::try_from(name.len()).map_err(|_| invalid("column name too long"))?,
6795 );
6796 out.extend_from_slice(name);
6797 put_type(&mut out, &field.ty)?;
6798 out.push(u8::from(field.not_null));
6799 }
6800 }
6801 out.extend_from_slice(NONZERO_COUNTS);
6802 for entry in entries {
6803 if entry.nonzero.len() != entry.fields.len() {
6804 return Err(invalid("nonzero count width differs from schema"));
6805 }
6806 for count in &entry.nonzero {
6807 match count {
6808 None => out.push(0),
6809 Some(count) => {
6810 out.push(1);
6811 put_u64(&mut out, *count);
6812 }
6813 }
6814 }
6815 }
6816 out.extend_from_slice(AGGREGATE_SUMS);
6817 for entry in entries {
6818 if entry.aggregates.len() != entry.fields.len() {
6819 return Err(invalid("aggregate sum width differs from schema"));
6820 }
6821 for summary in &entry.aggregates {
6822 match summary {
6823 None => out.push(0),
6824 Some((sum, count)) => {
6825 out.push(1);
6826 out.extend_from_slice(&sum.to_le_bytes());
6827 put_u64(&mut out, *count);
6828 }
6829 }
6830 }
6831 }
6832 Ok(out)
6833}
6834
6835fn put_long_text(out: &mut Vec<u8>, text: &str, what: &str) -> Result<()> {
6837 let bytes = text.as_bytes();
6838 put_u32(out, u32::try_from(bytes.len()).map_err(|_| invalid(&format!("{what} too long")))?);
6839 out.extend_from_slice(bytes);
6840 Ok(())
6841}
6842
6843fn decode_catalog(bytes: &[u8], size: u64) -> Result<(Vec<Entry>, Vec<ViewEntry>)> {
6846 let mut cur = Cursor::new(bytes);
6847 if cur.take(8)? != CATALOG {
6848 return Err(invalid("catalog magic differs"));
6849 }
6850 let count = cur.u32()? as usize;
6851 let mut entries: Vec<Entry> = Vec::with_capacity(count.min(1024));
6852 for _ in 0..count {
6853 let name = cur.text()?;
6854 let rows = usize::try_from(cur.u64()?).map_err(|_| invalid("row count does not fit"))?;
6855 let width = cur.u16()? as usize;
6856 let mut fields = Vec::with_capacity(width);
6857 for _ in 0..width {
6858 let name = cur.text()?;
6859 let ty = read_type(&mut cur)?;
6860 let not_null = match cur.u8()? {
6861 0 => false,
6862 1 => true,
6863 _ => return Err(invalid("nullability flag differs")),
6864 };
6865 fields.push(Field { name, ty, not_null });
6866 }
6867 let directory = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
6868 let end = directory
6869 .offset
6870 .checked_add(u64::from(directory.length))
6871 .ok_or_else(|| invalid("table directory offset overflow"))?;
6872 if directory.offset < HEADER
6873 || end > size
6874 || directory.length as usize > MAX_DIRECTORY
6875 || directory.length == 0
6876 {
6877 return Err(invalid("table directory range is outside the file"));
6878 }
6879 if entries.iter().any(|held| held.name == name) {
6880 return Err(invalid("two tables in the catalog have the same name"));
6881 }
6882 let nonzero = vec![None; fields.len()];
6883 let aggregates = vec![None; fields.len()];
6884 entries.push(Entry { name, fields, rows, directory, nonzero, aggregates });
6885 }
6886 let count = if cur.done() { 0 } else { cur.u32()? as usize };
6891 let mut views: Vec<ViewEntry> = Vec::with_capacity(count.min(1024));
6892 for _ in 0..count {
6893 let name = cur.text()?;
6894 let sql = cur.long_text()?;
6895 let statement = cur.long_text()?;
6896 let width = cur.u16()? as usize;
6897 let mut aliases = Vec::with_capacity(width);
6898 for _ in 0..width {
6899 aliases.push(cur.text()?);
6900 }
6901 let width = cur.u16()? as usize;
6902 let mut columns = Vec::with_capacity(width);
6903 for _ in 0..width {
6904 let name = cur.text()?;
6905 let ty = read_type(&mut cur)?;
6906 let not_null = match cur.u8()? {
6907 0 => false,
6908 1 => true,
6909 _ => return Err(invalid("nullability flag differs")),
6910 };
6911 columns.push(Field { name, ty, not_null });
6912 }
6913 if views.iter().any(|held| held.name == name) {
6917 return Err(invalid("two views in the catalog have the same name"));
6918 }
6919 if entries.iter().any(|held| held.name == name) {
6920 return Err(invalid("a table and a view in the catalog have the same name"));
6921 }
6922 views.push(ViewEntry { name, sql, statement, aliases, columns });
6923 }
6924 if !cur.done() {
6925 if cur.take(8)? != NONZERO_COUNTS {
6926 return Err(invalid("catalog extension magic differs"));
6927 }
6928 for entry in &mut entries {
6929 for (field, count) in entry.fields.iter().zip(&mut entry.nonzero) {
6930 *count = match cur.u8()? {
6931 0 => None,
6932 1 if matches!(
6933 field.ty,
6934 LogicalType::TinyInt
6935 | LogicalType::SmallInt
6936 | LogicalType::Integer
6937 | LogicalType::BigInt
6938 | LogicalType::UTinyInt
6939 | LogicalType::USmallInt
6940 | LogicalType::UInteger
6941 | LogicalType::UBigInt
6942 ) =>
6943 {
6944 let value = cur.u64()?;
6945 if value > entry.rows as u64 {
6946 return Err(invalid("nonzero count exceeds rows"));
6947 }
6948 Some(value)
6949 }
6950 _ => return Err(invalid("nonzero count tag or column type differs")),
6951 };
6952 }
6953 }
6954 }
6955 if !cur.done() {
6956 if cur.take(8)? != AGGREGATE_SUMS {
6957 return Err(invalid("aggregate catalog extension magic differs"));
6958 }
6959 for entry in &mut entries {
6960 for (field, summary) in entry.fields.iter().zip(&mut entry.aggregates) {
6961 *summary = match cur.u8()? {
6962 0 => None,
6963 1 if signed_integer(&field.ty) => {
6964 let sum = i128::from_le_bytes(
6965 cur.take(16)?
6966 .try_into()
6967 .map_err(|_| invalid("aggregate sum is truncated"))?,
6968 );
6969 let count = cur.u64()?;
6970 if count > entry.rows as u64 {
6971 return Err(invalid("aggregate count exceeds table rows"));
6972 }
6973 Some((sum, count))
6974 }
6975 _ => return Err(invalid("aggregate sum tag or column type differs")),
6976 };
6977 }
6978 }
6979 }
6980 if !cur.done() {
6981 return Err(invalid("catalog has trailing bytes"));
6982 }
6983 Ok((entries, views))
6984}
6985
6986struct Cursor<'a> {
6994 bytes: &'a [u8],
6995 at: usize,
6996 window: Option<Window<'a>>,
6997}
6998
6999struct Window<'a> {
7001 file: &'a File,
7002 offset: u64,
7003 length: usize,
7004 start: usize,
7006 held: Vec<u8>,
7007 size: usize,
7009}
7010
7011const DIRECTORY_WINDOW: usize = 64 << 10;
7013
7014impl<'a> Cursor<'a> {
7015 fn new(bytes: &'a [u8]) -> Self {
7016 Self { bytes, at: 0, window: None }
7017 }
7018
7019 fn over(file: &'a File, offset: u64, length: usize) -> Self {
7021 let window =
7022 Window { file, offset, length, start: 0, held: Vec::new(), size: DIRECTORY_WINDOW };
7023 Self { bytes: &[], at: 0, window: Some(window) }
7024 }
7025
7026 fn len(&self) -> usize {
7028 self.window.as_ref().map_or(self.bytes.len(), |window| window.length)
7029 }
7030
7031 fn ensure(&mut self, len: usize) -> Result<()> {
7033 let end = self.at.checked_add(len).ok_or_else(|| invalid("directory offset overflow"))?;
7034 if end > self.len() {
7035 return Err(invalid("directory is truncated"));
7036 }
7037 let Some(window) = &mut self.window else { return Ok(()) };
7038 if self.at < window.start || end > window.start + window.held.len() {
7039 let want = len.max(window.size).min(window.length - self.at);
7040 window.start = self.at;
7041 window.held.resize(want, 0);
7042 read_at(window.file, window.offset + self.at as u64, &mut window.held)?;
7043 }
7044 Ok(())
7045 }
7046
7047 fn held(&self, at: usize, len: usize) -> &[u8] {
7049 match &self.window {
7050 Some(window) => &window.held[at - window.start..at - window.start + len],
7051 None => &self.bytes[at..at + len],
7052 }
7053 }
7054
7055 #[inline]
7057 fn peek(&mut self, len: usize) -> Result<&[u8]> {
7058 if self.window.is_none() {
7059 let bytes = self.bytes;
7060 return Ok(&bytes[self.at..self.end(len)?]);
7061 }
7062 self.ensure(len)?;
7063 Ok(self.held(self.at, len))
7064 }
7065
7066 #[inline]
7072 fn take(&mut self, len: usize) -> Result<&[u8]> {
7073 if self.window.is_none() {
7074 let bytes = self.bytes;
7075 let (at, end) = (self.at, self.end(len)?);
7076 self.at = end;
7077 return Ok(&bytes[at..end]);
7078 }
7079 self.take_windowed(len)
7080 }
7081
7082 fn skip(&mut self, len: usize) -> Result<()> {
7084 let end = self.at.checked_add(len).ok_or_else(|| invalid("directory offset overflow"))?;
7085 if end > self.len() {
7086 return Err(invalid("directory is truncated"));
7087 }
7088 self.at = end;
7089 Ok(())
7090 }
7091
7092 fn skip_bound(&mut self) -> Result<()> {
7093 match self.u8()? {
7094 0 => Ok(()),
7095 1 => self.skip(16),
7096 2 => self.skip(8),
7097 3 => {
7098 let length = self.u32()? as usize;
7099 self.skip(length)
7100 }
7101 4 => self.skip(17),
7102 _ => Err(invalid("a stored bound has an unknown tag")),
7103 }
7104 }
7105
7106 #[inline]
7108 fn end(&self, len: usize) -> Result<usize> {
7109 let end = self.at.checked_add(len).ok_or_else(|| invalid("directory offset overflow"))?;
7110 if end > self.bytes.len() {
7111 return Err(invalid("directory is truncated"));
7112 }
7113 Ok(end)
7114 }
7115
7116 #[inline(never)]
7118 fn take_windowed(&mut self, len: usize) -> Result<&[u8]> {
7119 self.ensure(len)?;
7120 self.at += len;
7121 Ok(self.held(self.at - len, len))
7122 }
7123 #[inline]
7124 fn u8(&mut self) -> Result<u8> {
7125 Ok(self.take(1)?[0])
7126 }
7127 #[inline]
7128 fn u16(&mut self) -> Result<u16> {
7129 Ok(u16::from_le_bytes(self.take(2)?.try_into().expect("two bytes")))
7130 }
7131 #[inline]
7132 fn u32(&mut self) -> Result<u32> {
7133 Ok(u32::from_le_bytes(self.take(4)?.try_into().expect("four bytes")))
7134 }
7135 #[inline]
7136 fn u64(&mut self) -> Result<u64> {
7137 Ok(u64::from_le_bytes(self.take(8)?.try_into().expect("eight bytes")))
7138 }
7139 fn var_u64(&mut self) -> Result<u64> {
7140 let mut value = 0_u64;
7141 for shift in (0..=63).step_by(7) {
7142 let byte = self.u8()?;
7143 let part = u64::from(byte & 0x7f);
7144 if shift == 63 && part > 1 {
7145 return Err(invalid("frequency ordinal varint overflows"));
7146 }
7147 value |= part << shift;
7148 if byte & 0x80 == 0 {
7149 return Ok(value);
7150 }
7151 }
7152 Err(invalid("frequency ordinal varint is too long"))
7153 }
7154 fn bound(&mut self) -> Result<Option<Bound>> {
7163 let rest = self.len().saturating_sub(self.at);
7164 let mut want = 32;
7165 loop {
7166 let offered = self.peek(want.min(rest))?;
7167 let mut used = 0;
7168 match bounds::get(offered, &mut used) {
7169 Ok(bound) => {
7170 self.at += used;
7171 return Ok(bound);
7172 }
7173 Err(_) if want < rest => want *= 2,
7174 Err(error) => return Err(error),
7175 }
7176 }
7177 }
7178 fn text(&mut self) -> Result<String> {
7179 let len = self.u16()? as usize;
7180 String::from_utf8(self.take(len)?.to_vec()).map_err(|_| invalid("name is not UTF-8"))
7181 }
7182 fn done(&self) -> bool {
7185 self.at >= self.len()
7186 }
7187 fn long_text(&mut self) -> Result<String> {
7194 let len = self.u32()? as usize;
7195 String::from_utf8(self.take(len)?.to_vec()).map_err(|_| invalid("text is not UTF-8"))
7196 }
7197}
7198
7199fn decode_summary(
7201 cur: &mut Cursor<'_>,
7202 field: &Field,
7203 rows: usize,
7204 values: bool,
7205) -> Result<Option<FrequencySummary>> {
7206 Ok(match cur.u8()? {
7207 0 => None,
7208 1 => {
7209 let omitted_max = cur.u64()?;
7210 let count = cur.u32()? as usize;
7211 if count > FREQUENCY_ENTRIES {
7212 return Err(invalid("frequency entry count exceeds its bound"));
7213 }
7214 let mut entries = Vec::with_capacity(count);
7215 for _ in 0..count {
7217 let value = match cur.u8()? {
7218 0 => FrequencyValue::Null,
7219 1 => FrequencyValue::Integer(i128::from_le_bytes(
7220 cur.take(16)?.try_into().expect("sixteen bytes"),
7221 )),
7222 2 => FrequencyValue::Code(cur.u32()?),
7223 _ => return Err(invalid("frequency value tag differs")),
7224 };
7225 let valid = matches!(
7226 (&field.ty, value),
7227 (_, FrequencyValue::Null)
7228 | (LogicalType::Varchar, FrequencyValue::Code(_))
7229 | (
7230 LogicalType::TinyInt
7231 | LogicalType::SmallInt
7232 | LogicalType::Integer
7233 | LogicalType::BigInt
7234 | LogicalType::UTinyInt
7235 | LogicalType::USmallInt
7236 | LogicalType::UInteger
7237 | LogicalType::UBigInt
7238 | LogicalType::Date
7239 | LogicalType::Timestamp,
7240 FrequencyValue::Integer(_),
7241 )
7242 );
7243 if !valid {
7244 return Err(invalid("frequency value does not match its column"));
7245 }
7246 let count = cur.u64()?;
7247 if count == 0 || count > rows as u64 {
7248 return Err(invalid("frequency count is outside the table"));
7249 }
7250 entries.push(FrequencyEntry { value, count });
7251 }
7252 if entries.windows(2).any(|pair| pair[0].count < pair[1].count) {
7253 return Err(invalid("frequency entries are not descending"));
7254 }
7255 let ordinals = {
7256 let ordinal_count = cur.u32()? as usize;
7257 if ordinal_count > FREQUENCY_ORDINALS || ordinal_count > rows {
7258 return Err(invalid("frequency ordinal count exceeds its bound"));
7259 }
7260 let mut ordinals = Vec::with_capacity(ordinal_count);
7261 let mut previous = 0_u64;
7262 for at in 0..ordinal_count {
7263 let delta = cur.var_u64()?;
7264 if at != 0 && delta == 0 {
7265 return Err(invalid("frequency ordinals are not increasing"));
7266 }
7267 let ordinal = if at == 0 {
7268 delta
7269 } else {
7270 previous
7271 .checked_add(delta)
7272 .ok_or_else(|| invalid("frequency ordinal overflows"))?
7273 };
7274 if ordinal >= rows as u64 {
7275 return Err(invalid("frequency ordinal is outside the table"));
7276 }
7277 ordinals.push(ordinal);
7278 previous = ordinal;
7279 }
7280 ordinals
7281 };
7282 let ordinal_entries = if values {
7283 let mut ordinal_entries = Vec::with_capacity(ordinals.len());
7284 for _ in 0..ordinals.len() {
7285 let entry = cur.u16()?;
7286 if entry as usize >= entries.len() {
7287 return Err(invalid("frequency ordinal value is outside its entries"));
7288 }
7289 ordinal_entries.push(entry);
7290 }
7291 ordinal_entries
7292 } else {
7293 Vec::new()
7294 };
7295 Some(FrequencySummary { entries, omitted_max, ordinals, ordinal_entries })
7296 }
7297 _ => return Err(invalid("frequency summary tag differs")),
7298 })
7299}
7300
7301fn skip_summary(cur: &mut Cursor<'_>, values: bool, rows: usize) -> Result<()> {
7304 match cur.u8()? {
7305 0 => Ok(()),
7306 1 => {
7307 cur.skip(8)?;
7308 let entries = cur.u32()? as usize;
7309 if entries > FREQUENCY_ENTRIES {
7310 return Err(invalid("frequency entry count exceeds its bound"));
7311 }
7312 for _ in 0..entries {
7313 match cur.u8()? {
7314 0 => {}
7315 1 => cur.skip(16)?,
7316 2 => cur.skip(4)?,
7317 _ => return Err(invalid("frequency value tag differs")),
7318 }
7319 cur.skip(8)?;
7320 }
7321 let ordinals = cur.u32()? as usize;
7322 if ordinals > FREQUENCY_ORDINALS || ordinals > rows {
7323 return Err(invalid("frequency ordinal count exceeds its bound"));
7324 }
7325 for _ in 0..ordinals {
7326 cur.var_u64()?;
7327 }
7328 if values {
7329 cur.skip(ordinals * 2)?;
7330 }
7331 Ok(())
7332 }
7333 _ => Err(invalid("frequency summary tag differs")),
7334 }
7335}
7336
7337fn quick_nonzero(
7341 mut cur: Cursor<'_>,
7342 name: &str,
7343 fields: &[Field],
7344 rows: usize,
7345 wanted: usize,
7346) -> Result<Option<u64>> {
7347 if cur.take(8)? != DIRECTORY || cur.text()? != name {
7348 return Err(invalid("table directory differs from the catalog"));
7349 }
7350 let width = cur.u16()? as usize;
7351 if width != fields.len() {
7352 return Err(invalid("table directory width differs from the catalog"));
7353 }
7354 for field in fields {
7355 let stored =
7356 Field { name: cur.text()?, ty: read_type(&mut cur)?, not_null: cur.u8()? != 0 };
7357 if &stored != field {
7358 return Err(invalid("table directory schema differs from the catalog"));
7359 }
7360 }
7361 let mut dictionaries = Vec::with_capacity(width);
7362 for _ in 0..width {
7363 let held = match cur.u8()? {
7364 0 => false,
7365 1 => {
7366 cur.skip(20)?;
7367 true
7368 }
7369 _ => return Err(invalid("dictionary page tag differs")),
7370 };
7371 dictionaries.push(held);
7372 }
7373 for _ in 0..width {
7374 match cur.u8()? {
7375 0 => {}
7376 1 => cur.skip(8)?,
7377 _ => return Err(invalid("distinct count tag differs")),
7378 }
7379 }
7380 if cur.u64()? != rows as u64 {
7381 return Err(invalid("table row count differs from the catalog"));
7382 }
7383 let stripes = cur.u32()? as usize;
7384 let mut total = 0_usize;
7385 let mut nulls = 0_u64;
7386 for _ in 0..stripes {
7387 let parts = cur.u32()? as usize;
7388 if parts == 0 || parts > STRIPE_PARTS {
7389 return Err(invalid("stripe part count is outside its bound"));
7390 }
7391 let mut stripe_rows = 0_usize;
7392 for _ in 0..parts {
7393 stripe_rows = stripe_rows
7394 .checked_add(cur.u32()? as usize)
7395 .ok_or_else(|| invalid("stripe row count overflow"))?;
7396 }
7397 total =
7398 total.checked_add(stripe_rows).ok_or_else(|| invalid("stripe row count overflow"))?;
7399 cur.skip(12 + width * 12)?;
7400 for (field, held) in fields.iter().zip(&dictionaries) {
7401 if field.ty == LogicalType::Varchar && *held {
7402 cur.skip(20)?;
7403 }
7404 }
7405 for _ in 0..width * 2 {
7406 match cur.u8()? {
7407 0 => {}
7408 1 => cur.skip(20)?,
7409 _ => return Err(invalid("stripe page tag differs")),
7410 }
7411 }
7412 for column in 0..width {
7413 cur.skip_bound()?;
7414 cur.skip_bound()?;
7415 let count = cur.u32()? as u64;
7416 if count > stripe_rows as u64 {
7417 return Err(invalid("null count exceeds stripe rows"));
7418 }
7419 if column == wanted {
7420 nulls = nulls.checked_add(count).ok_or_else(|| invalid("null count overflow"))?;
7421 }
7422 cur.skip(1)?;
7423 match cur.u8()? {
7424 0 => {}
7425 1 => cur.skip(16)?,
7426 _ => return Err(invalid("a stripe sum has an unknown tag")),
7427 }
7428 }
7429 }
7430 if total != rows {
7431 return Err(invalid("table row count differs from stripes"));
7432 }
7433 if cur.done() {
7434 return Ok(None);
7435 }
7436 let magic = cur.take(8)?;
7437 let values = magic == FREQUENCIES;
7438 if !values && magic != FREQUENCIES_V2 {
7439 return Err(invalid("directory extension magic differs"));
7440 }
7441 if cur.u16()? as usize != width {
7442 return Err(invalid("frequency column count differs"));
7443 }
7444 for _ in 0..wanted {
7445 skip_summary(&mut cur, values, rows)?;
7446 }
7447 let Some(summary) = decode_summary(&mut cur, &fields[wanted], rows, values)? else {
7448 return Ok(None);
7449 };
7450 let zero = summary
7451 .entries
7452 .iter()
7453 .find(|entry| entry.value == FrequencyValue::Integer(0))
7454 .map(|entry| entry.count)
7455 .or_else(|| (summary.omitted_max == 0).then_some(0));
7456 Ok(zero.and_then(|zero| (rows as u64).checked_sub(nulls)?.checked_sub(zero)))
7457}
7458
7459fn decode_directory(bytes: &[u8], size: u64) -> Result<Table> {
7460 read_directory(Cursor::new(bytes), size, None)
7461}
7462
7463fn read_directory(mut cur: Cursor<'_>, size: u64, stored_at: Option<u64>) -> Result<Table> {
7468 if cur.take(8)? != DIRECTORY {
7469 return Err(invalid("directory magic differs"));
7470 }
7471 let name = cur.text()?;
7472 let width = cur.u16()? as usize;
7473 let mut fields = Vec::with_capacity(width);
7474 for _ in 0..width {
7475 let name = cur.text()?;
7476 let ty = read_type(&mut cur)?;
7477 let not_null = match cur.u8()? {
7478 0 => false,
7479 1 => true,
7480 _ => return Err(invalid("nullability flag differs")),
7481 };
7482 fields.push(Field { name, ty, not_null });
7483 }
7484 let mut dictionaries = Vec::with_capacity(width);
7485 for _ in 0..width {
7486 dictionaries.push(match cur.u8()? {
7487 0 => None,
7488 1 => {
7489 let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
7490 let end = page
7491 .offset
7492 .checked_add(u64::from(page.length))
7493 .ok_or_else(|| invalid("dictionary page offset overflow"))?;
7494 if page.offset < HEADER || end > size {
7499 return Err(invalid("dictionary page range is outside the file"));
7500 }
7501 Some(page)
7502 }
7503 _ => return Err(invalid("dictionary page tag differs")),
7504 });
7505 }
7506 let mut distincts = Vec::with_capacity(width);
7507 for _ in 0..width {
7508 distincts.push(match cur.u8()? {
7509 0 => None,
7510 1 => Some(cur.u64()?),
7511 _ => return Err(invalid("distinct count tag differs")),
7512 });
7513 }
7514 let rows = usize::try_from(cur.u64()?).map_err(|_| invalid("row count does not fit"))?;
7515 let count = cur.u32()? as usize;
7516 let mut stripes = Vec::with_capacity(count);
7517 let mut total = 0_usize;
7518 for _ in 0..count {
7519 let count = cur.u32()? as usize;
7520 if count == 0 || count > STRIPE_PARTS {
7521 return Err(invalid("stripe part count is outside its bound"));
7522 }
7523 let mut parts = Vec::with_capacity(count);
7524 let mut stripe_rows = 0_usize;
7525 for _ in 0..count {
7526 let rows = cur.u32()?;
7527 if rows == 0 {
7528 return Err(invalid("empty part"));
7529 }
7530 parts.push(rows);
7531 stripe_rows = stripe_rows
7532 .checked_add(rows as usize)
7533 .ok_or_else(|| invalid("stripe row count overflow"))?;
7534 }
7535 total =
7536 total.checked_add(stripe_rows).ok_or_else(|| invalid("stripe row count overflow"))?;
7537 let index = Span { offset: cur.u64()?, length: cur.u32()? };
7538 let section = index_section(count)?;
7539 let wanted = section
7540 .checked_mul(width)
7541 .and_then(|bytes| u32::try_from(bytes).ok())
7542 .ok_or_else(|| invalid("index page length overflow"))?;
7543 let end = index
7544 .offset
7545 .checked_add(u64::from(index.length))
7546 .ok_or_else(|| invalid("index page offset overflow"))?;
7547 if index.offset < HEADER || end > size || index.length != wanted {
7548 return Err(invalid("index page range is outside the file"));
7549 }
7550 let mut pages = Vec::with_capacity(width);
7551 for _ in 0..width {
7552 let offset = cur.u64()?;
7553 let length = cur.u32()?;
7554 let end = offset
7555 .checked_add(u64::from(length))
7556 .ok_or_else(|| invalid("page offset overflow"))?;
7557 if offset < HEADER || end > size || length as usize > MAX_PAGE {
7558 return Err(invalid("page range is outside the file"));
7559 }
7560 pages.push(Span { offset, length });
7561 }
7562 let mut memberships = vec![None; width];
7563 for (column, field) in fields.iter().enumerate() {
7564 if field.ty != LogicalType::Varchar || dictionaries[column].is_none() {
7565 continue;
7566 }
7567 let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
7568 let end = page
7569 .offset
7570 .checked_add(u64::from(page.length))
7571 .ok_or_else(|| invalid("membership page offset overflow"))?;
7572 if page.offset < HEADER || end > size || page.length as usize > MAX_PAGE {
7573 return Err(invalid("membership page range is outside the file"));
7574 }
7575 memberships[column] = Some(page);
7576 }
7577 let mut sieves = vec![None; width];
7578 for sieve in sieves.iter_mut().take(width) {
7579 match cur.u8()? {
7580 0 => continue,
7581 1 => {}
7582 _ => return Err(invalid("a sieve page has an unknown tag")),
7583 }
7584 let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
7585 let end = page
7586 .offset
7587 .checked_add(u64::from(page.length))
7588 .ok_or_else(|| invalid("sieve page offset overflow"))?;
7589 if page.offset < HEADER || end > size || page.length as usize > MAX_PAGE {
7590 return Err(invalid("sieve page range is outside the file"));
7591 }
7592 *sieve = Some(page);
7593 }
7594 let mut part_ranges = vec![None; width];
7595 for held in part_ranges.iter_mut().take(width) {
7596 match cur.u8()? {
7597 0 => continue,
7598 1 => {}
7599 _ => return Err(invalid("a part range page has an unknown tag")),
7600 }
7601 let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
7602 let end = page
7603 .offset
7604 .checked_add(u64::from(page.length))
7605 .ok_or_else(|| invalid("part range page offset overflow"))?;
7606 if page.offset < HEADER || end > size || page.length as usize > MAX_PAGE {
7607 return Err(invalid("part range page range is outside the file"));
7608 }
7609 *held = Some(page);
7610 }
7611 let mut ranges = Vec::with_capacity(width);
7612 for column in 0..width {
7613 let low = cur.bound()?;
7614 let high = cur.bound()?;
7615 let nulls = cur.u32()? as usize;
7616 if nulls > stripe_rows {
7617 return Err(invalid("null count exceeds stripe rows"));
7618 }
7619 let exact = cur.u8()? != 0;
7620 let sum = match cur.u8()? {
7621 0 => None,
7622 1 => Some(i128::from_le_bytes(
7623 cur.take(16)?.try_into().map_err(|_| invalid("a stripe sum is truncated"))?,
7624 )),
7625 _ => return Err(invalid("a stripe sum has an unknown tag")),
7626 };
7627 let ty = &fields.get(column).ok_or_else(|| invalid("a stripe range has no column"))?.ty;
7633 let low = low.map(|bound| scaled_as(bound, ty));
7634 let high = high.map(|bound| scaled_as(bound, ty));
7635 ranges.push(Range { low, high, nulls, exact, sum });
7636 }
7637 stripes.push(Stripe {
7638 rows: stripe_rows,
7639 parts,
7640 index,
7641 pages,
7642 memberships: Pages::from_slots(memberships)?,
7643 sieves: Pages::from_slots(sieves)?,
7644 part_ranges: Pages::from_slots(part_ranges)?,
7645 zone: Zone::from_ranges(ranges),
7646 });
7647 }
7648 if total != rows {
7649 return Err(invalid("table row count differs from stripes"));
7650 }
7651 let mut entry_counts = vec![0; width];
7654 let frequencies = if cur.done() {
7655 vec![None; width]
7656 } else {
7657 let frequency_magic = cur.take(8)?;
7658 let frequency_values = frequency_magic == FREQUENCIES;
7659 if !frequency_values && frequency_magic != FREQUENCIES_V2 {
7660 return Err(invalid("directory extension magic differs"));
7661 }
7662 if cur.u16()? as usize != width {
7663 return Err(invalid("frequency column count differs"));
7664 }
7665 let mut frequencies = Vec::with_capacity(width);
7666 for (field, entry_count) in fields.iter().zip(&mut entry_counts) {
7667 let start = cur.at;
7668 let summary = decode_summary(&mut cur, field, rows, frequency_values)?;
7669 *entry_count = summary.as_ref().map_or(0, |summary| summary.entries.len());
7670 frequencies.push(match (summary, stored_at) {
7671 (None, _) => None,
7672 (Some(summary), None) => Some(Frequencies::Held(summary)),
7673 (Some(_), Some(offset)) => Some(Frequencies::Stored {
7674 span: Span {
7675 offset: offset + start as u64,
7676 length: u32::try_from(cur.at - start)
7677 .map_err(|_| invalid("a frequency synopsis is too long"))?,
7678 },
7679 values: frequency_values,
7680 }),
7681 });
7682 }
7683 frequencies
7684 };
7685 let mut clustering = None;
7695 let mut sections = Vec::new();
7696 let mut pair_frequencies = Vec::new();
7697 let mut seen_pair_frequencies = false;
7698 let mut frequency_texts = vec![Vec::new(); width];
7699 let mut seen_frequency_texts = false;
7700 let mut host_groups = None;
7701 let mut seen_sections = false;
7702 let mut dictionary_payloads = Vec::new();
7703 let mut seen_payloads = false;
7704 let mut generation = 0;
7707 while !cur.done() {
7708 let mut tag = [0u8; 8];
7709 tag.copy_from_slice(cur.take(8)?);
7710 if &tag == PAIR_FREQUENCIES {
7711 if seen_pair_frequencies {
7712 return Err(invalid("directory names two pair frequency blocks"));
7713 }
7714 seen_pair_frequencies = true;
7715 let count = cur.u16()? as usize;
7716 if count > MAX_PAIR_FREQUENCIES {
7717 return Err(invalid("pair frequency count exceeds its bound"));
7718 }
7719 pair_frequencies = Vec::with_capacity(count);
7720 for _ in 0..count {
7721 let first = cur.u16()?;
7722 let second = cur.u16()?;
7723 let first_at = first as usize;
7724 let second_at = second as usize;
7725 if frequencies.get(first_at).and_then(Option::as_ref).is_none() {
7726 return Err(invalid("pair frequency first column has no synopsis"));
7727 }
7728 let first_entries = entry_counts[first_at];
7729 if !matches!(fields.get(second_at), Some(field) if field.ty == LogicalType::Varchar)
7730 || dictionaries.get(second_at).copied().flatten().is_none()
7731 {
7732 return Err(invalid("pair frequency second column has no stable dictionary"));
7733 }
7734 if pair_frequencies
7735 .iter()
7736 .any(|held: &PairFrequencySummary| held.first == first && held.second == second)
7737 {
7738 return Err(invalid("directory repeats a pair frequency summary"));
7739 }
7740 let omitted_max = cur.u64()?;
7741 if omitted_max > rows as u64 {
7742 return Err(invalid("pair frequency omitted count exceeds the table"));
7743 }
7744 let entries_count = cur.u16()? as usize;
7745 if entries_count > FREQUENCY_ENTRIES {
7746 return Err(invalid("pair frequency entry count exceeds its bound"));
7747 }
7748 let mut entries = Vec::with_capacity(entries_count);
7749 for _ in 0..entries_count {
7750 let first_entry = cur.u16()?;
7751 if first_entry as usize >= first_entries {
7752 return Err(invalid("pair frequency anchor is outside its synopsis"));
7753 }
7754 let second = match cur.u8()? {
7755 0 => None,
7756 1 => Some(cur.u32()?),
7757 _ => return Err(invalid("pair frequency string tag differs")),
7758 };
7759 let count = cur.u64()?;
7760 if count == 0 || count > rows as u64 {
7761 return Err(invalid("pair frequency count is outside the table"));
7762 }
7763 entries.push(PairFrequencyEntry { first_entry, second, count });
7764 }
7765 if entries.windows(2).any(|pair| pair[0].count < pair[1].count) {
7766 return Err(invalid("pair frequency entries are not descending"));
7767 }
7768 pair_frequencies.push(PairFrequencySummary { first, second, entries, omitted_max });
7769 }
7770 } else if &tag == FREQUENCY_TEXTS {
7771 if seen_frequency_texts {
7772 return Err(invalid("directory names two frequency text blocks"));
7773 }
7774 seen_frequency_texts = true;
7775 let columns = cur.u16()? as usize;
7776 if columns > width {
7777 return Err(invalid("frequency text column count exceeds the schema"));
7778 }
7779 for _ in 0..columns {
7780 let column = cur.u16()? as usize;
7781 if !frequency_texts.get(column).is_some_and(Vec::is_empty) {
7782 return Err(invalid("frequency text column is repeated or out of range"));
7783 }
7784 if !matches!(fields.get(column), Some(field) if field.ty == LogicalType::Varchar)
7785 || dictionaries.get(column).copied().flatten().is_none()
7786 || frequencies.get(column).and_then(Option::as_ref).is_none()
7787 {
7788 return Err(invalid("frequency texts belong to a non-string synopsis"));
7789 }
7790 let count = cur.u16()? as usize;
7791 if count == 0 || count != entry_counts[column] {
7792 return Err(invalid("frequency text count differs from its synopsis"));
7793 }
7794 let mut texts = Vec::with_capacity(count);
7795 for _ in 0..count {
7796 texts.push(match cur.u8()? {
7797 0 => None,
7798 1 => {
7799 let length = cur.u32()? as usize;
7800 let bytes = cur.take(length)?.to_vec();
7801 std::str::from_utf8(&bytes)
7802 .map_err(|_| invalid("frequency text is not UTF-8"))?;
7803 Some(bytes)
7804 }
7805 _ => return Err(invalid("frequency text tag differs")),
7806 });
7807 }
7808 frequency_texts[column] = texts;
7809 }
7810 } else if &tag == HOST_GROUPS {
7811 if host_groups.is_some() {
7812 return Err(invalid("directory names two host group blocks"));
7813 }
7814 let column = cur.u16()? as usize;
7815 if !matches!(fields.get(column), Some(field) if field.ty == LogicalType::Varchar)
7816 || dictionaries.get(column).copied().flatten().is_none()
7817 {
7818 return Err(invalid("host groups belong to a non-string dictionary"));
7819 }
7820 let omitted_max = cur.u64()?;
7821 if omitted_max > rows as u64 {
7822 return Err(invalid("host group bound exceeds the table"));
7823 }
7824 let count = cur.u16()? as usize;
7825 if count > host::CAPACITY {
7826 return Err(invalid("host group count exceeds its bound"));
7827 }
7828 let mut entries = Vec::with_capacity(count);
7829 let mut bytes = 0_usize;
7830 for _ in 0..count {
7831 let host_len = cur.u32()? as usize;
7832 bytes =
7833 bytes.checked_add(host_len).ok_or_else(|| invalid("host bytes overflow"))?;
7834 if bytes > host::BYTE_BUDGET {
7835 return Err(invalid("host groups exceed their byte budget"));
7836 }
7837 let host = std::str::from_utf8(cur.take(host_len)?)
7838 .map_err(|_| invalid("host is not UTF-8"))?
7839 .to_owned();
7840 let count = cur.u64()?;
7841 if count == 0 || count > rows as u64 {
7842 return Err(invalid("host group count exceeds the table"));
7843 }
7844 let bytes_sum = i128::from_le_bytes(
7845 cur.take(16)?
7846 .try_into()
7847 .map_err(|_| invalid("host length sum is truncated"))?,
7848 );
7849 if bytes_sum < 0 {
7850 return Err(invalid("host length sum is negative"));
7851 }
7852 let minimum_len = cur.u32()? as usize;
7853 bytes =
7854 bytes.checked_add(minimum_len).ok_or_else(|| invalid("host bytes overflow"))?;
7855 if bytes > host::BYTE_BUDGET {
7856 return Err(invalid("host groups exceed their byte budget"));
7857 }
7858 let minimum = std::str::from_utf8(cur.take(minimum_len)?)
7859 .map_err(|_| invalid("host minimum is not UTF-8"))?
7860 .to_owned();
7861 entries.push(host::HostEntry { host, count, bytes_sum, minimum });
7862 }
7863 if entries.windows(2).any(|pair| pair[0].count < pair[1].count)
7864 || entries.iter().any(|entry| entry.host.is_empty() || entry.minimum.is_empty())
7865 {
7866 return Err(invalid("host groups are not in certified order"));
7867 }
7868 host_groups = Some(host::HostSummary { column, omitted_max, entries });
7869 } else if &tag == CLUSTERING {
7870 if clustering.is_some() {
7871 return Err(invalid("directory names two clustering declarations"));
7872 }
7873 let bucket = Width::from_tag(cur.u8()?)
7874 .ok_or_else(|| invalid("clustering width tag differs"))?;
7875 let count = cur.u16()? as usize;
7876 let mut columns = Vec::with_capacity(count.min(fields.len()));
7877 for _ in 0..count {
7878 columns.push(u32::from(cur.u16()?));
7879 }
7880 clustering = Some(Clustering::new(columns, bucket, &fields).map_err(|_| {
7883 invalid("stored clustering declaration does not match the table it is on")
7884 })?);
7885 } else if &tag == SECTIONS {
7886 if seen_sections {
7887 return Err(invalid("directory names two section tables"));
7888 }
7889 seen_sections = true;
7890 generation = cur.u64()?;
7891 let count = cur.u16()? as usize;
7892 if count > MAX_SECTIONS {
7893 return Err(invalid("section count exceeds its bound"));
7894 }
7895 sections = Vec::with_capacity(count);
7896 for _ in 0..count {
7899 sections.push(Section::decode(cur.take(section::ENTRY_BYTES)?)?);
7900 }
7901 for held in §ions {
7902 let Some(end) = held.extent_page.checked_add(u64::from(held.extent_bytes)) else {
7903 return Err(invalid("a section's extent table overflows the file"));
7904 };
7905 if held.extent_bytes != 0 && (held.extent_page < HEADER || end > size) {
7909 return Err(invalid("a section's extent table is outside the file"));
7910 }
7911 if held.extents == 0 && held.extent_bytes != 0 {
7912 return Err(invalid("a section with no extents names an extent table"));
7913 }
7914 }
7915 } else if &tag == DICTIONARY_PAYLOADS {
7916 if seen_payloads {
7917 return Err(invalid("directory names two dictionary payload blocks"));
7918 }
7919 seen_payloads = true;
7920 let count = cur.u16()? as usize;
7921 if count != fields.len() {
7922 return Err(invalid("dictionary payload block does not match the table's columns"));
7923 }
7924 dictionary_payloads = Vec::with_capacity(count);
7925 for _ in 0..count {
7926 let bytes = cur.u64()?;
7927 if bytes > size {
7928 return Err(invalid("a dictionary payload is larger than the file"));
7929 }
7930 dictionary_payloads.push(bytes);
7931 }
7932 } else {
7933 return Err(invalid("directory extension magic differs"));
7934 }
7935 }
7936 if !cur.done() {
7937 return Err(invalid("directory has trailing bytes"));
7938 }
7939 Ok(Table {
7940 name,
7941 fields,
7942 stripes,
7943 rows,
7944 dictionaries,
7945 dictionary_payloads,
7946 distincts,
7947 frequencies,
7948 pair_frequencies,
7949 frequency_texts,
7950 host_groups,
7951 clustering,
7952 generation,
7953 sections,
7954 })
7955}
7956
7957fn put_bound(out: &mut Vec<u8>, bound: Option<&Bound>) -> Result<()> {
7959 bounds::put(out, bound)
7960}
7961
7962#[derive(Debug)]
7979struct Codes;
7980
7981impl chooser::Chooser for Codes {
7982 fn name(&self) -> &'static str {
7983 "codes"
7984 }
7985
7986 fn narrow_strings(
7987 &self,
7988 _values: &[&[u8]],
7989 offered: &[string::Kind],
7990 _depth: u8,
7991 ) -> Vec<string::Kind> {
7992 offered.to_vec()
7995 }
7996
7997 fn narrow_integers(
7998 &self,
7999 _values: &[i64],
8000 offered: &[integer::Kind],
8001 depth: u8,
8002 ) -> Vec<integer::Kind> {
8003 narrowed_to(Codes::keep(depth), offered)
8006 }
8007
8008 fn considers_integer(&self, kind: integer::Kind, depth: u8) -> bool {
8009 Codes::keep(depth).contains(&kind)
8010 }
8011}
8012
8013impl Codes {
8014 fn keep(depth: u8) -> &'static [integer::Kind] {
8015 if depth == 0 {
8016 &[integer::Kind::Constant, integer::Kind::Packed, integer::Kind::Rle]
8017 } else {
8018 &[integer::Kind::Constant, integer::Kind::Packed]
8019 }
8020 }
8021}
8022
8023fn narrowed_to(keep: &[integer::Kind], offered: &[integer::Kind]) -> Vec<integer::Kind> {
8031 let narrowed: Vec<integer::Kind> =
8032 offered.iter().copied().filter(|kind| keep.contains(kind)).collect();
8033 if narrowed.is_empty() { offered.to_vec() } else { narrowed }
8034}
8035
8036#[derive(Debug)]
8048struct Fixed;
8049
8050impl chooser::Chooser for Fixed {
8051 fn name(&self) -> &'static str {
8052 "fixed"
8053 }
8054
8055 fn narrow_strings(
8056 &self,
8057 _values: &[&[u8]],
8058 offered: &[string::Kind],
8059 _depth: u8,
8060 ) -> Vec<string::Kind> {
8061 offered.to_vec()
8062 }
8063
8064 fn narrow_integers(
8065 &self,
8066 _values: &[i64],
8067 offered: &[integer::Kind],
8068 depth: u8,
8069 ) -> Vec<integer::Kind> {
8070 narrowed_to(Fixed::keep(depth), offered)
8071 }
8072
8073 fn considers_integer(&self, kind: integer::Kind, depth: u8) -> bool {
8074 Fixed::keep(depth).contains(&kind)
8075 }
8076}
8077
8078impl Fixed {
8079 fn keep(depth: u8) -> &'static [integer::Kind] {
8080 if depth == 0 {
8081 &[
8082 integer::Kind::Constant,
8083 integer::Kind::Packed,
8084 integer::Kind::Delta,
8085 integer::Kind::Rle,
8086 integer::Kind::Sparse,
8087 integer::Kind::Strided,
8088 ]
8089 } else {
8090 &[integer::Kind::Constant, integer::Kind::Packed, integer::Kind::Delta]
8091 }
8092 }
8093}
8094
8095fn widened(data: &Data) -> Option<Vec<i64>> {
8102 match data {
8103 Data::Int8(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
8104 Data::UInt8(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
8105 Data::Int16(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
8106 Data::UInt16(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
8107 Data::Int32(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
8108 Data::UInt32(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
8109 Data::Int64(values) => Some(values.to_vec()),
8110 _ => None,
8111 }
8112}
8113
8114trait Narrow: Copy {
8121 const BIASED: (u32, u64);
8126
8127 fn narrow(value: i64) -> Self;
8129}
8130
8131#[allow(clippy::cast_sign_loss, reason = "a residue is a bit pattern and not a number")]
8148fn residue<T: Narrow>(value: i64) -> u64 {
8149 let (bits, bias) = T::BIASED;
8150 (value as u64).wrapping_add(bias) >> bits
8151}
8152
8153macro_rules! narrows {
8158 ($($ty:ty => $bias:expr),* $(,)?) => {$(
8159 impl Narrow for $ty {
8160 const BIASED: (u32, u64) = (<$ty>::BITS, $bias);
8161
8162 #[allow(
8163 clippy::cast_possible_truncation,
8164 clippy::cast_sign_loss,
8165 reason = "the caller has checked the bits this truncates away"
8166 )]
8167 fn narrow(value: i64) -> Self {
8168 value as Self
8169 }
8170 }
8171 )*};
8172}
8173
8174narrows! {
8175 i8 => 1 << 7,
8176 u8 => 0,
8177 i16 => 1 << 15,
8178 u16 => 0,
8179 i32 => 1 << 31,
8180 u32 => 0,
8181}
8182
8183fn fit<T: Narrow>(values: &[i64]) -> Result<Vec<T>> {
8196 let mut spilled = 0u64;
8197 for value in values {
8198 spilled |= residue::<T>(*value);
8199 }
8200 if spilled != 0 {
8201 return Err(invalid("page value is not of its type"));
8202 }
8203 Ok(values.iter().map(|value| T::narrow(*value)).collect())
8204}
8205
8206fn narrowed(ty: &LogicalType, values: Vec<i64>) -> Result<Data> {
8211 Ok(match ty {
8212 LogicalType::TinyInt => Data::Int8(fit::<i8>(&values)?.into()),
8213 LogicalType::UTinyInt => Data::UInt8(fit::<u8>(&values)?.into()),
8214 LogicalType::SmallInt => Data::Int16(fit::<i16>(&values)?.into()),
8215 LogicalType::USmallInt => Data::UInt16(fit::<u16>(&values)?.into()),
8216 LogicalType::Integer | LogicalType::Date => Data::Int32(fit::<i32>(&values)?.into()),
8217 LogicalType::UInteger => Data::UInt32(fit::<u32>(&values)?.into()),
8218 LogicalType::BigInt
8219 | LogicalType::Timestamp
8220 | LogicalType::Time
8221 | LogicalType::TimeTz
8222 | LogicalType::TimestampTz
8223 | LogicalType::TimestampS
8224 | LogicalType::TimestampMs
8225 | LogicalType::TimestampNs => Data::Int64(values.into()),
8226 LogicalType::Decimal { .. } => match ty.physical() {
8229 PhysicalType::Int16 => Data::Int16(fit::<i16>(&values)?.into()),
8230 PhysicalType::Int32 => Data::Int32(fit::<i32>(&values)?.into()),
8231 PhysicalType::Int64 => Data::Int64(values.into()),
8232 _ => return Err(invalid("cascade codec belongs to a decimal that is not an integer")),
8233 },
8234 _ => return Err(invalid("cascade codec belongs to a page that is not integers")),
8235 })
8236}
8237
8238fn plain_width(ty: &LogicalType) -> Option<usize> {
8241 Some(match ty {
8242 LogicalType::TinyInt | LogicalType::UTinyInt => 1,
8243 LogicalType::SmallInt | LogicalType::USmallInt => 2,
8244 LogicalType::Integer | LogicalType::UInteger | LogicalType::Date => 4,
8245 LogicalType::BigInt
8246 | LogicalType::Timestamp
8247 | LogicalType::Time
8248 | LogicalType::TimeTz
8249 | LogicalType::TimestampTz
8250 | LogicalType::TimestampS
8251 | LogicalType::TimestampMs
8252 | LogicalType::TimestampNs => 8,
8253 LogicalType::Decimal { .. } => match ty.physical() {
8254 PhysicalType::Int16 => 2,
8255 PhysicalType::Int32 => 4,
8256 PhysicalType::Int64 => 8,
8257 _ => return None,
8260 },
8261 _ => return None,
8262 })
8263}
8264
8265fn cascaded(
8271 flat: &Vector,
8272 ty: &LogicalType,
8273 packed: Option<&Packed<'_>>,
8274 settling: &mut Settling,
8275) -> Result<Option<Vec<u8>>> {
8276 let (Some(width), Some(data)) = (plain_width(ty), flat.data()) else { return Ok(None) };
8277 let Some(values) = widened(data) else { return Ok(None) };
8278 let plain = values.len().saturating_mul(width);
8279 let best = match packed {
8280 Some(packed) => plain.min(21 + size_of_val(packed.words())),
8282 None => plain,
8283 };
8284 let out = settling.encode(&values)?;
8285 Ok((out.len() < best).then_some(out))
8286}
8287
8288const SEARCH_EVERY: usize = 16;
8295
8296#[derive(Debug, Default)]
8301struct Settling {
8302 shape: Option<Shape>,
8305 since: usize,
8307}
8308
8309impl Settling {
8310 fn encode(&mut self, values: &[i64]) -> Result<Vec<u8>> {
8317 if let Some(shape) = self.shape.as_ref().filter(|_| self.since < SEARCH_EVERY) {
8318 let replay = chooser::Replay::new(&shape.kinds, &Fixed).expecting(&shape.offered);
8319 let out = integer::encode_with(values, &replay)?;
8320 if !replay.held() {
8321 self.settle(&out, values.len(), replay.first_offered())?;
8322 return Ok(out);
8323 }
8324 let grown = (out.len() as u128) * (shape.rows as u128) * 4;
8325 if grown <= (shape.len as u128) * (values.len() as u128) * 5 {
8326 self.since += 1;
8327 return Ok(out);
8328 }
8329 }
8330 let search = chooser::Replay::new(&[], &Fixed);
8332 let out = integer::encode_with(values, &search)?;
8333 self.settle(&out, values.len(), search.first_offered())?;
8334 Ok(out)
8335 }
8336
8337 fn settle(&mut self, out: &[u8], rows: usize, offered: Vec<integer::Kind>) -> Result<()> {
8338 let kinds = integer::shape(out)?;
8339 self.shape = Some(Shape { kinds, offered, len: out.len().max(1), rows: rows.max(1) });
8340 self.since = 0;
8341 Ok(())
8342 }
8343}
8344
8345#[derive(Debug)]
8347struct Shape {
8348 kinds: Vec<integer::Kind>,
8349 offered: Vec<integer::Kind>,
8350 len: usize,
8351 rows: usize,
8352}
8353
8354fn text_compressed(flat: &Vector) -> Result<Option<Vec<u8>>> {
8392 let mut values: Vec<&[u8]> = Vec::with_capacity(flat.len());
8393 let mut payload = 0_usize;
8394 for row in 0..flat.len() {
8395 let text = flat.text_at(row).unwrap_or("").as_bytes();
8396 payload = payload.saturating_add(text.len());
8397 values.push(text);
8398 }
8399 let plain = (flat.len() + 1).saturating_mul(4).saturating_add(payload);
8401 let Some(out) = string::encode_only(string::Kind::Fsst, &values)? else {
8402 return Ok(None);
8403 };
8404 Ok((out.len() < plain).then_some(out))
8405}
8406
8407fn encoded_codes(codes: &[u32]) -> Result<Option<Vec<u8>>> {
8408 let wide: Vec<i64> = codes.iter().map(|code| i64::from(*code)).collect();
8409 let coded = integer::encode_with(&wide, &Codes)?;
8410 let plain = codes.len().saturating_mul(size_of::<u32>());
8411 Ok((coded.len() < plain).then_some(coded))
8412}
8413
8414fn push_validity(out: &mut Vec<u8>, flat: &Vector) {
8417 let flag = match flat.validity() {
8418 Validity::AllValid => 0,
8419 Validity::AllInvalid => 1,
8420 Validity::Mask(_) => 2,
8421 };
8422 out.push(flag);
8423 if flag == 2 {
8424 for group in (0..flat.len()).step_by(8) {
8425 let mut bits = 0_u8;
8426 for bit in 0..8 {
8427 if group + bit < flat.len() && !flat.is_null_at(group + bit) {
8428 bits |= 1 << bit;
8429 }
8430 }
8431 out.push(bits);
8432 }
8433 }
8434}
8435
8436fn coded_page(codes: &[u32], validity: &[u8]) -> Result<Vec<u8>> {
8443 let coded = encoded_codes(codes)?;
8444 let mut out = Vec::with_capacity(
8445 1 + validity.len() + coded.as_ref().map_or(size_of_val(codes), Vec::len),
8446 );
8447 out.push(if coded.is_some() { 4 } else { 3 });
8448 out.extend_from_slice(validity);
8449 match coded {
8450 Some(coded) => out.extend_from_slice(&coded),
8451 None => {
8452 for &code in codes {
8453 put_u32(&mut out, code);
8454 }
8455 }
8456 }
8457 Ok(out)
8458}
8459
8460fn encode(vector: &Vector, settling: &mut Settling) -> Result<Vec<u8>> {
8463 let ty = vector.logical_type();
8464 let flat = vector.flatten()?;
8466 let mut out = Vec::new();
8467 let dictionary = if ty == &LogicalType::Varchar { string_dictionary(&flat)? } else { None };
8468 let compressed_text = if dictionary.is_none() && ty == &LogicalType::Varchar {
8469 text_compressed(&flat)?
8470 } else {
8471 None
8472 };
8473 let packed_vector = if dictionary.is_none() { Some(flat.bit_packed()?) } else { None };
8474 let packed = packed_vector.as_ref().and_then(Vector::packed_parts);
8475 let cascade =
8479 if dictionary.is_none() { cascaded(&flat, ty, packed.as_ref(), settling)? } else { None };
8480 out.push(if cascade.is_some() {
8481 5
8482 } else if dictionary.is_some() {
8483 1
8484 } else if compressed_text.is_some() {
8485 6
8486 } else if packed.is_some() {
8487 2
8488 } else {
8489 0
8490 });
8491 push_validity(&mut out, &flat);
8492 if let Some(cascade) = cascade {
8493 out.extend_from_slice(&cascade);
8494 return Ok(out);
8495 }
8496 if let Some(dictionary) = dictionary {
8497 out.extend_from_slice(&dictionary);
8498 return Ok(out);
8499 }
8500 if let Some(compressed_text) = compressed_text {
8501 out.extend_from_slice(&compressed_text);
8502 return Ok(out);
8503 }
8504 if let Some(packed) = packed {
8505 if packed.offset() != 0 {
8506 return Err(invalid("writer received a sliced packed vector"));
8507 }
8508 out.push(u8::try_from(packed.width()).map_err(|_| invalid("packed width overflow"))?);
8509 out.extend_from_slice(&packed.base().to_le_bytes());
8510 put_u32(
8511 &mut out,
8512 u32::try_from(packed.words().len()).map_err(|_| invalid("too many packed words"))?,
8513 );
8514 for word in packed.words() {
8515 put_u64(&mut out, *word);
8516 }
8517 return Ok(out);
8518 }
8519 let data = flat.data().ok_or_else(|| invalid("scalar column did not flatten"))?;
8520 match (ty, data) {
8521 (LogicalType::TinyInt, Data::Int8(values)) => {
8522 for value in &**values {
8523 out.extend_from_slice(&value.to_le_bytes());
8524 }
8525 }
8526 (LogicalType::UTinyInt, Data::UInt8(values)) => {
8527 for value in &**values {
8528 out.extend_from_slice(&value.to_le_bytes());
8529 }
8530 }
8531 (LogicalType::SmallInt, Data::Int16(values)) => {
8532 for value in &**values {
8533 out.extend_from_slice(&value.to_le_bytes());
8534 }
8535 }
8536 (LogicalType::USmallInt, Data::UInt16(values)) => {
8537 for value in &**values {
8538 out.extend_from_slice(&value.to_le_bytes());
8539 }
8540 }
8541 (LogicalType::UInteger, Data::UInt32(values)) => {
8542 for value in &**values {
8543 out.extend_from_slice(&value.to_le_bytes());
8544 }
8545 }
8546 (LogicalType::UBigInt, Data::UInt64(values)) => {
8547 for value in &**values {
8548 out.extend_from_slice(&value.to_le_bytes());
8549 }
8550 }
8551 (LogicalType::Integer | LogicalType::Date, Data::Int32(values)) => {
8552 for value in &**values {
8553 out.extend_from_slice(&value.to_le_bytes());
8554 }
8555 }
8556 (
8557 LogicalType::BigInt
8558 | LogicalType::Timestamp
8559 | LogicalType::Time
8560 | LogicalType::TimeTz
8561 | LogicalType::TimestampTz
8562 | LogicalType::TimestampS
8563 | LogicalType::TimestampMs
8564 | LogicalType::TimestampNs,
8565 Data::Int64(values),
8566 ) => {
8567 for value in &**values {
8568 out.extend_from_slice(&value.to_le_bytes());
8569 }
8570 }
8571 (LogicalType::HugeInt | LogicalType::Uuid, Data::Int128(values)) => {
8574 for value in &**values {
8575 out.extend_from_slice(&value.to_le_bytes());
8576 }
8577 }
8578 (LogicalType::UHugeInt, Data::UInt128(values)) => {
8579 for value in &**values {
8580 out.extend_from_slice(&value.to_le_bytes());
8581 }
8582 }
8583 (LogicalType::Float, Data::Float32(values)) => {
8586 for value in &**values {
8587 out.extend_from_slice(&value.to_le_bytes());
8588 }
8589 }
8590 (LogicalType::Double, Data::Float64(values)) => {
8591 for value in &**values {
8592 out.extend_from_slice(&value.to_le_bytes());
8593 }
8594 }
8595 (LogicalType::Interval, Data::Interval(values)) => {
8599 for (months, days, micros) in &**values {
8600 out.extend_from_slice(&months.to_le_bytes());
8601 out.extend_from_slice(&days.to_le_bytes());
8602 out.extend_from_slice(µs.to_le_bytes());
8603 }
8604 }
8605 (LogicalType::Boolean, Data::Bool(values)) => {
8606 for value in &**values {
8607 out.push(u8::from(*value));
8608 }
8609 }
8610 (LogicalType::Decimal { .. }, Data::Int16(values)) => {
8613 for value in &**values {
8614 out.extend_from_slice(&value.to_le_bytes());
8615 }
8616 }
8617 (LogicalType::Decimal { .. }, Data::Int32(values)) => {
8618 for value in &**values {
8619 out.extend_from_slice(&value.to_le_bytes());
8620 }
8621 }
8622 (LogicalType::Decimal { .. }, Data::Int64(values)) => {
8623 for value in &**values {
8624 out.extend_from_slice(&value.to_le_bytes());
8625 }
8626 }
8627 (LogicalType::Decimal { .. }, Data::Int128(values)) => {
8628 for value in &**values {
8629 out.extend_from_slice(&value.to_le_bytes());
8630 }
8631 }
8632 (LogicalType::Varchar | LogicalType::Blob | LogicalType::Bit, Data::Varlen(values)) => {
8637 let mut bytes = Vec::new();
8638 put_u32(&mut out, 0);
8639 for row in 0..vector.len() {
8640 let value = values.bytes(row).ok_or_else(|| invalid("string view is invalid"))?;
8641 bytes.extend_from_slice(value);
8642 put_u32(
8643 &mut out,
8644 u32::try_from(bytes.len())
8645 .map_err(|_| invalid("string payload exceeds 4GiB"))?,
8646 );
8647 }
8648 out.extend_from_slice(&bytes);
8649 }
8650 _ => return Err(Error::not_implemented(format!("native page for {ty}"))),
8651 }
8652 Ok(out)
8653}
8654
8655fn put_varint(out: &mut Vec<u8>, mut value: u32) {
8656 while value >= 0x80 {
8657 out.push((value as u8 & 0x7f) | 0x80);
8658 value >>= 7;
8659 }
8660 out.push(value as u8);
8661}
8662
8663fn unique_codes(codes: &[u32]) -> Vec<u32> {
8665 let mut unique = codes.to_vec();
8666 unique.sort_unstable();
8667 unique.dedup();
8668 unique
8669}
8670
8671fn merged_codes(lists: Vec<Vec<u32>>) -> Vec<u32> {
8677 let mut lists = lists;
8678 while lists.len() > 1 {
8679 let mut next = Vec::with_capacity(lists.len().div_ceil(2));
8680 for pair in lists.chunks(2) {
8681 match pair {
8682 [left, right] => next.push(merged_pair(left, right)),
8683 [only] => next.push(only.clone()),
8684 _ => {}
8685 }
8686 }
8687 lists = next;
8688 }
8689 lists.pop().unwrap_or_default()
8690}
8691
8692fn merged_pair(left: &[u32], right: &[u32]) -> Vec<u32> {
8693 let mut out = Vec::with_capacity(left.len().saturating_add(right.len()));
8694 let mut at = 0;
8695 let mut to = 0;
8696 while at < left.len() && to < right.len() {
8697 match left[at].cmp(&right[to]) {
8698 Ordering::Less => {
8699 out.push(left[at]);
8700 at += 1;
8701 }
8702 Ordering::Greater => {
8703 out.push(right[to]);
8704 to += 1;
8705 }
8706 Ordering::Equal => {
8707 out.push(left[at]);
8708 at += 1;
8709 to += 1;
8710 }
8711 }
8712 }
8713 out.extend_from_slice(&left[at..]);
8714 out.extend_from_slice(&right[to..]);
8715 out
8716}
8717
8718fn merged_range(ranges: impl Iterator<Item = Range>) -> Range {
8723 let mut merged = Range::default();
8724 let mut first = true;
8725 for range in ranges {
8726 merged.nulls = merged.nulls.saturating_add(range.nulls);
8727 merged.sum = match (merged.sum.take(), range.sum) {
8731 (Some(held), Some(next)) if !first => held.checked_add(next),
8732 (_, next) if first => next,
8733 _ => None,
8734 };
8735 merged.exact = if first { range.exact } else { merged.exact && range.exact };
8736 if first {
8737 merged.low = range.low;
8738 merged.high = range.high;
8739 first = false;
8740 continue;
8741 }
8742 merged.low = match (merged.low.take(), range.low) {
8743 (Some(held), Some(next)) => Some(held.smaller(next)),
8744 _ => None,
8745 };
8746 merged.high = match (merged.high.take(), range.high) {
8747 (Some(held), Some(next)) => Some(held.larger(next)),
8748 _ => None,
8749 };
8750 }
8751 merged
8752}
8753
8754fn shortened(bound: Option<Bound>, high: bool) -> Option<Bound> {
8767 match bound {
8768 Some(Bound::Bytes(mut value)) if value.len() > PART_BOUND_BYTES => {
8769 value.truncate(PART_BOUND_BYTES);
8770 if !high {
8771 return Some(Bound::Bytes(value));
8772 }
8773 while let Some(last) = value.pop() {
8774 if last < u8::MAX {
8775 value.push(last + 1);
8776 return Some(Bound::Bytes(value));
8777 }
8778 }
8779 None
8780 }
8781 other => other,
8782 }
8783}
8784
8785fn encode_part_ranges(ranges: &[Range]) -> Result<Vec<u8>> {
8793 let mut out = Vec::new();
8794 put_u32(
8795 &mut out,
8796 u32::try_from(ranges.len()).map_err(|_| invalid("too many parts in a stripe"))?,
8797 );
8798 for range in ranges {
8799 put_bound(&mut out, shortened(range.low.clone(), false).as_ref())?;
8800 put_bound(&mut out, shortened(range.high.clone(), true).as_ref())?;
8801 put_u32(&mut out, u32::try_from(range.nulls).map_err(|_| invalid("null count overflow"))?);
8802 }
8803 Ok(out)
8804}
8805
8806fn decode_part_ranges(bytes: &[u8]) -> Result<Vec<Range>> {
8808 let mut cur = Cursor::new(bytes);
8809 let parts = cur.u32()? as usize;
8810 let mut out = Vec::new();
8811 for _ in 0..parts {
8812 let low = cur.bound()?;
8813 let high = cur.bound()?;
8814 let nulls = cur.u32()? as usize;
8815 out.push(Range { low, high, nulls, exact: false, sum: None });
8816 }
8817 Ok(out)
8818}
8819
8820fn encode_sieves<'a>(sieves: impl Iterator<Item = &'a Option<Sieve>>) -> Result<Vec<u8>> {
8821 let held: Vec<&Option<Sieve>> = sieves.collect();
8822 let mut out = Vec::new();
8823 put_u32(
8824 &mut out,
8825 u32::try_from(held.len()).map_err(|_| invalid("too many parts in a stripe"))?,
8826 );
8827 for sieve in &held {
8828 let length = sieve.as_ref().map_or(0, Sieve::len);
8829 put_u32(&mut out, u32::try_from(length).map_err(|_| invalid("sieve length overflow"))?);
8830 }
8831 for sieve in held.into_iter().flatten() {
8833 out.extend_from_slice(&sieve.to_bytes());
8834 }
8835 Ok(out)
8836}
8837
8838fn decode_sieves(bytes: &[u8]) -> Result<Vec<Option<Sieve>>> {
8844 let parts = u32::from_le_bytes(
8845 bytes
8846 .get(..4)
8847 .ok_or_else(|| invalid("sieve page is truncated"))?
8848 .try_into()
8849 .map_err(|_| invalid("sieve page is truncated"))?,
8850 ) as usize;
8851 let mut lengths = Vec::with_capacity(parts);
8852 for part in 0..parts {
8853 let at = 4 + part * 4;
8854 let field = bytes.get(at..at + 4).ok_or_else(|| invalid("sieve page is truncated"))?;
8855 lengths.push(u32::from_le_bytes(
8856 field.try_into().map_err(|_| invalid("sieve page is truncated"))?,
8857 ) as usize);
8858 }
8859 let mut at = 4 + parts * 4;
8860 let mut out = Vec::with_capacity(parts);
8861 for length in lengths {
8862 if length == 0 {
8863 out.push(None);
8864 continue;
8865 }
8866 let end = at.checked_add(length).ok_or_else(|| invalid("sieve page is truncated"))?;
8867 let field = bytes.get(at..end).ok_or_else(|| invalid("sieve page is truncated"))?;
8868 out.push(Sieve::from_bytes(field));
8869 at = end;
8870 }
8871 if at != bytes.len() {
8872 return Err(invalid("sieve page has trailing bytes"));
8873 }
8874 Ok(out)
8875}
8876
8877fn encode_membership(unique: &[u32]) -> Vec<u8> {
8883 let mut out = Vec::with_capacity(unique.len().saturating_mul(2).saturating_add(5));
8884 put_varint(&mut out, u32::try_from(unique.len()).unwrap_or(u32::MAX));
8885 let mut previous = 0;
8886 for (at, &code) in unique.iter().enumerate() {
8887 put_varint(&mut out, if at == 0 { code } else { code - previous });
8888 previous = code;
8889 }
8890 out
8891}
8892
8893fn take_varint(bytes: &[u8], at: &mut usize) -> Result<u32> {
8894 let mut value = 0_u32;
8895 for shift in (0..35).step_by(7) {
8896 let byte = *bytes.get(*at).ok_or_else(|| invalid("membership varint is truncated"))?;
8897 *at += 1;
8898 let part = u32::from(byte & 0x7f);
8899 if shift == 28 && part > 0x0f {
8900 return Err(invalid("membership varint overflow"));
8901 }
8902 value = value
8903 .checked_add(
8904 part.checked_shl(shift).ok_or_else(|| invalid("membership varint overflow"))?,
8905 )
8906 .ok_or_else(|| invalid("membership varint overflow"))?;
8907 if byte & 0x80 == 0 {
8908 return Ok(value);
8909 }
8910 }
8911 Err(invalid("membership varint is too long"))
8912}
8913
8914fn decode_membership(bytes: &[u8]) -> Result<Vec<u32>> {
8915 let mut at = 0;
8916 let count = take_varint(bytes, &mut at)? as usize;
8917 let mut codes = Vec::with_capacity(count);
8918 let mut previous = 0_u32;
8919 for index in 0..count {
8920 let delta = take_varint(bytes, &mut at)?;
8921 let code = if index == 0 {
8922 delta
8923 } else {
8924 previous.checked_add(delta).ok_or_else(|| invalid("membership code overflow"))?
8925 };
8926 if index > 0 && code <= previous {
8927 return Err(invalid("membership codes are not increasing"));
8928 }
8929 codes.push(code);
8930 previous = code;
8931 }
8932 if at != bytes.len() {
8933 return Err(invalid("membership page has trailing bytes"));
8934 }
8935 Ok(codes)
8936}
8937
8938fn string_dictionary(vector: &Vector) -> Result<Option<Vec<u8>>> {
8939 let mut by_text = HashMap::new();
8940 let mut values = Vec::new();
8941 let mut codes = Vec::with_capacity(vector.len());
8942 let mut plain_bytes = 0_usize;
8943 for row in 0..vector.len() {
8944 let text = vector.text_at(row).unwrap_or("");
8945 plain_bytes = plain_bytes.saturating_add(text.len());
8946 let code = match by_text.get(text) {
8947 Some(&code) => code,
8948 None => {
8949 let code = u32::try_from(values.len())
8950 .map_err(|_| invalid("too many dictionary values"))?;
8951 by_text.insert(text, code);
8952 values.push(text);
8953 code
8954 }
8955 };
8956 codes.push(code);
8957 }
8958 let dictionary_bytes = values.iter().map(|value| value.len()).sum::<usize>();
8959 let encoded = 8_usize
8960 .saturating_add((values.len() + 1).saturating_mul(4))
8961 .saturating_add(dictionary_bytes)
8962 .saturating_add(codes.len().saturating_mul(4));
8963 let plain = (vector.len() + 1).saturating_mul(4).saturating_add(plain_bytes);
8964 if encoded >= plain {
8965 return Ok(None);
8966 }
8967 let mut out = Vec::with_capacity(encoded);
8968 put_u32(
8969 &mut out,
8970 u32::try_from(values.len()).map_err(|_| invalid("too many dictionary values"))?,
8971 );
8972 put_u32(
8973 &mut out,
8974 u32::try_from(dictionary_bytes).map_err(|_| invalid("dictionary payload exceeds 4GiB"))?,
8975 );
8976 let mut offset = 0_u32;
8977 put_u32(&mut out, offset);
8978 for value in &values {
8979 offset = offset
8980 .checked_add(
8981 u32::try_from(value.len()).map_err(|_| invalid("dictionary value is too long"))?,
8982 )
8983 .ok_or_else(|| invalid("dictionary payload exceeds 4GiB"))?;
8984 put_u32(&mut out, offset);
8985 }
8986 for value in values {
8987 out.extend_from_slice(value.as_bytes());
8988 }
8989 for code in codes {
8990 put_u32(&mut out, code);
8991 }
8992 Ok(Some(out))
8993}
8994
8995struct EncodedDictionary {
8996 index: Vec<u8>,
8997 ranks: Vec<u8>,
8998 grams: Vec<u8>,
8999}
9000
9001fn sort_by_value<'a>(codes: &mut [u32], values: impl Fn(u32) -> &'a [u8]) {
9042 let mut work = vec![(0, codes.len(), 0)];
9043 let mut keyed: Vec<(u64, u8, u32)> = Vec::new();
9044 while let Some((from, to, depth)) = work.pop() {
9045 let part = &mut codes[from..to];
9046 keyed.clear();
9047 keyed.extend(part.iter().map(|&code| {
9048 let value = values(code);
9049 let rest = value.get(depth..).unwrap_or_default();
9050 (head(rest), rest.len().min(8) as u8, code)
9051 }));
9052 keyed.sort_unstable();
9053 for (slot, entry) in part.iter_mut().zip(keyed.iter()) {
9054 *slot = entry.2;
9055 }
9056 let mut start = 0;
9057 while start < keyed.len() {
9058 let (key, taken, _) = keyed[start];
9059 let mut end = start + 1;
9060 while end < keyed.len() && keyed[end].0 == key && keyed[end].1 == taken {
9061 end += 1;
9062 }
9063 if taken == 8 && end - start > 1 {
9064 work.push((from + start, from + end, depth + 8));
9065 }
9066 start = end;
9067 }
9068 }
9069}
9070
9071const PARALLEL_SORT_MIN: usize = 1 << 16;
9073
9074const BUCKETS_PER_WORKER: usize = 4;
9077
9078const SAMPLES_PER_BUCKET: usize = 32;
9080
9081fn sort_by_value_across<'a>(
9099 codes: &mut [u32],
9100 values: impl Fn(u32) -> &'a [u8] + Sync,
9101 workers: usize,
9102) {
9103 if workers <= 1 || codes.len() < PARALLEL_SORT_MIN {
9104 sort_by_value(codes, values);
9105 return;
9106 }
9107 let buckets = workers * BUCKETS_PER_WORKER;
9108 let wanted = buckets * SAMPLES_PER_BUCKET;
9109 let mut sample = (0..wanted).map(|at| codes[at * codes.len() / wanted]).collect::<Vec<_>>();
9110 sort_by_value(&mut sample, &values);
9111 let splitters =
9112 (1..buckets).map(|cut| values(sample[cut * sample.len() / buckets])).collect::<Vec<_>>();
9113 let values = &values;
9114 let splitters = &splitters;
9115 let per = codes.len().div_ceil(workers);
9116 let places = std::thread::scope(|scope| {
9118 codes
9119 .chunks(per)
9120 .map(|run| {
9121 scope.spawn(move || {
9122 run.iter()
9123 .map(|&code| {
9124 let value = values(code);
9125 splitters.partition_point(|splitter| *splitter <= value) as u32
9126 })
9127 .collect::<Vec<_>>()
9128 })
9129 })
9130 .collect::<Vec<_>>()
9131 .into_iter()
9132 .flat_map(|handle| {
9133 handle.join().unwrap_or_else(|panic| std::panic::resume_unwind(panic))
9134 })
9135 .collect::<Vec<_>>()
9136 });
9137 let mut starts = vec![0_usize; buckets + 1];
9138 for &place in &places {
9139 starts[place as usize + 1] += 1;
9140 }
9141 for bucket in 0..buckets {
9142 starts[bucket + 1] += starts[bucket];
9143 }
9144 let mut laid = vec![0_u32; codes.len()];
9145 let mut next = starts.clone();
9146 for (&code, &place) in codes.iter().zip(&places) {
9147 laid[next[place as usize]] = code;
9148 next[place as usize] += 1;
9149 }
9150 drop(places);
9151 let mut runs = Vec::with_capacity(buckets);
9152 let mut rest = laid.as_mut_slice();
9153 for bucket in 0..buckets {
9154 let (run, after) = rest.split_at_mut(starts[bucket + 1] - starts[bucket]);
9155 runs.push(run);
9156 rest = after;
9157 }
9158 runs.sort_by_key(|run| run.len());
9160 let queue = Mutex::new(runs);
9161 std::thread::scope(|scope| {
9162 for _ in 0..workers {
9163 scope.spawn(|| {
9164 loop {
9165 let taken =
9166 queue.lock().unwrap_or_else(std::sync::PoisonError::into_inner).pop();
9167 let Some(run) = taken else { break };
9168 sort_by_value(run, values);
9169 }
9170 });
9171 }
9172 });
9173 codes.copy_from_slice(&laid);
9174}
9175
9176fn head(bytes: &[u8]) -> u64 {
9178 let mut word = [0; 8];
9179 let take = bytes.len().min(8);
9180 word[..take].copy_from_slice(&bytes[..take]);
9181 u64::from_be_bytes(word)
9182}
9183
9184fn encode_global_dictionary(
9195 dictionary: &GlobalDictionary,
9196 order: &[(u64, u32)],
9197 places: &[Placed],
9198 scattered: bool,
9199) -> Result<EncodedDictionary> {
9200 let values = dictionary.values();
9201 if order.len() != values {
9202 return Err(invalid("global dictionary order does not cover its values"));
9203 }
9204 let blocks = values.div_ceil(TEXT_PAYLOAD_VALUES);
9205 if places.len() != blocks {
9206 return Err(invalid("global dictionary payload is not the blocks it says it is"));
9207 }
9208 if dictionary.grams.len() != blocks {
9209 return Err(invalid("global dictionary signatures do not cover its blocks"));
9210 }
9211 let (ranks, rank_ends) = encode_ranks(order, code_width(values))?;
9212 let rank_blocks = values.div_ceil(TEXT_RANK_BLOCK);
9213 let offset_bits = offset_width(&dictionary.ends);
9214 let payload_words = if scattered { 3 } else { 2 };
9215 let index_len = DICTIONARY_HEADER
9216 .checked_add(offset_bytes(values, offset_bits))
9217 .and_then(|len| len.checked_add(blocks.checked_mul(payload_words * 8)?))
9218 .and_then(|len| len.checked_add(rank_blocks.checked_mul(16)?))
9219 .and_then(|len| len.checked_add(8))
9220 .ok_or_else(|| invalid("global dictionary index length overflow"))?;
9221 let mut index = Vec::with_capacity(index_len);
9222 put_u32(
9223 &mut index,
9224 u32::try_from(values).map_err(|_| invalid("global dictionary has too many values"))?,
9225 );
9226 put_u32(&mut index, TEXT_PAYLOAD_VALUES as u32);
9227 put_u32(
9228 &mut index,
9229 u32::try_from(blocks).map_err(|_| invalid("global dictionary has too many blocks"))?,
9230 );
9231 let flag = (if scattered { DICTIONARY_SCATTERED } else { 0 }) | DICTIONARY_GRAMS;
9232 put_u32(&mut index, offset_bits as u32 | flag);
9233 encode_offsets(&dictionary.ends, offset_bits, &mut index)?;
9234 let mut end = 0_u64;
9239 for place in places {
9240 if scattered {
9241 put_u64(&mut index, place.start);
9242 put_u64(&mut index, place.length);
9243 } else {
9244 end = end
9245 .checked_add(place.length)
9246 .ok_or_else(|| invalid("global dictionary payload overflow"))?;
9247 put_u64(&mut index, end);
9248 }
9249 }
9250 for place in places {
9251 put_u64(&mut index, place.hash);
9252 }
9253 if rank_ends.len() != rank_blocks {
9256 return Err(invalid("global dictionary order is not the blocks it says it is"));
9257 }
9258 for end in &rank_ends {
9259 put_u64(&mut index, *end);
9260 }
9261 let mut at = 0_usize;
9262 for end in &rank_ends {
9263 let end = usize::try_from(*end).map_err(|_| invalid("global dictionary order overflow"))?;
9264 put_u64(&mut index, checksum(&ranks[at..end]));
9265 at = end;
9266 }
9267 let gram_len = blocks
9268 .checked_mul(TEXT_GRAM_BYTES)
9269 .ok_or_else(|| invalid("global dictionary signature count overflow"))?;
9270 let mut grams = Vec::with_capacity(gram_len);
9271 for block in &dictionary.grams {
9272 grams.extend_from_slice(block);
9273 }
9274 put_u64(&mut index, checksum(&grams));
9275 if index.len() != index_len {
9276 return Err(invalid("global dictionary index is not the length it was laid out for"));
9277 }
9278 Ok(EncodedDictionary { index, ranks, grams })
9279}
9280
9281const PAYLOAD_SAMPLE_BLOCKS: usize = 8;
9288
9289fn payload_shapes() -> Vec<chooser::Settled> {
9315 let integers = vec![integer::Kind::Packed];
9316 [
9317 vec![string::Kind::Front, string::Kind::Lz],
9318 vec![string::Kind::Lz, string::Kind::Fsst],
9319 vec![string::Kind::Lz, string::Kind::Plain],
9320 vec![string::Kind::Fsst],
9321 vec![string::Kind::Plain],
9322 ]
9323 .into_iter()
9324 .map(|strings| chooser::Settled::new(strings, integers.clone()))
9325 .collect()
9326}
9327
9328fn synced(file: &File, profile: Option<&LoadProfile>) -> Result<()> {
9350 let started = profile.map(|_| std::time::Instant::now());
9351 file.sync_all().map_err(io)?;
9352 if let (Some(profile), Some(started)) = (profile, started) {
9353 profile.waited(
9354 Stage::Publish,
9355 u64::try_from(started.elapsed().as_nanos()).unwrap_or(u64::MAX),
9356 );
9357 }
9358 Ok(())
9359}
9360
9361fn encode_ready(dictionaries: &mut [Option<GlobalDictionary>]) -> Result<()> {
9362 for dictionary in dictionaries.iter_mut().flatten() {
9363 dictionary.settle()?;
9364 }
9365 encode_waiting(dictionaries, false)
9370}
9371
9372fn finish_dictionaries(dictionaries: &mut [Option<GlobalDictionary>]) -> Result<()> {
9379 for dictionary in dictionaries.iter_mut().flatten() {
9380 dictionary.seal_rest();
9381 }
9382 encode_waiting(dictionaries, true)
9383}
9384
9385fn encode_waiting(dictionaries: &mut [Option<GlobalDictionary>], closing: bool) -> Result<()> {
9388 let jobs = dictionaries
9389 .iter()
9390 .enumerate()
9391 .filter(|(_, held)| held.as_ref().is_some_and(|held| closing || held.shape.is_some()))
9392 .flat_map(|(column, held)| {
9393 (0..held.as_ref().map_or(0, |held| held.waiting.len())).map(move |at| (column, at))
9394 })
9395 .collect::<Vec<_>>();
9396 if jobs.is_empty() {
9397 return Ok(());
9398 }
9399 let one = |column: usize, at: usize| -> Result<(usize, usize, Vec<u8>)> {
9400 let held = dictionaries[column].as_ref().ok_or_else(|| Error::internal("no dictionary"))?;
9401 Ok((column, at, held.encode_waiting(at)?))
9402 };
9403 let workers = std::thread::available_parallelism()
9404 .map_or(1, usize::from)
9405 .min(MAX_FREQUENCY_WORKERS)
9406 .min(jobs.len());
9407 let made = if workers <= 1 {
9408 jobs.iter().map(|&(column, at)| one(column, at)).collect::<Result<Vec<_>>>()?
9409 } else {
9410 let next = AtomicUsize::new(0);
9411 let jobs = &jobs;
9412 let pieces = std::thread::scope(|scope| {
9413 (0..workers)
9414 .map(|_| {
9415 scope.spawn(|| {
9416 let mut mine = Vec::new();
9417 loop {
9418 let job = next.fetch_add(1, Atomic::Relaxed);
9419 let Some(&(column, at)) = jobs.get(job) else { break };
9420 mine.push(one(column, at)?);
9421 }
9422 Ok(mine)
9423 })
9424 })
9425 .collect::<Vec<_>>()
9426 .into_iter()
9427 .map(|handle| {
9428 handle
9429 .join()
9430 .map_err(|_| Error::internal("a dictionary encode worker panicked"))?
9431 })
9432 .collect::<Result<Vec<_>>>()
9433 })?;
9434 pieces.into_iter().flatten().collect()
9435 };
9436 let mut done: Vec<Vec<(usize, Vec<u8>)>> =
9437 (0..dictionaries.len()).map(|_| Vec::new()).collect();
9438 for (column, at, bytes) in made {
9439 done[column].push((at, bytes));
9440 }
9441 for (column, mut made) in done.into_iter().enumerate() {
9442 if made.is_empty() {
9443 continue;
9444 }
9445 let Some(held) = dictionaries[column].as_mut() else { continue };
9446 made.sort_by_key(|(at, _)| *at);
9447 let waiting = std::mem::take(&mut held.waiting);
9448 for ((block, _), (_, bytes)) in waiting.into_iter().zip(made) {
9449 if held.encoded() != block {
9450 return Err(Error::internal("a dictionary block was encoded out of order"));
9451 }
9452 held.blocks.push(bytes);
9453 }
9454 }
9455 Ok(())
9456}
9457
9458fn settle_shape(sample: &[Vec<&[u8]>]) -> Result<chooser::Settled> {
9468 let mut best: Option<(chooser::Settled, usize)> = None;
9469 for shape in payload_shapes() {
9470 let mut size = 0;
9471 for block in sample {
9472 size += string::encode_with(block, &shape)?.len();
9473 }
9474 if best.as_ref().is_none_or(|(_, smallest)| size < *smallest) {
9475 best = Some((shape, size));
9476 }
9477 }
9478 best.map(|(shape, _)| shape)
9479 .ok_or_else(|| invalid("no shape applies to a global dictionary payload"))
9480}
9481
9482fn encode_ranks(order: &[(u64, u32)], code_bits: usize) -> Result<(Vec<u8>, Vec<u64>)> {
9489 let mut out = Vec::with_capacity(order.len() * 4);
9490 let mut ends = Vec::with_capacity(order.len().div_ceil(TEXT_RANK_BLOCK));
9491 let mut heads = Vec::with_capacity(TEXT_RANK_BLOCK);
9492 let mut codes = Vec::with_capacity(TEXT_RANK_BLOCK);
9493 for block in order.chunks(TEXT_RANK_BLOCK) {
9494 let base = block.first().map_or(0, |&(head, _)| head);
9497 let span = block.last().map_or(0, |&(head, _)| head.wrapping_sub(base));
9498 let width = (u64::BITS - span.leading_zeros()) as usize;
9499 heads.clear();
9500 codes.clear();
9501 for &(head, code) in block {
9502 heads.push(head.wrapping_sub(base));
9503 codes.push(u64::from(code));
9504 }
9505 put_u64(&mut out, base);
9506 out.push(width as u8);
9507 bitpack::pack_tail(&heads, width, &mut out)
9508 .map_err(|_| invalid("global dictionary heads do not pack"))?;
9509 bitpack::pack_tail(&codes, code_bits, &mut out)
9510 .map_err(|_| invalid("global dictionary codes do not pack"))?;
9511 ends.push(out.len() as u64);
9512 }
9513 Ok((out, ends))
9514}
9515
9516fn open_global_dictionary(
9523 file: Arc<File>,
9524 page: Page,
9525 ty: &LogicalType,
9526 keep_budget: usize,
9527) -> Result<Vector> {
9528 if ty != &LogicalType::Varchar {
9529 return Err(invalid("global dictionary belongs to a non-string column"));
9530 }
9531 let mut header = [0; DICTIONARY_HEADER];
9532 read_at(&file, page.offset, &mut header)?;
9533 let count = u32::from_le_bytes(header[0..4].try_into().expect("four bytes")) as usize;
9534 let per_block = u32::from_le_bytes(header[4..8].try_into().expect("four bytes")) as usize;
9535 let blocks = u32::from_le_bytes(header[8..12].try_into().expect("four bytes")) as usize;
9536 let width = u32::from_le_bytes(header[12..16].try_into().expect("four bytes"));
9537 let scattered = width & DICTIONARY_SCATTERED != 0;
9538 let has_grams = width & DICTIONARY_GRAMS != 0;
9539 let offset_bits = (width & !(DICTIONARY_SCATTERED | DICTIONARY_GRAMS)) as usize;
9540 if per_block != TEXT_PAYLOAD_VALUES {
9541 return Err(invalid("global dictionary block width differs"));
9542 }
9543 if blocks != count.div_ceil(TEXT_PAYLOAD_VALUES) {
9544 return Err(invalid("global dictionary block count differs from its value count"));
9545 }
9546 if offset_bits > u32::BITS as usize {
9547 return Err(invalid("global dictionary packs offsets past a payload"));
9548 }
9549 let offset_len = offset_bytes(count, offset_bits);
9550 let ranks = count;
9555 let rank_blocks = ranks.div_ceil(TEXT_RANK_BLOCK);
9556 let payload_words = if scattered { 3 } else { 2 };
9560 let hash_len = blocks
9561 .checked_mul(payload_words * 8)
9562 .and_then(|len| len.checked_add(rank_blocks.checked_mul(16)?))
9563 .and_then(|len| len.checked_add(usize::from(has_grams) * 8))
9564 .ok_or_else(|| invalid("global dictionary block count overflow"))?;
9565 let gram_len = if has_grams {
9566 blocks
9567 .checked_mul(TEXT_GRAM_BYTES)
9568 .ok_or_else(|| invalid("global dictionary signature count overflow"))?
9569 } else {
9570 0
9571 };
9572 let index_len = DICTIONARY_HEADER
9573 .checked_add(offset_len)
9574 .and_then(|len| len.checked_add(hash_len))
9575 .ok_or_else(|| invalid("global dictionary header overflow"))?;
9576 if index_len > page.length as usize {
9577 return Err(invalid("global dictionary offset index exceeds its page"));
9578 }
9579 let mut index = vec![0; index_len];
9580 index[..DICTIONARY_HEADER].copy_from_slice(&header);
9581 read_at(&file, page.offset + DICTIONARY_HEADER as u64, &mut index[DICTIONARY_HEADER..])?;
9582 if checksum(&index) != page.hash {
9583 return Err(invalid("global dictionary index checksum differs"));
9584 }
9585 let offsets = index[DICTIONARY_HEADER..DICTIONARY_HEADER + offset_len].to_vec();
9586 let word_end = index_len - usize::from(has_grams) * 8;
9587 let gram_hash = has_grams
9588 .then(|| u64::from_le_bytes(index[word_end..index_len].try_into().expect("eight bytes")));
9589 let mut words = index[DICTIONARY_HEADER + offset_len..word_end]
9590 .chunks_exact(8)
9591 .map(|part| u64::from_le_bytes(part.try_into().expect("eight bytes")))
9592 .collect::<Vec<_>>();
9593 let mut rest = words.split_off(blocks * payload_words);
9594 let rank_hashes = rest.split_off(rank_blocks);
9595 let rank_ends = rest;
9596 if rank_ends.windows(2).any(|pair| pair[0] >= pair[1]) {
9599 return Err(invalid("global dictionary order blocks do not rise"));
9600 }
9601 let rank_len = usize::try_from(rank_ends.last().copied().unwrap_or_default())
9602 .map_err(|_| invalid("global dictionary rank overflow"))?;
9603 let body_len = index_len
9604 .checked_add(rank_len)
9605 .ok_or_else(|| invalid("global dictionary header overflow"))?;
9606 if body_len > page.length as usize {
9607 return Err(invalid("global dictionary order exceeds its page"));
9608 }
9609 let gram_end = body_len
9610 .checked_add(gram_len)
9611 .ok_or_else(|| invalid("global dictionary signature length overflow"))?;
9612 if gram_end > page.length as usize {
9613 return Err(invalid("global dictionary signatures exceed their page"));
9614 }
9615 let grams = gram_hash.map(|hash| NativeGrams {
9616 start: page.offset + body_len as u64,
9617 length: gram_len,
9618 hash,
9619 loaded: OnceLock::new(),
9620 });
9621 let hashes = words.split_off(blocks * (payload_words - 1));
9622 let (starts, lengths) = if scattered {
9623 let mut starts = Vec::with_capacity(blocks);
9624 let mut lengths = Vec::with_capacity(blocks);
9625 for pair in words.chunks_exact(2) {
9626 starts.push(pair[0]);
9627 lengths.push(pair[1]);
9628 }
9629 (starts, lengths)
9630 } else {
9631 let base = page.offset + gram_end as u64;
9635 let mut starts = Vec::with_capacity(blocks);
9636 let mut lengths = Vec::with_capacity(blocks);
9637 let mut at = 0_u64;
9638 for &end in &words {
9639 let len = end
9640 .checked_sub(at)
9641 .ok_or_else(|| invalid("global dictionary block ends before it starts"))?;
9642 starts.push(base + at);
9643 lengths.push(len);
9644 at = end;
9645 }
9646 (starts, lengths)
9647 };
9648 let stored_len = page.length as u64 - gram_end as u64;
9654 if scattered && stored_len == 0 {
9655 let size = file.metadata().map_err(io)?.len();
9656 let inside = starts.iter().zip(&lengths).all(|(&start, &len)| {
9657 start >= HEADER && start.checked_add(len).is_some_and(|end| end <= size)
9658 });
9659 if !inside {
9660 return Err(invalid("global dictionary block lies outside the file"));
9661 }
9662 } else if lengths.iter().try_fold(0_u64, |sum, len| sum.checked_add(*len)) != Some(stored_len) {
9663 return Err(invalid("global dictionary blocks do not bound the payload"));
9664 }
9665 Vector::external_text(
9666 LogicalType::Varchar,
9667 Arc::new(NativeText {
9668 file,
9669 values: count,
9670 offsets,
9671 offset_bits,
9672 value_ends: OnceLock::new(),
9673 value_lens: OnceLock::new(),
9674 ends_asked: AtomicUsize::new(0),
9675 ranks,
9676 rank_at: page.offset + index_len as u64,
9677 rank_ends,
9678 rank_hashes,
9679 rank_blocks: (0..rank_blocks).map(|_| OnceLock::new()).collect(),
9680 code_bits: code_width(count),
9681 code_ranks: OnceLock::new(),
9682 starts,
9683 lengths,
9684 hashes,
9685 grams,
9686 blocks: (0..blocks).map(|_| OnceLock::new()).collect(),
9687 keep_budget,
9688 payload_kept: AtomicUsize::new(0),
9689 swept: (0..blocks).map(|_| AtomicBool::new(false)).collect(),
9690 searched: Mutex::new(HashMap::new()),
9691 }),
9692 )
9693}
9694
9695fn page_encoding(ty: &LogicalType, rows: usize, bytes: &[u8]) -> String {
9708 fn cascade_at(rows: usize, bytes: &[u8]) -> Result<(u8, usize)> {
9710 let mut cur = Cursor::new(bytes);
9711 let codec = cur.u8()?;
9712 if cur.u8()? == 2 {
9713 cur.take(rows.div_ceil(8))?;
9714 }
9715 Ok((codec, cur.at))
9716 }
9717 let Ok((codec, at)) = cascade_at(rows, bytes) else {
9718 return "UNREADABLE".to_string();
9719 };
9720 let tail = &bytes[at..];
9721 let described = |described: Result<String>| described.unwrap_or_else(|_| "UNREADABLE".into());
9722 match codec {
9723 0 => match ty {
9724 LogicalType::Varchar | LogicalType::Blob => "PLAIN".to_string(),
9725 _ => "FIXED".to_string(),
9726 },
9727 1 => "DICT(PLAIN)".to_string(),
9728 2 => "FOR+BITPACK".to_string(),
9729 3 => "TABLE DICT".to_string(),
9730 4 => format!("TABLE DICT({})", described(integer::describe(tail))),
9731 5 => described(integer::describe(tail)),
9732 6 => described(string::describe(tail)),
9733 other => format!("CODEC {other}"),
9734 }
9735}
9736
9737fn decode_selected_stable_codes(
9742 rows: usize,
9743 bytes: &[u8],
9744 positions: &[usize],
9745 out: &mut Vec<Option<u32>>,
9746) -> Result<bool> {
9747 if positions.windows(2).any(|pair| pair[0] >= pair[1])
9748 || positions.last().is_some_and(|&position| position >= rows)
9749 {
9750 return Err(invalid("selected code positions are not sorted and in range"));
9751 }
9752 let mut cur = Cursor::new(bytes);
9753 let codec = cur.u8()?;
9754 if codec != 3 && codec != 4 {
9755 return Ok(false);
9756 }
9757 let flag = cur.u8()?;
9758 let mask = match flag {
9759 0 | 1 => None,
9760 2 => {
9761 let at = cur.at;
9762 let len = rows.div_ceil(8);
9763 cur.take(len)?;
9764 Some((at, len))
9765 }
9766 _ => return Err(invalid("page validity tag differs")),
9767 };
9768 let valid = |row: usize| match flag {
9769 0 => true,
9770 1 => false,
9771 2 => mask.is_some_and(|(at, _)| bytes[at + row / 8] >> (row % 8) & 1 == 1),
9772 _ => unreachable!("the validity tag was checked"),
9773 };
9774 if codec == 4 {
9775 let wide = integer::decode_selected(&bytes[cur.at..], positions)?;
9776 for (&row, code) in positions.iter().zip(wide) {
9777 let code = u32::try_from(code).map_err(|_| invalid("code is not a code"))?;
9778 out.push(valid(row).then_some(code));
9779 }
9780 return Ok(true);
9781 }
9782 let codes_at = cur.at;
9783 let codes_len = rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?;
9784 cur.take(codes_len)?;
9785 if cur.at != bytes.len() {
9786 return Err(invalid("global code page has trailing bytes"));
9787 }
9788 let codes = &bytes[codes_at..codes_at + codes_len];
9789 for &row in positions {
9790 let at = row.checked_mul(4).ok_or_else(|| invalid("dictionary code offset overflow"))?;
9791 let code = u32::from_le_bytes(
9792 codes[at..at + 4].try_into().map_err(|_| invalid("dictionary code is truncated"))?,
9793 );
9794 out.push(valid(row).then_some(code));
9795 }
9796 Ok(true)
9797}
9798
9799fn decode(
9800 ty: &LogicalType,
9801 rows: usize,
9802 bytes: &[u8],
9803 global: Option<Arc<Vector>>,
9804) -> Result<Vector> {
9805 let mut cur = Cursor::new(bytes);
9806 let codec = cur.u8()?;
9807 let flag = cur.u8()?;
9808 let validity = match flag {
9809 0 => Validity::AllValid,
9810 1 => Validity::AllInvalid,
9811 2 => {
9812 let mask = cur.take(rows.div_ceil(8))?;
9813 Validity::from_iter(rows, |row| mask[row / 8] >> (row % 8) & 1 == 1)
9814 }
9815 _ => return Err(invalid("page validity tag differs")),
9816 };
9817 if codec == 1 {
9818 if ty != &LogicalType::Varchar {
9819 return Err(invalid("dictionary codec belongs to a non-string page"));
9820 }
9821 let count = cur.u32()? as usize;
9822 let payload_len = cur.u32()? as usize;
9823 let offset_bytes = cur.take(
9824 (count + 1)
9825 .checked_mul(4)
9826 .ok_or_else(|| invalid("dictionary offset count overflow"))?,
9827 )?;
9828 let offsets = offset_bytes
9829 .chunks_exact(4)
9830 .map(|part| u32::from_le_bytes(part.try_into().expect("four bytes")))
9831 .collect::<Vec<_>>();
9832 let payload = cur.take(payload_len)?.to_vec();
9833 if offsets.first() != Some(&0)
9834 || offsets.last().copied().map(|last| last as usize) != Some(payload.len())
9835 || offsets.windows(2).any(|pair| pair[0] > pair[1])
9836 {
9837 return Err(invalid("dictionary offsets do not bound the payload"));
9838 }
9839 let mut strings = StringColumn::over(Buffer::from_vec(payload).into_page());
9842 for pair in offsets.windows(2) {
9843 strings.push_in_place(pair[0] as usize, (pair[1] - pair[0]) as usize)?;
9844 }
9845 let mut codes = Vec::with_capacity(rows);
9846 for _ in 0..rows {
9847 codes.push(cur.u32()?);
9848 }
9849 if codes.iter().any(|code| *code as usize >= count) {
9850 return Err(invalid("dictionary code is out of range"));
9851 }
9852 if cur.at != bytes.len() {
9853 return Err(invalid("dictionary page has trailing bytes"));
9854 }
9855 let dictionary = Vector::flat(LogicalType::Varchar, Data::Varlen(strings))?;
9856 return Ok(Vector::dictionary(codes, dictionary)?.with_validity(validity));
9857 }
9858 if codec == 3 || codec == 4 {
9859 let dictionary = global.ok_or_else(|| invalid("global code page has no dictionary"))?;
9860 let codes = if codec == 4 {
9861 let wide = integer::decode(&bytes[cur.at..])?;
9864 if wide.len() != rows {
9865 return Err(invalid("encoded code page holds the wrong number of rows"));
9866 }
9867 let seen = wide.iter().fold(0_i64, |seen, &code| seen | code);
9874 if seen < 0 || seen > i64::from(u32::MAX) {
9875 return Err(invalid("code is not a code"));
9876 }
9877 wide.iter().map(|&code| code as u32).collect()
9878 } else {
9879 let mut codes = Vec::with_capacity(rows);
9880 for _ in 0..rows {
9881 codes.push(cur.u32()?);
9882 }
9883 if cur.at != bytes.len() {
9884 return Err(invalid("global code page has trailing bytes"));
9885 }
9886 codes
9887 };
9888 let highest = codes.iter().copied().max();
9889 return Ok(Vector::stable_dictionary_validated(codes, dictionary, highest)?
9890 .with_validity(validity));
9891 }
9892 if codec == 6 {
9893 if ty != &LogicalType::Varchar {
9894 return Err(invalid("compressed text codec belongs to a non-string page"));
9895 }
9896 let (payload, ends) = string::decode_flat(&bytes[cur.at..])?.into_parts();
9900 if ends.len() != rows {
9901 return Err(invalid("compressed text page holds the wrong number of rows"));
9902 }
9903 let mut values = StringColumn::over(Buffer::from_vec(payload).into_page());
9906 let mut start = 0;
9907 for end in ends {
9908 let len = end
9909 .checked_sub(start)
9910 .ok_or_else(|| invalid("compressed text value ends before it starts"))?;
9911 values.push_in_place(start, len)?;
9912 start = end;
9913 }
9914 return Ok(Vector::flat(ty.clone(), Data::Varlen(values))?.with_validity(validity));
9915 }
9916 if codec == 5 {
9917 let values = integer::decode(&bytes[cur.at..])?;
9919 if values.len() != rows {
9920 return Err(invalid("cascade page holds the wrong number of rows"));
9921 }
9922 let data = narrowed(ty, values)?;
9923 return Ok(Vector::flat(ty.clone(), data)?.with_validity(validity));
9924 }
9925 if codec == 2 {
9926 let width = u32::from(cur.u8()?);
9927 let base = i128::from_le_bytes(cur.take(16)?.try_into().expect("sixteen bytes"));
9928 let count = cur.u32()? as usize;
9929 let length = count.checked_mul(8).ok_or_else(|| invalid("packed page is too long"))?;
9930 let words: Vec<u64> = cur
9931 .take(length)?
9932 .chunks_exact(8)
9933 .map(|word| u64::from_le_bytes(word.try_into().expect("eight bytes")))
9934 .collect();
9935 if cur.at != bytes.len() {
9936 return Err(invalid("packed page has trailing bytes"));
9937 }
9938 return Ok(Vector::packed(ty.clone(), words, width, base, rows)?.with_validity(validity));
9939 }
9940 if codec != 0 {
9941 return Err(invalid("page codec is unknown"));
9942 }
9943 let data = match ty {
9944 LogicalType::TinyInt => {
9945 let values = cur.take(rows)?;
9946 Data::Int8(values.iter().map(|item| *item as i8).collect::<Vec<_>>().into())
9947 }
9948 LogicalType::UTinyInt => Data::UInt8(cur.take(rows)?.to_vec().into()),
9949 LogicalType::SmallInt => {
9950 let values =
9951 cur.take(rows.checked_mul(2).ok_or_else(|| invalid("page size overflow"))?)?;
9952 Data::Int16(
9953 values
9954 .chunks_exact(2)
9955 .map(|item| i16::from_le_bytes(item.try_into().expect("two bytes")))
9956 .collect::<Vec<_>>()
9957 .into(),
9958 )
9959 }
9960 LogicalType::USmallInt => {
9961 let values =
9962 cur.take(rows.checked_mul(2).ok_or_else(|| invalid("page size overflow"))?)?;
9963 Data::UInt16(
9964 values
9965 .chunks_exact(2)
9966 .map(|item| u16::from_le_bytes(item.try_into().expect("two bytes")))
9967 .collect::<Vec<_>>()
9968 .into(),
9969 )
9970 }
9971 LogicalType::UInteger => {
9972 let values =
9973 cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
9974 Data::UInt32(
9975 values
9976 .chunks_exact(4)
9977 .map(|item| u32::from_le_bytes(item.try_into().expect("four bytes")))
9978 .collect::<Vec<_>>()
9979 .into(),
9980 )
9981 }
9982 LogicalType::UBigInt => {
9983 let values =
9984 cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
9985 Data::UInt64(
9986 values
9987 .chunks_exact(8)
9988 .map(|item| u64::from_le_bytes(item.try_into().expect("eight bytes")))
9989 .collect::<Vec<_>>()
9990 .into(),
9991 )
9992 }
9993 LogicalType::Integer | LogicalType::Date => {
9994 let values =
9995 cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
9996 Data::Int32(
9997 values
9998 .chunks_exact(4)
9999 .map(|item| i32::from_le_bytes(item.try_into().expect("four bytes")))
10000 .collect::<Vec<_>>()
10001 .into(),
10002 )
10003 }
10004 LogicalType::BigInt
10005 | LogicalType::Timestamp
10006 | LogicalType::Time
10007 | LogicalType::TimeTz
10008 | LogicalType::TimestampTz
10009 | LogicalType::TimestampS
10010 | LogicalType::TimestampMs
10011 | LogicalType::TimestampNs => {
10012 let values =
10013 cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
10014 Data::Int64(
10015 values
10016 .chunks_exact(8)
10017 .map(|item| i64::from_le_bytes(item.try_into().expect("eight bytes")))
10018 .collect::<Vec<_>>()
10019 .into(),
10020 )
10021 }
10022 LogicalType::HugeInt | LogicalType::Uuid => {
10023 let values =
10024 cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
10025 Data::Int128(
10026 values
10027 .chunks_exact(16)
10028 .map(|item| i128::from_le_bytes(item.try_into().expect("sixteen bytes")))
10029 .collect::<Vec<_>>()
10030 .into(),
10031 )
10032 }
10033 LogicalType::UHugeInt => {
10034 let values =
10035 cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
10036 Data::UInt128(
10037 values
10038 .chunks_exact(16)
10039 .map(|item| u128::from_le_bytes(item.try_into().expect("sixteen bytes")))
10040 .collect::<Vec<_>>()
10041 .into(),
10042 )
10043 }
10044 LogicalType::Float => {
10045 let values =
10046 cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
10047 Data::Float32(
10048 values
10049 .chunks_exact(4)
10050 .map(|item| f32::from_le_bytes(item.try_into().expect("four bytes")))
10051 .collect::<Vec<_>>()
10052 .into(),
10053 )
10054 }
10055 LogicalType::Double => {
10056 let values =
10057 cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
10058 Data::Float64(
10059 values
10060 .chunks_exact(8)
10061 .map(|item| f64::from_le_bytes(item.try_into().expect("eight bytes")))
10062 .collect::<Vec<_>>()
10063 .into(),
10064 )
10065 }
10066 LogicalType::Interval => {
10067 let values =
10068 cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
10069 Data::Interval(
10070 values
10071 .chunks_exact(16)
10072 .map(|item| {
10073 (
10074 i32::from_le_bytes(item[..4].try_into().expect("four bytes")),
10075 i32::from_le_bytes(item[4..8].try_into().expect("four bytes")),
10076 i64::from_le_bytes(item[8..].try_into().expect("eight bytes")),
10077 )
10078 })
10079 .collect::<Vec<_>>()
10080 .into(),
10081 )
10082 }
10083 LogicalType::Boolean => {
10084 let values = cur.take(rows)?;
10085 if values.iter().any(|value| *value > 1) {
10086 return Err(invalid("boolean page has another value"));
10087 }
10088 Data::Bool(values.iter().map(|value| *value == 1).collect::<Vec<_>>().into())
10089 }
10090 LogicalType::Decimal { .. } => match ty.physical() {
10093 PhysicalType::Int16 => {
10094 let values =
10095 cur.take(rows.checked_mul(2).ok_or_else(|| invalid("page size overflow"))?)?;
10096 Data::Int16(
10097 values
10098 .chunks_exact(2)
10099 .map(|item| i16::from_le_bytes(item.try_into().expect("two bytes")))
10100 .collect::<Vec<_>>()
10101 .into(),
10102 )
10103 }
10104 PhysicalType::Int32 => {
10105 let values =
10106 cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
10107 Data::Int32(
10108 values
10109 .chunks_exact(4)
10110 .map(|item| i32::from_le_bytes(item.try_into().expect("four bytes")))
10111 .collect::<Vec<_>>()
10112 .into(),
10113 )
10114 }
10115 PhysicalType::Int64 => {
10116 let values =
10117 cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
10118 Data::Int64(
10119 values
10120 .chunks_exact(8)
10121 .map(|item| i64::from_le_bytes(item.try_into().expect("eight bytes")))
10122 .collect::<Vec<_>>()
10123 .into(),
10124 )
10125 }
10126 _ => {
10127 let values =
10128 cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
10129 Data::Int128(
10130 values
10131 .chunks_exact(16)
10132 .map(|item| i128::from_le_bytes(item.try_into().expect("sixteen bytes")))
10133 .collect::<Vec<_>>()
10134 .into(),
10135 )
10136 }
10137 },
10138 LogicalType::Varchar | LogicalType::Blob | LogicalType::Bit => {
10139 let offset_bytes = cur
10140 .take((rows + 1).checked_mul(4).ok_or_else(|| invalid("offset count overflow"))?)?;
10141 let offsets = offset_bytes
10142 .chunks_exact(4)
10143 .map(|part| u32::from_le_bytes(part.try_into().expect("four bytes")))
10144 .collect::<Vec<_>>();
10145 let payload = cur.take(bytes.len() - cur.at)?.to_vec();
10146 if offsets.first() != Some(&0)
10147 || offsets.last().copied().map(|last| last as usize) != Some(payload.len())
10148 || offsets.windows(2).any(|pair| pair[0] > pair[1])
10149 {
10150 return Err(invalid("string offsets do not bound the payload"));
10151 }
10152 let mut values = StringColumn::over(Buffer::from_vec(payload).into_page());
10160 let text = ty == &LogicalType::Varchar;
10161 for pair in offsets.windows(2) {
10162 let (at, len) = (pair[0] as usize, (pair[1] - pair[0]) as usize);
10163 if text {
10164 values.push_in_place(at, len)?;
10165 } else {
10166 values.push_bytes_in_place(at, len)?;
10167 }
10168 }
10169 Data::Varlen(values)
10170 }
10171 _ => return Err(Error::not_implemented(format!("native page for {ty}"))),
10172 };
10173 if cur.at != bytes.len() {
10174 return Err(invalid("page has trailing bytes"));
10175 }
10176 Ok(Vector::flat(ty.clone(), data)?.with_validity(validity))
10177}
10178
10179#[cfg(test)]
10180mod tests {
10181 use std::fs;
10182 use std::io::{Seek, SeekFrom, Write};
10183 use std::path::PathBuf;
10184 use std::time::{SystemTime, UNIX_EPOCH};
10185
10186 use rudb_common::Stat;
10187 use rudb_common::Value;
10188 use rudb_common::bounds::{Frequencies, Op, Remainder, Zones};
10189 use rudb_common::stat::Provenance;
10190
10191 use super::*;
10192
10193 #[test]
10194 fn a_name_taken_in_pieces_is_the_name_of_the_pieces_joined() {
10195 let bytes: Vec<u8> =
10196 (0..300_u32).map(|at| (at.wrapping_mul(2_654_435_761) >> 13) as u8).collect();
10197 for length in [0, 1, 7, 31, 32, 33, 63, 64, 65, 100, 300] {
10198 let whole = content_name(&bytes[..length]);
10199 for step in [1, 3, 8, 31, 32, 33, 64, 301] {
10200 let mut namer = ContentNamer::default();
10201 bytes[..length].chunks(step).for_each(|piece| namer.update(piece));
10202 assert_eq!(namer.finish(), whole, "{length} bytes in pieces of {step}");
10203 }
10204 }
10205 }
10206
10207 #[derive(Debug)]
10210 struct TestsEverything<'a>(&'a dyn chooser::Chooser);
10211
10212 impl chooser::Chooser for TestsEverything<'_> {
10213 fn name(&self) -> &'static str {
10214 "tests everything"
10215 }
10216
10217 fn narrow_strings(
10218 &self,
10219 values: &[&[u8]],
10220 offered: &[string::Kind],
10221 depth: u8,
10222 ) -> Vec<string::Kind> {
10223 self.0.narrow_strings(values, offered, depth)
10224 }
10225
10226 fn narrow_integers(
10227 &self,
10228 values: &[i64],
10229 offered: &[integer::Kind],
10230 depth: u8,
10231 ) -> Vec<integer::Kind> {
10232 self.0.narrow_integers(values, offered, depth)
10233 }
10234 }
10235
10236 #[test]
10237 fn ruling_kinds_out_before_testing_for_them_writes_the_same_bytes() {
10238 let columns: Vec<Vec<i64>> = vec![
10239 vec![],
10240 vec![5; 1000],
10241 (0..1000).collect(),
10242 (0..1000).map(|row| 1_600_000_000_000_000 + row * 1_000_000).collect(),
10243 (0..1000).map(|row| row / 50).collect(),
10244 (0..1000).map(|row| if row % 97 == 0 { row } else { 0 }).collect(),
10245 (0..1000).map(|row| (row * 7919) % 13).collect(),
10246 (0..1000).map(|row| (row * 2_654_435_761) % 1_000_003).collect(),
10247 (0..1000).map(|row| [3, 3, 3, 9, 9, 1][row as usize % 6]).collect(),
10248 (0..1000).map(|row| i64::MIN + row % 3).collect(),
10249 ];
10250 let choosers: [&dyn chooser::Chooser; 2] = [&Fixed, &Codes];
10251 for column in &columns {
10252 for chooser in choosers {
10253 let quick = integer::encode_with(column, chooser).unwrap();
10254 let full = integer::encode_with(column, &TestsEverything(chooser)).unwrap();
10255 assert_eq!(
10256 quick,
10257 full,
10258 "{} on {:?}",
10259 chooser.name(),
10260 &column[..column.len().min(8)]
10261 );
10262 }
10263 }
10264 }
10265
10266 #[test]
10269 fn parts_that_look_alike_replay_to_the_bytes_a_search_writes() {
10270 let mut settling = Settling::default();
10271 for part in 0..STRIPE_PARTS as i64 {
10272 let values: Vec<i64> = (0..2048)
10273 .map(|row| 1_600_000_000_000_000 + (part * 2048 + row) * 1_000_000 + row % 7)
10274 .collect();
10275 let searched = integer::encode_with(&values, &Fixed).unwrap();
10276 assert_eq!(settling.encode(&values).unwrap(), searched, "part {part}");
10277 }
10278 }
10279
10280 #[test]
10284 fn a_column_that_changes_under_the_shape_is_searched_again() {
10285 let mut state = 0x9e37_79b9_7f4a_7c15_u64;
10286 let mut noise = move || {
10287 state ^= state << 13;
10288 state ^= state >> 7;
10289 state ^= state << 17;
10290 (state % 1_000_000) as i64
10291 };
10292 let mut settling = Settling::default();
10293 for part in 0..STRIPE_PARTS as i64 {
10294 let values: Vec<i64> = match part / 16 {
10295 0 => (0..2048).map(|row| (part * 2048 + row) / 300).collect(),
10296 1 => (0..2048).map(|_| noise()).collect(),
10297 2 => (0..2048).map(|row| if row % 97 == 0 { row } else { 42 }).collect(),
10298 _ => (0..2048).map(|row| 5 + (part * 2048 + row) * 1_000_000).collect(),
10299 };
10300 let settled = settling.encode(&values).unwrap();
10301 assert_eq!(integer::decode(&settled).unwrap(), values, "part {part}");
10302 let searched = integer::encode_with(&values, &Fixed).unwrap();
10303 assert!(
10304 settled.len() * 4 <= searched.len() * 5,
10305 "part {part}: {} settled against {} searched, {} against {}",
10306 settled.len(),
10307 searched.len(),
10308 integer::describe(&settled).unwrap(),
10309 integer::describe(&searched).unwrap(),
10310 );
10311 }
10312 }
10313
10314 #[test]
10315 fn checksum_matches_fixed_vectors() {
10316 assert_eq!(checksum(b""), 0xef46_db37_51d8_e999);
10317 assert_eq!(checksum(b"a"), 0xd24e_c4f1_a98c_6e5b);
10318 assert_eq!(checksum(b"abc"), 0x44bc_2cf5_ad77_0999);
10319 }
10320
10321 #[test]
10322 fn sorting_across_threads_matches_sorting_on_one() {
10323 let mut state = 0x9e37_79b9_7f4a_7c15_u64;
10324 let mut next = move || {
10325 state ^= state << 13;
10326 state ^= state >> 7;
10327 state ^= state << 17;
10328 state
10329 };
10330 let mut values = Vec::new();
10331 for at in 0..150_000_u64 {
10332 let value = match next() % 6 {
10333 0 => Vec::new(),
10334 1 => format!("https://example.com/{}", next() % 5_000).into_bytes(),
10335 2 => format!("https://example.com/path/{at}").into_bytes(),
10336 3 => b"same".to_vec(),
10337 4 => vec![0xff; (next() % 12) as usize],
10338 _ => (0..next() % 20).map(|_| (next() % 3) as u8).collect(),
10339 };
10340 values.push(value);
10341 }
10342 let value = |code: u32| values[code as usize].as_slice();
10343 for workers in [1, 2, 3, 8, 32] {
10344 let mut one = (0..values.len() as u32).rev().collect::<Vec<_>>();
10345 let mut across = one.clone();
10346 sort_by_value(&mut one, value);
10347 sort_by_value_across(&mut across, value, workers);
10348 assert_eq!(one, across, "{workers} workers");
10349 }
10350 let mut sorted = (0..values.len() as u32).collect::<Vec<_>>();
10351 sort_by_value_across(&mut sorted, value, 8);
10352 assert!(sorted.windows(2).all(|pair| value(pair[0]) <= value(pair[1])));
10353 }
10354
10355 fn path(label: &str) -> PathBuf {
10356 let stamp = SystemTime::now().duration_since(UNIX_EPOCH).expect("time advances").as_nanos();
10357 std::env::temp_dir().join(format!("rudb-native-{label}-{}-{stamp}.rdb", std::process::id()))
10358 }
10359
10360 fn dictionary_values(dictionary: &GlobalDictionary) -> Vec<Vec<u8>> {
10365 let (flat, bases) = dictionary.decoded(None).expect("the blocks decode");
10366 (0..dictionary.values())
10367 .map(|code| {
10368 let (from, to) = GlobalDictionary::value_span(&dictionary.ends, &bases, code);
10369 flat[from..to].to_vec()
10370 })
10371 .collect()
10372 }
10373
10374 fn attached(table: &Table) -> Vec<&Section> {
10381 table.sections().iter().filter(|held| !held.among(section::STATISTICS_KINDS)).collect()
10382 }
10383
10384 #[test]
10386 fn a_read_at_an_offset_ignores_where_another_thread_left_the_cursor() {
10387 const SPANS: usize = 64;
10388 const SPAN: usize = 512;
10389 let path = path("positional");
10390 let content: Vec<u8> =
10391 (0..SPANS).flat_map(|span| std::iter::repeat_n(span as u8, SPAN)).collect();
10392 fs::write(&path, &content).expect("the file is written");
10393 let file = Arc::new(File::open(&path).expect("the file opens"));
10394 std::thread::scope(|scope| {
10395 for _ in 0..8 {
10396 let file = Arc::clone(&file);
10397 scope.spawn(move || {
10398 for _ in 0..64 {
10399 for span in 0..SPANS {
10400 let mut bytes = [0_u8; SPAN];
10401 read_at(&file, (span * SPAN) as u64, &mut bytes)
10402 .expect("the span reads");
10403 assert!(
10404 bytes.iter().all(|byte| *byte == span as u8),
10405 "span {span} came back as {}",
10406 bytes[0],
10407 );
10408 }
10409 }
10410 });
10411 }
10412 });
10413 let mut past = [0_u8; SPAN];
10414 let end = (SPANS * SPAN) as u64;
10415 let error = read_at(&file, end, &mut past).expect_err("a read past the end is refused");
10416 assert!(error.message().contains("ends before its declared length"), "{error}");
10417 drop(file);
10418 let _ = fs::remove_file(&path);
10419 }
10420
10421 #[test]
10427 fn a_writer_puts_a_page_where_it_said_it_did_wherever_the_cursor_has_got_to() {
10428 let path = path("cursor");
10429 let mut writer = Writer::create(
10430 &path,
10431 "items",
10432 vec![
10433 Field::required("id", LogicalType::Integer),
10434 Field::new("text", LogicalType::Varchar),
10435 ],
10436 )
10437 .expect("new file");
10438 writer.append(&sample()).expect("first part");
10439 writer.file.seek(SeekFrom::Start(0)).expect("the cursor goes back to the header");
10440 writer.append(&sample()).expect("second part");
10441 writer.file.seek(SeekFrom::Start(1)).expect("and somewhere useless again");
10442 writer.finish().expect("commit");
10443 let reader = Reader::open(&path).expect("reopen from disk");
10444 assert_eq!(reader.table().rows(), 6);
10445 let ids = reader.read(0, &[0]).expect("the integer page reads back");
10446 assert_eq!(ids.value_at(0, 0), Value::Integer(4));
10447 assert_eq!(ids.value_at(2, 0), Value::Integer(-2));
10448 let text = reader.read(1, &[1]).expect("the text page reads back");
10449 assert_eq!(text.value_at(1, 0), Value::Null);
10450 assert_eq!(text.value_at(2, 0), Value::Varchar("long text after a slash".into()));
10451 let end = reader.table().stripes().iter().flat_map(|stripe| {
10454 stripe
10455 .pages
10456 .iter()
10457 .map(|page| page.offset + u64::from(page.length))
10458 .chain(std::iter::once(stripe.index.offset + u64::from(stripe.index.length)))
10459 });
10460 let last = end.fold(HEADER, u64::max);
10461 let directory = fs::metadata(&path).expect("the file is there").len();
10462 assert!(last <= directory, "a page runs to {last} in a file of {directory} bytes");
10463 fs::remove_file(path).expect("remove scratch file");
10464 }
10465
10466 fn dictionary_index_len(header: &[u8; DICTIONARY_HEADER]) -> u64 {
10472 let count = u64::from(u32::from_le_bytes(header[0..4].try_into().expect("four bytes")));
10473 let blocks = u64::from(u32::from_le_bytes(header[8..12].try_into().expect("four bytes")));
10474 let width = u32::from_le_bytes(header[12..16].try_into().expect("four bytes"));
10475 let bits = (width & !(DICTIONARY_SCATTERED | DICTIONARY_GRAMS)) as usize;
10476 let payload_words = if width & DICTIONARY_SCATTERED == 0 { 2 } else { 3 };
10477 let rank_blocks = count.div_ceil(TEXT_RANK_BLOCK as u64);
10478 DICTIONARY_HEADER as u64
10479 + offset_bytes(count as usize, bits) as u64
10480 + blocks * payload_words * 8
10481 + rank_blocks * 16
10482 + if width & DICTIONARY_GRAMS == 0 { 0 } else { 8 }
10483 }
10484
10485 fn sample() -> Chunk {
10486 Chunk::new(vec![
10487 Vector::from_values(
10488 LogicalType::Integer,
10489 &[Value::Integer(4), Value::Integer(9), Value::Integer(-2)],
10490 )
10491 .expect("integers"),
10492 Vector::from_values(
10493 LogicalType::Varchar,
10494 &[
10495 Value::Varchar("alpha".into()),
10496 Value::Null,
10497 Value::Varchar("long text after a slash".into()),
10498 ],
10499 )
10500 .expect("strings"),
10501 ])
10502 .expect("matching rows")
10503 }
10504
10505 fn sample_ids() -> Chunk {
10506 Chunk::new(vec![
10507 Vector::flat(LogicalType::Integer, Data::Int32(vec![7, 8, 9].into()))
10508 .expect("integers"),
10509 ])
10510 .expect("one column")
10511 }
10512
10513 #[test]
10514 fn the_planner_gets_the_null_count_off_the_same_directory_the_bounds_are_in() {
10515 let path = path("nulls_for_the_planner");
10518 let mut writer =
10519 Writer::create(&path, "items", vec![Field::new("a", LogicalType::Integer)])
10520 .expect("new file");
10521 let rows = Chunk::new(vec![
10522 Vector::from_values(
10523 LogicalType::Integer,
10524 &[
10525 Value::Integer(4),
10526 Value::Null,
10527 Value::Integer(9),
10528 Value::Null,
10529 Value::Integer(1),
10530 Value::Integer(2),
10531 ],
10532 )
10533 .expect("integers"),
10534 ])
10535 .expect("one column");
10536 writer.append(&rows).expect("the only part");
10537 writer.finish().expect("commit");
10538 let reader = Reader::open(&path).expect("reopen from disk");
10539 let stripes = Stripes::new(reader);
10540 let column = stripes.column("a").expect("the file has that column");
10541 assert_eq!(stripes.nulls(column), Stat::exact(2, Provenance::NullCount));
10542 assert_eq!(stripes.nulls(column + 1), Stat::Unknown);
10545 fs::remove_file(&path).expect("clean up");
10546 }
10547
10548 #[test]
10549 fn the_planner_gets_a_row_count_per_value_off_a_complete_synopsis() {
10550 let path = path("frequencies_for_the_planner");
10555 let mut writer =
10556 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
10557 .expect("new file");
10558 let rows = Chunk::new(vec![
10559 Vector::from_values(
10560 LogicalType::Integer,
10561 &[
10562 Value::Integer(4),
10563 Value::Integer(4),
10564 Value::Integer(4),
10565 Value::Integer(9),
10566 Value::Integer(9),
10567 Value::Integer(1),
10568 ],
10569 )
10570 .expect("integers"),
10571 ])
10572 .expect("one column");
10573 writer.append(&rows).expect("the only part");
10574 writer.finish().expect("commit");
10575 let reader = Reader::open(&path).expect("reopen from disk");
10576 let common = Common::new(reader);
10577 assert_eq!(common.rows(), 6);
10578 let column = common.column("id").expect("the file has that column");
10579 assert_eq!(common.column("nothing"), None);
10580 assert_eq!(
10581 common.rows_with(column, &Bound::Int(4)),
10582 Stat::exact(3, Provenance::FrequencySynopsis)
10583 );
10584 assert_eq!(
10586 common.rows_with(column, &Bound::Int(7)),
10587 Stat::exact(0, Provenance::FrequencySynopsis)
10588 );
10589 assert_eq!(common.rows_with(column, &Bound::Bytes(b"four".to_vec())), Stat::Unknown);
10592 assert_eq!(common.remainder(column), None);
10595 fs::remove_file(&path).expect("clean up");
10596 }
10597
10598 #[test]
10599 fn string_frequency_estimates_do_not_open_the_global_dictionary() {
10600 let path = path("string_frequencies_for_the_planner");
10601 let mut writer =
10602 Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
10603 .expect("new file");
10604 let rows = Chunk::new(vec![
10605 Vector::from_values(
10606 LogicalType::Varchar,
10607 &[
10608 Value::Varchar(String::new()),
10609 Value::Varchar("alpha".into()),
10610 Value::Varchar(String::new()),
10611 Value::Varchar("beta".into()),
10612 Value::Varchar(String::new()),
10613 ],
10614 )
10615 .expect("strings"),
10616 ])
10617 .expect("one column");
10618 writer.append(&rows).expect("the only part");
10619 writer.finish().expect("commit");
10620
10621 let reader = Reader::open(&path).expect("reopen from disk");
10622 assert_eq!(reader.reads().dictionaries, 0, "open reads only the directory");
10623 let common = Common::new(reader.clone());
10624 let column = common.column("text").expect("the file has that column");
10625 assert_eq!(
10626 common.rows_with(column, &Bound::Bytes(Vec::new())),
10627 Stat::exact(3, Provenance::FrequencySynopsis)
10628 );
10629 assert_eq!(
10630 common.rows_with(column, &Bound::Bytes(b"missing".to_vec())),
10631 Stat::exact(0, Provenance::FrequencySynopsis)
10632 );
10633 assert_eq!(
10634 reader.reads().dictionaries,
10635 0,
10636 "the bounded spellings answer without opening the dictionary index"
10637 );
10638 fs::remove_file(&path).expect("clean up");
10639 }
10640
10641 #[test]
10642 fn host_groups_certify_omitted_hosts_and_keep_exact_aggregates() {
10643 let path = path("certified_host_groups");
10644 let mut writer =
10645 Writer::create(&path, "hits", vec![Field::required("Referer", LogicalType::Varchar)])
10646 .expect("new file");
10647 let mut values = vec![Value::Varchar("http://www.example.com/a".into()); 150];
10648 values.extend(vec![Value::Varchar("https://example.com/b".into()); 70]);
10649 values.extend((0..550).map(|at| Value::Varchar(format!("https://site{at}.test/x"))));
10650 values.push(Value::Varchar(String::new()));
10651 for part in values.chunks(512) {
10652 writer
10653 .append(
10654 &Chunk::new(vec![
10655 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
10656 ])
10657 .expect("one column"),
10658 )
10659 .expect("part written");
10660 }
10661 writer.finish().expect("commit");
10662 let reader = Reader::open(&path).expect("reopen");
10663 let summary = reader.table.host_groups.as_ref().expect("bounded host metadata");
10664 assert!(summary.omitted_max < 220);
10665 assert!(reader.host_groups(0, summary.omitted_max).expect("valid column").is_none());
10666 let groups = reader.host_groups(0, 220).expect("valid column").expect("certified");
10667 let example = groups.iter().find(|entry| entry.host == "example.com").expect("leader");
10668 assert_eq!(example.count, 220);
10669 assert_eq!(example.bytes_sum, 150 * 24 + 70 * 21);
10670 assert_eq!(example.minimum, "http://www.example.com/a");
10671 assert_eq!(reader.reads().dictionaries, 0, "the directory settles the question");
10672 fs::remove_file(&path).expect("clean up");
10673 }
10674
10675 fn bare_table(sections: Vec<Section>) -> Table {
10680 Table {
10681 name: "linked".to_owned(),
10682 fields: vec![Field::required("id", LogicalType::Integer)],
10683 stripes: Vec::new(),
10684 rows: 0,
10685 dictionaries: vec![None],
10686 dictionary_payloads: Vec::new(),
10687 distincts: vec![None],
10688 frequencies: vec![None],
10689 pair_frequencies: Vec::new(),
10690 frequency_texts: Vec::new(),
10691 host_groups: None,
10692 clustering: None,
10693 generation: 1,
10694 sections,
10695 }
10696 }
10697
10698 fn a_key_map_section() -> Section {
10699 Section {
10700 kind: *section::KEY_MAP,
10701 id: 1,
10702 generation: 3,
10703 extents: 1,
10704 extent_page: HEADER,
10705 extent_bytes: section::EXTENT_BYTES as u32,
10706 hash: 0x1234_5678_9abc_def0,
10707 flags: 0,
10708 header_bytes: 24,
10709 }
10710 }
10711
10712 #[test]
10713 fn a_section_table_round_trips_through_a_directory() {
10714 let mut later = a_key_map_section();
10715 later.kind = *b"RUDBZZ9\0";
10716 later.id = 2;
10717 let table = bare_table(vec![a_key_map_section(), later]);
10718 let directory = encode_directory(&table).expect("directory");
10719 let decoded = decode_directory(&directory, 1 << 20).expect("reopen");
10720 assert_eq!(decoded.sections(), &[a_key_map_section(), later]);
10721 assert!(decoded.sections()[0].known());
10725 assert!(!decoded.sections()[1].known());
10726 }
10727
10728 #[test]
10729 fn a_directory_written_before_the_section_table_reads_as_a_table_with_none() {
10730 let directory = encode_directory(&bare_table(Vec::new())).expect("directory");
10734 let block = SECTIONS.len() + size_of::<u64>() + size_of::<u16>();
10735 let older = &directory[..directory.len() - block];
10736 let decoded = decode_directory(older, 1 << 20).expect("a directory from before sections");
10737 assert!(decoded.sections().is_empty());
10738 assert_eq!(decoded.generation(), 0, "a format 22 table recorded no generation");
10739 assert_eq!(decoded.name(), "linked");
10740 assert_eq!(decoded.fields().len(), 1, "everything before the block still decodes");
10741 }
10742
10743 #[test]
10744 fn a_file_stamped_with_the_previous_format_still_opens_and_reads() {
10745 let path = path("format_twenty_two");
10752 let mut writer =
10753 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
10754 .expect("new file");
10755 let rows = Chunk::new(vec![
10756 Vector::from_values(
10757 LogicalType::Integer,
10758 &[Value::Integer(1), Value::Integer(2), Value::Integer(3)],
10759 )
10760 .expect("integers"),
10761 ])
10762 .expect("one column");
10763 writer.append(&rows).expect("the only part");
10764 writer.finish().expect("commit");
10765
10766 let file = OpenOptions::new().write(true).open(&path).expect("reopen to patch");
10767 write_at(&file, 8, &22_u32.to_le_bytes()).expect("stamp the older format");
10768 drop(file);
10769
10770 let reader = Reader::open(&path).expect("a format 22 file opens unchanged");
10771 assert_eq!(reader.table().rows(), 3);
10772 assert_eq!(reader.read(0, &[0]).expect("the part still reads").len(), 3);
10777
10778 let file = OpenOptions::new().write(true).open(&path).expect("reopen to patch");
10781 write_at(&file, 8, &21_u32.to_le_bytes()).expect("stamp an unreadable format");
10782 drop(file);
10783 let error = Reader::open(&path).expect_err("format 21 is not readable");
10784 assert!(error.to_string().contains("format 21"), "{error}");
10785
10786 fs::remove_file(&path).expect("clean up");
10787 }
10788
10789 #[test]
10790 fn a_section_whose_extent_table_is_outside_the_file_is_refused() {
10791 let mut past = a_key_map_section();
10796 past.extent_page = 1 << 30;
10797 let directory = encode_directory(&bare_table(vec![past])).expect("directory");
10798 let error = decode_directory(&directory, 1 << 20).expect_err("refused");
10799 assert!(error.to_string().contains("outside the file"), "{error}");
10800
10801 let mut inside_the_header = a_key_map_section();
10802 inside_the_header.extent_page = 8;
10803 let directory = encode_directory(&bare_table(vec![inside_the_header])).expect("directory");
10804 assert!(
10805 decode_directory(&directory, 1 << 20).is_err(),
10806 "a section may not overlap a header"
10807 );
10808 }
10809
10810 #[test]
10811 fn a_section_recorded_as_not_built_is_legal_and_names_no_bytes() {
10812 let not_built = Section {
10816 kind: *section::FORWARD_LINK,
10817 id: 9,
10818 generation: 3,
10819 extents: 0,
10820 extent_page: 0,
10821 extent_bytes: 0,
10822 hash: 0,
10823 flags: 0,
10824 header_bytes: 0,
10825 };
10826 let directory = encode_directory(&bare_table(vec![not_built])).expect("directory");
10827 let decoded = decode_directory(&directory, 1 << 20).expect("reopen");
10828 assert_eq!(decoded.sections(), &[not_built]);
10829
10830 let mut incoherent = not_built;
10833 incoherent.extent_bytes = 28;
10834 incoherent.extent_page = HEADER;
10835 let directory = encode_directory(&bare_table(vec![incoherent])).expect("directory");
10836 assert!(decode_directory(&directory, 1 << 20).is_err());
10837 }
10838
10839 #[test]
10840 fn a_directory_naming_more_sections_than_the_bound_is_refused() {
10841 let directory = encode_directory(&bare_table(Vec::new())).expect("directory");
10842 let mut torn = directory.clone();
10843 let count_at = torn.len() - size_of::<u16>();
10844 torn[count_at..].copy_from_slice(&u16::MAX.to_le_bytes());
10845 assert!(decode_directory(&torn, 1 << 20).is_err());
10848 }
10849
10850 fn linked_file(label: &str, rows: i32) -> PathBuf {
10852 let path = path(label);
10853 let mut writer =
10854 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
10855 .expect("new file");
10856 let values = (0..rows).map(Value::Integer).collect::<Vec<_>>();
10857 let chunk =
10858 Chunk::new(vec![Vector::from_values(LogicalType::Integer, &values).expect("integers")])
10859 .expect("one column");
10860 writer.append(&chunk).expect("the only part");
10861 writer.finish().expect("commit");
10862 path
10863 }
10864
10865 fn a_key_map_payload() -> Vec<u8> {
10866 (0..512_u32).flat_map(u32::to_le_bytes).collect()
10869 }
10870
10871 #[test]
10872 fn a_section_attached_to_a_committed_file_reads_back_byte_for_byte() {
10873 let path = linked_file("attach", 64);
10874 let payload = a_key_map_payload();
10875 let table = attach(
10876 &path,
10877 "items",
10878 &[section::Attachment {
10879 kind: *section::KEY_MAP,
10880 id: 0,
10881 flags: 2,
10882 header_bytes: 40,
10883 bytes: &payload,
10884 }],
10885 )
10886 .expect("attach a key map");
10887 assert_eq!(attached(&table).len(), 1);
10888
10889 let reader = Reader::open(&path).expect("reopen after the attach");
10890 let held = attached(reader.table());
10891 assert_eq!(held.len(), 1);
10892 assert_eq!(held[0].kind, *section::KEY_MAP);
10893 assert_eq!(held[0].flags, 2, "the form a reader must not have to guess");
10894 assert_eq!(held[0].header_bytes, 40);
10895 assert_eq!(held[0].generation, 1);
10899 assert!(held[0].usable(reader.table().generation()));
10900 assert_eq!(reader.payload(held[0]).expect("read the payload"), payload);
10901 assert_eq!(reader.extents(held[0]).expect("extent table").len(), 1);
10902
10903 fs::remove_file(&path).expect("clean up");
10904 }
10905
10906 #[test]
10907 fn attaching_a_section_answers_every_row_exactly_as_before() {
10908 let path = linked_file("attach_changes_nothing", 300);
10913 let before = Reader::open(&path).expect("open before");
10914 let rows = before.table().rows();
10915 let first = before.read(0, &[0]).expect("read before");
10916 let values = (0..rows).map(|at| first.value_at(at, 0)).collect::<Vec<_>>();
10917 let layout = before.layout().columns_total();
10918 drop(before);
10919
10920 let payload = a_key_map_payload();
10921 attach(
10922 &path,
10923 "items",
10924 &[section::Attachment {
10925 kind: *section::KEY_MAP,
10926 id: 0,
10927 flags: 0,
10928 header_bytes: 0,
10929 bytes: &payload,
10930 }],
10931 )
10932 .expect("attach");
10933
10934 let after = Reader::open(&path).expect("open after");
10935 assert_eq!(after.table().rows(), rows);
10936 let read = after.read(0, &[0]).expect("read after");
10937 for (at, value) in values.iter().enumerate() {
10938 assert_eq!(&read.value_at(at, 0), value, "row {at} moved");
10939 }
10940 assert_eq!(
10941 after.layout().columns_total(),
10942 layout,
10943 "an attach appends and does not rewrite a column page"
10944 );
10945
10946 fs::remove_file(&path).expect("clean up");
10947 }
10948
10949 #[test]
10950 fn a_rebuilt_section_replaces_the_one_it_supersedes() {
10951 let path = linked_file("attach_twice", 32);
10955 let one = a_key_map_payload();
10956 let two = vec![7_u8; 1024];
10957 let entry = |bytes| section::Attachment {
10958 kind: *section::KEY_MAP,
10959 id: 4,
10960 flags: 1,
10961 header_bytes: 0,
10962 bytes,
10963 };
10964 attach(&path, "items", &[entry(&one)]).expect("first build");
10965 attach(&path, "items", &[entry(&two)]).expect("rebuild");
10966
10967 let reader = Reader::open(&path).expect("reopen");
10968 let held = attached(reader.table());
10969 assert_eq!(held.len(), 1, "one map per column and not one per build");
10970 assert_eq!(reader.payload(held[0]).expect("payload"), two);
10971
10972 fs::remove_file(&path).expect("clean up");
10973 }
10974
10975 #[test]
10976 fn an_attach_carries_through_a_kind_it_does_not_know() {
10977 let path = linked_file("attach_unknown", 16);
10981 let payload = vec![3_u8; 96];
10982 attach(
10983 &path,
10984 "items",
10985 &[section::Attachment {
10986 kind: *b"RUDBZZ9\0",
10987 id: 1,
10988 flags: 0,
10989 header_bytes: 0,
10990 bytes: &payload,
10991 }],
10992 )
10993 .expect("a kind this build does not know still writes");
10994 let key_map = a_key_map_payload();
10995 attach(
10996 &path,
10997 "items",
10998 &[section::Attachment {
10999 kind: *section::KEY_MAP,
11000 id: 0,
11001 flags: 0,
11002 header_bytes: 0,
11003 bytes: &key_map,
11004 }],
11005 )
11006 .expect("attach beside it");
11007
11008 let reader = Reader::open(&path).expect("reopen");
11009 let held = attached(reader.table());
11010 assert_eq!(held.len(), 2, "the unfamiliar entry survived a directory rewrite");
11011 let unknown = held.iter().find(|one| !one.known()).expect("the unfamiliar one");
11012 assert_eq!(reader.payload(unknown).expect("its bytes are still there"), payload);
11013
11014 fs::remove_file(&path).expect("clean up");
11015 }
11016
11017 #[test]
11018 fn a_payload_of_nothing_is_a_relationship_recorded_as_not_built() {
11019 let path = linked_file("attach_not_built", 8);
11020 attach(
11021 &path,
11022 "items",
11023 &[section::Attachment {
11024 kind: *section::FORWARD_LINK,
11025 id: 2,
11026 flags: 0,
11027 header_bytes: 0,
11028 bytes: &[],
11029 }],
11030 )
11031 .expect("record a link that did not fit the budget");
11032
11033 let reader = Reader::open(&path).expect("reopen");
11034 let held = attached(reader.table());
11035 assert_eq!(held.len(), 1);
11036 assert_eq!(held[0].extents, 0);
11037 assert_eq!(held[0].extent_page, 0, "an entry that names no bytes points at none");
11038 assert!(reader.extents(held[0]).expect("no extent table").is_empty());
11039 assert!(reader.payload(held[0]).expect("no payload").is_empty());
11040
11041 fs::remove_file(&path).expect("clean up");
11042 }
11043
11044 #[test]
11045 fn a_payload_past_one_extent_is_split_and_joined_back() {
11046 let path = linked_file("attach_two_extents", 8);
11050 let payload = vec![0x5a_u8; section::MAX_EXTENT as usize + 1];
11051 attach(
11052 &path,
11053 "items",
11054 &[section::Attachment {
11055 kind: *section::KEY_MAP,
11056 id: 0,
11057 flags: 0,
11058 header_bytes: 0,
11059 bytes: &payload,
11060 }],
11061 )
11062 .expect("attach a payload past the bound");
11063
11064 let reader = Reader::open(&path).expect("reopen");
11065 let held = attached(reader.table());
11066 let extents = reader.extents(held[0]).expect("extent table");
11067 assert_eq!(extents.len(), 2, "one byte past the bound is two extents");
11068 assert_eq!(extents[0].length, section::MAX_EXTENT);
11069 assert_eq!(extents[1].length, 1);
11070 assert_eq!(extents[1].first, u64::from(section::MAX_EXTENT));
11071 assert_eq!(reader.extent(&extents[1]).expect("the last extent"), vec![0x5a]);
11073 assert_eq!(reader.payload(held[0]).expect("the whole payload").len(), payload.len());
11074
11075 fs::remove_file(&path).expect("clean up");
11076 }
11077
11078 #[test]
11079 fn a_torn_extent_is_refused_rather_than_decoded() {
11080 let path = linked_file("attach_torn", 8);
11081 let payload = a_key_map_payload();
11082 attach(
11083 &path,
11084 "items",
11085 &[section::Attachment {
11086 kind: *section::KEY_MAP,
11087 id: 0,
11088 flags: 0,
11089 header_bytes: 0,
11090 bytes: &payload,
11091 }],
11092 )
11093 .expect("attach");
11094
11095 let reader = Reader::open(&path).expect("reopen");
11096 let extent = reader.extents(&reader.table().sections()[0]).expect("extent table")[0];
11097 let file = OpenOptions::new().write(true).open(&path).expect("reopen to corrupt");
11098 write_at(&file, extent.offset + 7, &[0xff]).expect("flip a byte of the payload");
11099 drop(file);
11100
11101 let reader = Reader::open(&path).expect("the table still opens");
11102 let error = reader
11103 .payload(&reader.table().sections()[0])
11104 .expect_err("a corrupt payload is not handed out");
11105 assert!(error.to_string().contains("checksum"), "{error}");
11106 assert_eq!(reader.read(0, &[0]).expect("the column is untouched").width(), 1);
11109
11110 fs::remove_file(&path).expect("clean up");
11111 }
11112
11113 #[test]
11114 fn attaching_to_a_file_of_the_previous_format_is_refused_rather_than_done() {
11115 let path = linked_file("attach_old_format", 8);
11118 let file = OpenOptions::new().write(true).open(&path).expect("reopen to patch");
11119 write_at(&file, 8, &22_u32.to_le_bytes()).expect("stamp the older format");
11120 drop(file);
11121
11122 let payload = a_key_map_payload();
11123 let error = attach(
11124 &path,
11125 "items",
11126 &[section::Attachment {
11127 kind: *section::KEY_MAP,
11128 id: 0,
11129 flags: 0,
11130 header_bytes: 0,
11131 bytes: &payload,
11132 }],
11133 )
11134 .expect_err("format 22 cannot gain a section");
11135 assert!(error.to_string().contains("format 22"), "{error}");
11136 assert!(Reader::open(&path).expect("and the file is untouched").table().rows() == 8);
11137
11138 fs::remove_file(&path).expect("clean up");
11139 }
11140
11141 #[test]
11142 fn a_section_header_longer_than_its_payload_is_refused_at_the_write() {
11143 let path = linked_file("attach_bad_header", 8);
11144 let error = attach(
11145 &path,
11146 "items",
11147 &[section::Attachment {
11148 kind: *section::KEY_MAP,
11149 id: 0,
11150 flags: 0,
11151 header_bytes: 40,
11152 bytes: &[1, 2, 3],
11153 }],
11154 )
11155 .expect_err("a writer's bug stops at the write");
11156 assert!(error.to_string().contains("header is longer"), "{error}");
11157
11158 fs::remove_file(&path).expect("clean up");
11159 }
11160
11161 #[test]
11162 fn attaching_to_a_name_the_file_does_not_hold_says_so() {
11163 let path = linked_file("attach_wrong_name", 8);
11164 let error = attach(&path, "orders", &[]).expect_err("no such table");
11165 assert!(error.to_string().contains("orders"), "{error}");
11166 fs::remove_file(&path).expect("clean up");
11167 }
11168
11169 #[test]
11170 fn the_planner_gets_an_exact_count_for_a_leading_value_of_an_incomplete_synopsis() {
11171 let path = path("frequency_prefix_for_the_planner");
11178 let mut writer =
11179 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
11180 .expect("new file");
11181 let mut values = vec![Value::Integer(1); 10_000];
11182 for _ in 0..10 {
11183 values.extend((0..600).map(|tail| Value::Integer(1_000 + tail)));
11184 }
11185 for part in values.chunks(8_000) {
11188 let rows = Chunk::new(vec![
11189 Vector::from_values(LogicalType::Integer, part).expect("integers"),
11190 ])
11191 .expect("one column");
11192 writer.append(&rows).expect("a part");
11193 }
11194 writer.finish().expect("commit");
11195 let reader = Reader::open(&path).expect("reopen from disk");
11196 let prefix =
11197 reader.frequency_prefix(0).expect("a readable synopsis").expect("the column has one");
11198 assert_eq!(prefix.entries.len(), 512);
11201 assert_eq!(prefix.omitted_max, 10);
11202 let common = Common::new(reader);
11203 assert_eq!(common.rows(), 16_000);
11204 let column = common.column("id").expect("the file has that column");
11205 assert_eq!(
11206 common.rows_with(column, &Bound::Int(1)),
11207 Stat::exact(10_000, Provenance::FrequencySynopsis)
11208 );
11209 assert_eq!(
11211 common.rows_with(column, &Bound::Int(1_100)),
11212 Stat::exact(10, Provenance::FrequencySynopsis)
11213 );
11214 assert_eq!(common.rows_with(column, &Bound::Int(1_550)), Stat::Unknown);
11217 assert_eq!(common.rows_with(column, &Bound::Int(9_999)), Stat::Unknown);
11220 let remainder = common.remainder(column).expect("the list is a prefix");
11224 assert_eq!(remainder, Remainder { rows: 890, listed: 512, most: 10 });
11225 assert_eq!(remainder.rows / (601 - remainder.listed), 10);
11226 fs::remove_file(&path).expect("clean up");
11227 }
11228
11229 #[test]
11231 fn a_file_holding_no_table_commits_and_opens_and_a_table_can_be_added_to_it() {
11232 let path = path("empty");
11233 Writer::empty(&path, &[]).expect("a file with nothing in it");
11234 let catalog = Catalog::open(&path).expect("the empty file opens");
11235 assert_eq!(catalog.len(), 0);
11236 assert!(catalog.is_empty());
11237 assert_eq!(catalog.names().count(), 0);
11238 let mut writer =
11241 Writer::open(&path, "items", vec![Field::required("id", LogicalType::Integer)])
11242 .expect("a table goes into the empty file");
11243 writer.append(&sample_ids()).expect("rows");
11244 writer.finish().expect("commit");
11245 let catalog = Catalog::open(&path).expect("the file opens again");
11246 assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
11247 fs::remove_file(&path).expect("clean up");
11248 }
11249
11250 #[test]
11260 fn a_committed_empty_table_gives_up_its_name_and_one_with_rows_does_not() {
11261 let path = path("empty-name");
11262 let field = || vec![Field::required("id", LogicalType::Integer)];
11263 Writer::create(&path, "items", field()).expect("new file").finish().expect("commit");
11264 let catalog = Catalog::open(&path).expect("the file opens");
11265 assert_eq!(catalog.rows().collect::<Vec<_>>(), vec![("items", 0)]);
11266
11267 let mut writer = Writer::open(&path, "items", field()).expect("the empty name is free");
11268 writer.append(&sample_ids()).expect("rows");
11269 writer.finish().expect("commit");
11270 let catalog = Catalog::open(&path).expect("the file opens again");
11271 assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
11273 let held = catalog.rows().collect::<Vec<_>>();
11274 assert_eq!(held.len(), 1);
11275 assert!(held[0].1 > 0, "the rows that were appended are the ones the catalog counts");
11276
11277 let error = Writer::open(&path, "items", field()).expect_err("a name with rows is taken");
11279 assert!(error.to_string().contains("same name"), "{error}");
11280 fs::remove_file(&path).expect("clean up");
11281 }
11282
11283 fn sample_view(name: &str) -> ViewEntry {
11285 ViewEntry {
11286 name: name.to_string(),
11287 sql: "SELECT id FROM items WHERE id > 0".to_string(),
11288 statement: format!("CREATE VIEW {name} AS SELECT id FROM items WHERE (id > 0);"),
11289 aliases: vec!["n".to_string()],
11290 columns: vec![Field::new("n", LogicalType::Integer)],
11291 }
11292 }
11293
11294 #[test]
11295 fn a_view_written_into_the_catalog_comes_back_whole() {
11296 let path = path("views");
11297 let mut writer =
11298 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
11299 .expect("new file");
11300 writer.append(&sample_ids()).expect("rows");
11301 writer.with_views(vec![sample_view("v")]).finish().expect("commit");
11302 let catalog = Catalog::open(&path).expect("reopen");
11303 assert_eq!(catalog.views().cloned().collect::<Vec<_>>(), vec![sample_view("v")]);
11304 assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
11307 fs::remove_file(&path).expect("clean up");
11308 }
11309
11310 #[test]
11312 fn appending_a_table_carries_the_views_forward() {
11313 let path = path("viewscarry");
11314 let mut writer =
11315 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
11316 .expect("new file");
11317 writer.append(&sample_ids()).expect("rows");
11318 writer.with_views(vec![sample_view("v")]).finish().expect("commit");
11319 let mut writer =
11320 Writer::open(&path, "other", vec![Field::required("id", LogicalType::Integer)])
11321 .expect("a second table");
11322 writer.append(&sample_ids()).expect("rows");
11323 writer.finish().expect("commit");
11324 let catalog = Catalog::open(&path).expect("reopen");
11325 assert_eq!(catalog.views().count(), 1);
11326 assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items", "other"]);
11327 fs::remove_file(&path).expect("clean up");
11328 }
11329
11330 #[test]
11332 fn restating_the_views_leaves_every_table_where_it_was() {
11333 let path = path("restate");
11334 let mut writer =
11335 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
11336 .expect("new file");
11337 writer.append(&sample_ids()).expect("rows");
11338 writer.finish().expect("commit");
11339 let before = fs::metadata(&path).expect("the file is there").len();
11340 Writer::restate(&path, &[sample_view("v"), sample_view("w")]).expect("two views");
11341 let catalog = Catalog::open(&path).expect("reopen");
11342 assert_eq!(catalog.views().count(), 2);
11343 assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
11344 let after = fs::metadata(&path).expect("the file is there").len();
11347 assert!(after > before, "a generation was written");
11348 assert!(after - before < before, "the table was not written again");
11349 let reader = Catalog::open(&path).expect("reopen").table("items").expect("the table");
11352 assert_eq!(reader.table().rows, 3);
11353 Writer::restate(&path, &[]).expect("no views at all");
11356 assert_eq!(Catalog::open(&path).expect("reopen").views().count(), 0);
11357 fs::remove_file(&path).expect("clean up");
11358 }
11359
11360 #[test]
11362 fn a_view_named_after_a_table_is_refused_when_the_catalog_is_read() {
11363 let bytes = encode_catalog(
11364 &[Entry {
11365 name: "items".to_string(),
11366 fields: vec![Field::required("id", LogicalType::Integer)],
11367 rows: 1,
11368 directory: Page { offset: HEADER, length: 8, hash: 0 },
11369 nonzero: vec![None],
11370 aggregates: vec![None],
11371 }],
11372 &[sample_view("items")],
11373 )
11374 .expect("it encodes, because encoding does not look");
11375 let error = decode_catalog(&bytes, HEADER + 8).expect_err("and decoding does");
11376 assert!(error.to_string().contains("same name"), "{error}");
11377 }
11378
11379 #[test]
11380 fn committed_file_reopens_and_reads_only_requested_columns() {
11381 let path = path("reopen");
11382 let mut writer = Writer::create(
11383 &path,
11384 "items",
11385 vec![
11386 Field::required("id", LogicalType::Integer),
11387 Field::new("text", LogicalType::Varchar),
11388 ],
11389 )
11390 .expect("new file");
11391 writer.append(&sample()).expect("first part");
11392 writer.append(&sample()).expect("second part");
11393 writer.finish().expect("commit");
11394 let reader = Reader::open(&path).expect("reopen from disk");
11395 assert_eq!(reader.table().rows(), 6);
11396 assert_eq!(reader.table().stripes().len(), 1);
11399 assert_eq!(reader.parts(), 2);
11400 assert_eq!(reader.part_rows(0), 3);
11401 assert_eq!(reader.part_rows(1), 3);
11402 let text = reader.read(1, &[1]).expect("only text page");
11403 assert_eq!(text.width(), 1);
11404 assert_eq!(text.value_at(1, 0), Value::Null);
11405 assert_eq!(text.value_at(2, 0), Value::Varchar("long text after a slash".into()));
11406 let sparse = reader.read_sparse(1, &[1]).expect("one part without its whole page");
11407 assert_eq!(sparse.width(), 1);
11408 assert_eq!(sparse.value_at(1, 0), Value::Null);
11409 assert_eq!(sparse.value_at(2, 0), Value::Varchar("long text after a slash".into()));
11410 assert!(!reader.skips_codes(0, 1, &[0]).expect("alpha is in the stripe"));
11411 assert!(!reader.skips_codes(0, 1, &[2]).expect("long text is in the stripe"));
11412 assert!(reader.skips_codes(0, 1, &[3]).expect("unknown code is absent"));
11413 let count = reader.read(0, &[]).expect("no page is needed for count");
11414 assert_eq!(count.len(), 3);
11415 assert!(reader.skips(0, &[Probe { column: 0, op: Op::Greater, value: Bound::Int(100) }]));
11416 assert!(!reader.skips(0, &[Probe { column: 0, op: Op::Greater, value: Bound::Int(0) }]));
11417 let integers = reader.top_frequencies(0, 1).expect("valid integer synopsis").expect("kept");
11418 assert_eq!(
11419 integers,
11420 vec![(Value::Integer(-2), 2), (Value::Integer(4), 2), (Value::Integer(9), 2),]
11421 );
11422 let strings = reader.top_frequencies(1, 1).expect("valid string synopsis").expect("kept");
11423 assert_eq!(strings.len(), 3);
11424 assert!(strings.contains(&(Value::Null, 2)));
11425 assert!(strings.contains(&(Value::Varchar("alpha".into()), 2)));
11426 assert!(strings.contains(&(Value::Varchar("long text after a slash".into()), 2)));
11427 fs::remove_file(path).expect("remove scratch file");
11428 }
11429
11430 #[test]
11438 fn runs_handed_over_out_of_order_still_read_back_in_source_order() {
11439 let path = path("interleaved-runs");
11440 let mut writer =
11441 Writer::create(&path, "interleaved", vec![Field::new("v", LogicalType::BigInt)])
11442 .expect("new file");
11443 for morsel in [2_u64, 0, 3, 1] {
11444 let parts = (0..4_u64)
11445 .map(|chunk| {
11446 let first = i64::try_from(morsel * 32 + chunk * 8).expect("small");
11447 let values =
11448 (0..8_i64).map(|row| Value::BigInt(first + row)).collect::<Vec<_>>();
11449 let column =
11450 Vector::from_values(LogicalType::BigInt, &values).expect("a column");
11451 ((morsel, chunk), Chunk::new(vec![column]).expect("one column"))
11452 })
11453 .collect::<Vec<_>>();
11454 writer.append_stripe(parts).expect("a stripe");
11455 }
11456 writer.finish().expect("commit");
11457
11458 let reader = Reader::open(&path).expect("valid directory");
11459 assert_eq!(reader.table().stripes().len(), 4, "a run is a stripe of its own");
11460 assert_eq!(reader.table().rows(), 128);
11461 for part in 0..16_usize {
11462 let read = reader.read(part, &[0]).expect("a part back");
11463 for row in 0..8_usize {
11464 let want = i64::try_from(part * 8 + row).expect("small");
11465 assert_eq!(read.value_at(row, 0), Value::BigInt(want), "part {part} row {row}");
11466 }
11467 }
11468 fs::remove_file(path).expect("remove scratch file");
11469 }
11470
11471 #[test]
11474 fn runs_that_overlap_each_other_are_refused_at_commit() {
11475 let path = path("overlapping-runs");
11476 let mut writer =
11477 Writer::create(&path, "overlapping", vec![Field::new("v", LogicalType::BigInt)])
11478 .expect("new file");
11479 let one = |order: (u64, u64)| {
11480 let column =
11481 Vector::from_values(LogicalType::BigInt, &[Value::BigInt(1)]).expect("a column");
11482 (order, Chunk::new(vec![column]).expect("one column"))
11483 };
11484 writer.append_stripe(vec![one((0, 0)), one((0, 2))]).expect("a stripe");
11487 writer.append_stripe(vec![one((0, 1))]).expect("a stripe");
11488 let error = writer.finish().expect_err("the runs overlap");
11489 assert!(error.message().contains("source order"), "{error}");
11490 fs::remove_file(path).expect("remove scratch file");
11491 }
11492
11493 #[test]
11496 fn a_run_longer_than_a_stripe_is_refused() {
11497 let path = path("overlong-run");
11498 let mut writer =
11499 Writer::create(&path, "overlong", vec![Field::new("v", LogicalType::BigInt)])
11500 .expect("new file");
11501 let parts = (0..=STRIPE_PARTS)
11502 .map(|at| {
11503 let column = Vector::from_values(LogicalType::BigInt, &[Value::BigInt(1)])
11504 .expect("a column");
11505 let chunk = Chunk::new(vec![column]).expect("one column");
11506 ((0, u64::try_from(at).expect("small")), chunk)
11507 })
11508 .collect::<Vec<_>>();
11509 let error = writer.append_stripe(parts).expect_err("one part too many");
11510 assert!(error.message().contains("more parts than it holds"), "{error}");
11511 fs::remove_file(path).expect("remove scratch file");
11512 }
11513
11514 #[test]
11520 fn parts_past_the_stripe_bound_start_a_new_stripe() {
11521 let path = path("stripe-bound");
11522 let mut writer = Writer::create(
11523 &path,
11524 "items",
11525 vec![
11526 Field::required("id", LogicalType::Integer),
11527 Field::new("text", LogicalType::Varchar),
11528 ],
11529 )
11530 .expect("new file");
11531 let parts = STRIPE_PARTS * 2 + 3;
11532 for part in 0..parts {
11533 let id = part as i32;
11534 let chunk = Chunk::new(vec![
11535 Vector::from_values(
11536 LogicalType::Integer,
11537 &[Value::Integer(id), Value::Integer(-id)],
11538 )
11539 .expect("integers"),
11540 Vector::from_values(
11541 LogicalType::Varchar,
11542 &[Value::Varchar(format!("value {part}")), Value::Null],
11543 )
11544 .expect("strings"),
11545 ])
11546 .expect("matching rows");
11547 writer.append(&chunk).expect("one part");
11548 }
11549 writer.finish().expect("commit");
11550
11551 let reader = Reader::open(&path).expect("reopen from disk");
11552 assert_eq!(reader.parts(), parts);
11553 assert_eq!(reader.table().rows(), parts * 2);
11554 assert_eq!(reader.table().stripes().len(), parts.div_ceil(STRIPE_PARTS));
11555 assert_eq!(reader.table().stripes()[0].parts(), STRIPE_PARTS);
11556 assert_eq!(reader.table().stripes()[0].rows(), STRIPE_PARTS * 2);
11557 assert_eq!(reader.table().stripes()[2].parts(), 3);
11558 for part in (0..parts).rev() {
11561 let dense = reader.read(part, &[0, 1]).expect("a whole page read");
11562 let sparse = reader.read_sparse(part, &[0, 1]).expect("one part read");
11563 for chunk in [&dense, &sparse] {
11564 assert_eq!(chunk.len(), 2, "part {part} has its own row count");
11565 assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
11566 assert_eq!(chunk.value_at(1, 0), Value::Integer(-(part as i32)));
11567 assert_eq!(chunk.value_at(0, 1), Value::Varchar(format!("value {part}")));
11568 assert_eq!(chunk.value_at(1, 1), Value::Null);
11569 }
11570 }
11571 let above = [Probe { column: 0, op: Op::Greater, value: Bound::Int(100) }];
11574 assert!(reader.skips(0, &above), "the first stripe stops at 63");
11575 assert!(!reader.skips(STRIPE_PARTS * 2, &above), "the third stripe reaches 130");
11576 fs::remove_file(path).expect("remove scratch file");
11577 }
11578
11579 fn scattered(n: i64) -> i64 {
11581 n.wrapping_mul(-7_046_029_254_386_353_131)
11582 }
11583
11584 #[test]
11590 fn a_part_is_skipped_when_its_sieve_does_not_hold_the_constant() {
11591 let path = path("sieve-skip");
11592 let mut writer =
11593 Writer::create(&path, "hits", vec![Field::required("id", LogicalType::BigInt)])
11594 .expect("new file");
11595 let parts = STRIPE_PARTS + 3;
11596 let per_part = 128;
11600 for part in 0..parts {
11601 let held: Vec<Value> = (0..per_part)
11602 .map(|row| Value::BigInt(scattered((part * per_part + row) as i64)))
11603 .collect();
11604 let chunk =
11605 Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
11606 .expect("one column");
11607 writer.append(&chunk).expect("one part");
11608 }
11609 writer.finish().expect("commit");
11610
11611 let reader = Reader::open(&path).expect("reopen from disk");
11612 let probe = |value: i64| Probe {
11613 column: 0,
11614 op: Op::Equal,
11615 value: Bound::Int(i128::from(scattered(value))),
11616 };
11617 for wanted in [0_i64, (per_part + 1) as i64, (parts * per_part - 1) as i64] {
11618 let tests = [probe(wanted)];
11619 let kept: Vec<usize> = (0..parts).filter(|&part| !reader.skips(part, &tests)).collect();
11620 let home = wanted as usize / per_part;
11621 assert!(kept.contains(&home), "the part holding {wanted} is read");
11622 assert!(kept.len() <= 2, "{wanted} keeps {kept:?}, which is more than one stray part");
11626 }
11627 let absent = [probe((parts * per_part) as i64 + 1)];
11628 let kept = (0..parts).filter(|&part| !reader.skips(part, &absent)).count();
11629 assert!(kept <= 1, "{kept} parts of {parts} kept a value no part holds");
11630 let tests = [probe(0)];
11633 assert!(
11634 reader.table().stripes().iter().all(|stripe| !stripe.zone.skips(&tests)),
11635 "the bounds rule out no stripe at all"
11636 );
11637 fs::remove_file(path).expect("remove scratch file");
11638 }
11639
11640 #[test]
11646 fn a_part_is_skipped_when_its_own_bounds_rule_out_a_comparison_the_stripe_keeps() {
11647 let path = path("part-range-skip");
11648 let mut writer =
11649 Writer::create(&path, "hits", vec![Field::required("at", LogicalType::BigInt)])
11650 .expect("new file");
11651 let parts = STRIPE_PARTS + 3;
11652 let per_part = 128;
11653 for part in 0..parts {
11654 let held: Vec<Value> = (0..per_part)
11658 .map(|row| {
11659 Value::BigInt((part * 1_000) as i64 + (scattered(row as i64).rem_euclid(900)))
11660 })
11661 .collect();
11662 let chunk =
11663 Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
11664 .expect("one column");
11665 writer.append(&chunk).expect("one part");
11666 }
11667 writer.finish().expect("commit");
11668
11669 let reader = Reader::open(&path).expect("reopen from disk");
11670 let under = [Probe { column: 0, op: Op::Less, value: Bound::Int(3_000) }];
11671 let kept: Vec<usize> = (0..parts).filter(|&part| !reader.skips(part, &under)).collect();
11672 assert_eq!(kept, vec![0, 1, 2], "only the three parts that start under three thousand");
11673 assert!(!reader.stripe_skips(0, &under), "the stripe reaches from zero and keeps itself");
11675 fs::remove_file(path).expect("remove scratch file");
11676 }
11677
11678 #[test]
11682 fn a_part_is_waved_through_when_its_own_bounds_pass_a_comparison_the_stripe_cannot() {
11683 let path = path("part-range-certain");
11684 let mut writer =
11685 Writer::create(&path, "hits", vec![Field::required("at", LogicalType::BigInt)])
11686 .expect("new file");
11687 let parts = STRIPE_PARTS + 3;
11688 let per_part = 128;
11689 for part in 0..parts {
11690 let held: Vec<Value> = (0..per_part)
11691 .map(|row| {
11692 Value::BigInt((part * 1_000) as i64 + (scattered(row as i64).rem_euclid(900)))
11693 })
11694 .collect();
11695 let chunk =
11696 Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
11697 .expect("one column");
11698 writer.append(&chunk).expect("one part");
11699 }
11700 writer.finish().expect("commit");
11701
11702 let reader = Reader::open(&path).expect("reopen from disk");
11703 let under = [Probe { column: 0, op: Op::Less, value: Bound::Int(3_000) }];
11704 let waved: Vec<usize> = (0..parts).filter(|&part| reader.certain(part, &under)).collect();
11705 assert_eq!(waved, vec![0, 1, 2], "the three parts that end under three thousand");
11706 assert!(!reader.stripe_skips(0, &under), "the stripe straddles the comparison");
11709 fs::remove_file(path).expect("remove scratch file");
11710 }
11711
11712 #[test]
11715 fn a_stripe_of_one_part_writes_no_range_page_and_a_stripe_of_many_does() {
11716 for (parts, wanted) in [(1_usize, false), (STRIPE_PARTS, true)] {
11717 let path = path("part-range-page");
11718 let mut writer =
11719 Writer::create(&path, "hits", vec![Field::required("at", LogicalType::BigInt)])
11720 .expect("new file");
11721 for part in 0..parts {
11722 let held: Vec<Value> = (0..128)
11723 .map(|row| {
11724 Value::BigInt((part * 1_000) as i64 + scattered(row as i64).rem_euclid(900))
11725 })
11726 .collect();
11727 let chunk = Chunk::new(vec![
11728 Vector::from_values(LogicalType::BigInt, &held).expect("numbers"),
11729 ])
11730 .expect("one column");
11731 writer.append(&chunk).expect("one part");
11732 }
11733 writer.finish().expect("commit");
11734 let reader = Reader::open(&path).expect("reopen from disk");
11735 let bytes = reader.layout().columns[0].part_ranges;
11736 assert_eq!(bytes > 0, wanted, "{parts} parts wrote {bytes} bytes of ranges");
11737 fs::remove_file(path).expect("remove scratch file");
11738 }
11739 }
11740
11741 #[test]
11744 fn a_string_end_that_is_cut_down_still_covers_the_value_it_came_from() {
11745 let long = vec![b'a'; PART_BOUND_BYTES * 2];
11746 let low = shortened(Some(Bound::Bytes(long.clone())), false).expect("a low end");
11747 let high = shortened(Some(Bound::Bytes(long.clone())), true).expect("a high end");
11748 let Bound::Bytes(low) = low else { panic!("a string stays a string") };
11749 let Bound::Bytes(high) = high else { panic!("a string stays a string") };
11750 assert!(low.len() <= PART_BOUND_BYTES && high.len() <= PART_BOUND_BYTES);
11751 assert!(low.as_slice() <= long.as_slice(), "the low end is at or under the value");
11752 assert!(high.as_slice() >= long.as_slice(), "the high end is at or over the value");
11753 }
11754
11755 #[test]
11758 fn a_string_end_with_no_room_to_step_up_gives_up_the_bound() {
11759 let long = vec![u8::MAX; PART_BOUND_BYTES * 2];
11760 assert_eq!(shortened(Some(Bound::Bytes(long.clone())), true), None);
11761 let low = shortened(Some(Bound::Bytes(long)), false).expect("a low end is still a prefix");
11762 assert_eq!(low, Bound::Bytes(vec![u8::MAX; PART_BOUND_BYTES]));
11763 }
11764
11765 #[test]
11777 fn what_a_column_is_stored_as_follows_the_order_the_rows_were_written_in() {
11778 let parts = 4;
11779 let per_part = 1024;
11780 let rows = parts * per_part;
11781 let written = |name: &str, keys: &[i64]| {
11782 let path = path(name);
11783 let fields = vec![Field::required("key", LogicalType::BigInt)];
11784 let mut writer = Writer::create(&path, "keys", fields).expect("new file");
11785 for part in 0..parts {
11786 let values: Vec<Value> = keys[part * per_part..(part + 1) * per_part]
11787 .iter()
11788 .map(|key| Value::BigInt(*key))
11789 .collect();
11790 let chunk = Chunk::new(vec![
11791 Vector::from_values(LogicalType::BigInt, &values).expect("numbers"),
11792 ])
11793 .expect("one column");
11794 writer.append(&chunk).expect("one part");
11795 }
11796 writer.finish().expect("commit");
11797 path
11798 };
11799 let climbing = |step: &dyn Fn(usize) -> i64| {
11802 let mut key = 0;
11803 (0..rows)
11804 .map(|row| {
11805 key += step(row);
11806 key
11807 })
11808 .collect::<Vec<i64>>()
11809 };
11810 let ascending = climbing(&|row| (row % 3) as i64);
11811 let sparse = climbing(&|row| ((row * 2_654_435_761) % 4096) as i64);
11815 let near_path = written("stored-near", &ascending);
11816 let far_path = written("stored-far", &sparse);
11817
11818 let one = Reader::open(&near_path).expect("reopen from disk");
11819 let other = Reader::open(&far_path).expect("reopen from disk");
11820 let near = one.stored(0).expect("the column is stored");
11821 let far = other.stored(0).expect("the column is stored");
11822 assert_eq!(near.len(), parts, "one row per part");
11823 assert_eq!(far.len(), parts);
11824 let total = |stored: &[StoredPart]| stored.iter().map(|part| part.bytes).sum::<u64>();
11827 assert_eq!(total(&near), one.layout().columns[0].pages);
11828 assert_eq!(total(&far), other.layout().columns[0].pages);
11829 assert!(
11830 total(&near) * 2 < total(&far),
11831 "the sparse keys cost more, {} against {}",
11832 total(&far),
11833 total(&near)
11834 );
11835 for (at, part) in near.iter().enumerate() {
11837 assert_eq!(part.part, at);
11838 assert_eq!(part.row, at * per_part);
11839 assert_eq!(part.rows, per_part);
11840 let held = &ascending[at * per_part..(at + 1) * per_part];
11841 assert_eq!(part.low, Some(Value::BigInt(held[0])));
11842 assert_eq!(part.high, Some(Value::BigInt(held[per_part - 1])));
11843 assert_eq!(part.nulls, Some(0));
11844 }
11845 assert!(near[0].encoding.contains("DELTA"), "{}", near[0].encoding);
11848 assert!(far[0].encoding.contains("DELTA"), "{}", far[0].encoding);
11849 assert_ne!(near[0].encoding, far[0].encoding);
11850 fs::remove_file(near_path).expect("remove scratch file");
11851 fs::remove_file(far_path).expect("remove scratch file");
11852 }
11853
11854 #[test]
11864 fn a_sieve_larger_than_the_part_it_indexes_is_not_written() {
11865 let path = path("sieve-pays");
11866 let fields = vec![
11867 Field::required("spread", LogicalType::BigInt),
11868 Field::required("repeated", LogicalType::BigInt),
11869 ];
11870 let mut writer = Writer::create(&path, "hits", fields).expect("new file");
11871 let parts = 3;
11872 let per_part = 1024;
11873 for part in 0..parts {
11874 let base = (part * per_part) as i64;
11875 let spread: Vec<Value> =
11876 (0..per_part).map(|row| Value::BigInt(scattered(base + row as i64))).collect();
11877 let repeated: Vec<Value> =
11878 (0..per_part).map(|row| Value::BigInt(scattered((row / 256) as i64))).collect();
11879 let chunk = Chunk::new(vec![
11880 Vector::from_values(LogicalType::BigInt, &spread).expect("numbers"),
11881 Vector::from_values(LogicalType::BigInt, &repeated).expect("numbers"),
11882 ])
11883 .expect("two columns");
11884 writer.append(&chunk).expect("one part");
11885 }
11886 writer.finish().expect("commit");
11887
11888 let reader = Reader::open(&path).expect("reopen from disk");
11889 let layout = reader.layout();
11890 let spread = &layout.columns[0];
11891 let repeated = &layout.columns[1];
11892 assert!(spread.sieves > 0, "a column whose parts are worth a filter keeps one");
11893 assert_eq!(
11894 repeated.sieves, 0,
11895 "a column whose filter costs more than its parts keeps none"
11896 );
11897 for column in &layout.columns {
11900 assert!(
11901 column.sieves < column.pages,
11902 "{} spends {} on sieves over {} of data",
11903 column.name,
11904 column.sieves,
11905 column.pages
11906 );
11907 }
11908 let absent = [Probe {
11910 column: 0,
11911 op: Op::Equal,
11912 value: Bound::Int(i128::from(scattered((parts * per_part) as i64 + 1))),
11913 }];
11914 assert!((0..parts).all(|part| reader.skips(part, &absent)), "no part holds it");
11915 fs::remove_file(path).expect("remove scratch file");
11916 }
11917
11918 #[test]
11924 fn a_damaged_sieve_page_is_read_through_rather_than_refused() {
11925 let path = path("sieve-damaged");
11926 let mut writer =
11927 Writer::create(&path, "hits", vec![Field::required("id", LogicalType::BigInt)])
11928 .expect("new file");
11929 let rows = 128;
11930 let held: Vec<Value> = (0..rows).map(|row| Value::BigInt(scattered(row))).collect();
11931 let chunk =
11932 Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
11933 .expect("one column");
11934 writer.append(&chunk).expect("one part");
11935 writer.finish().expect("commit");
11936
11937 let page = Reader::open(&path).expect("reopen").table.stripes[0]
11938 .sieves
11939 .get(0)
11940 .expect("a sieve page");
11941 let mut file = OpenOptions::new().write(true).open(&path).expect("open the sieve page");
11942 file.seek(SeekFrom::Start(page.offset + u64::from(page.length) - 1)).expect("seek");
11943 file.write_all(&[0xff]).expect("damage one byte");
11944 drop(file);
11945
11946 let reader = Reader::open(&path).expect("reopen the damaged file");
11947 let absent =
11948 [Probe { column: 0, op: Op::Equal, value: Bound::Int(i128::from(scattered(99))) }];
11949 assert!(!reader.skips(0, &absent), "a sieve that cannot be read skips nothing");
11950 assert_eq!(
11951 reader.read(0, &[0]).expect("the rows are untouched").len(),
11952 usize::try_from(rows).expect("a small count")
11953 );
11954 fs::remove_file(path).expect("remove scratch file");
11955 }
11956
11957 #[test]
11968 fn workers_that_want_the_same_stripe_read_it_once() {
11969 let path = path("single-flight");
11970 let mut writer =
11971 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
11972 .expect("new file");
11973 for part in 0..STRIPE_PARTS {
11974 let id = part as i32;
11975 let chunk = Chunk::new(vec![
11976 Vector::from_values(
11977 LogicalType::Integer,
11978 &[Value::Integer(id), Value::Integer(-id)],
11979 )
11980 .expect("integers"),
11981 ])
11982 .expect("matching rows");
11983 writer.append(&chunk).expect("one part");
11984 }
11985 writer.finish().expect("commit");
11986
11987 let reader = Reader::open(&path).expect("reopen from disk");
11988 assert_eq!(reader.table().stripes().len(), 1, "one stripe is the point of the test");
11989 let barrier = std::sync::Barrier::new(8);
11990 std::thread::scope(|scope| {
11991 for worker in 0..8 {
11992 let reader = &reader;
11993 let barrier = &barrier;
11994 scope.spawn(move || {
11995 barrier.wait();
11996 for part in (worker..STRIPE_PARTS).step_by(8) {
11997 let chunk = reader.read(part, &[0]).expect("a whole page read");
11998 assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
11999 assert_eq!(chunk.value_at(1, 0), Value::Integer(-(part as i32)));
12000 }
12001 });
12002 }
12003 });
12004 assert_eq!(reader.pages.load(Atomic::Relaxed), 1, "one stripe, one page read, whoever won");
12005 fs::remove_file(path).expect("remove scratch file");
12006 }
12007
12008 #[test]
12021 fn opening_costs_the_same_over_a_thousand_times_the_rows() {
12022 let opened = |label: &str, rows_per_part: i32| {
12023 let path = path(label);
12024 let mut writer =
12025 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
12026 .expect("new file");
12027 for part in 0..STRIPE_PARTS * 3 {
12028 let values = (0..rows_per_part)
12032 .map(|row| {
12033 Value::Integer((part as i32 * rows_per_part + row).wrapping_mul(2_654_435))
12034 })
12035 .collect::<Vec<_>>();
12036 let chunk = Chunk::new(vec![
12037 Vector::from_values(LogicalType::Integer, &values).expect("integers"),
12038 ])
12039 .expect("matching rows");
12040 writer.append(&chunk).expect("one part");
12041 }
12042 writer.finish().expect("commit");
12043 let reader = Reader::open(&path).expect("reopen from disk");
12044 let size = fs::metadata(&path).expect("the file is there").len();
12045 let out = (reader.reads(), reader.table().stripes().len(), size);
12046 fs::remove_file(path).expect("remove scratch file");
12047 out
12048 };
12049
12050 let (thin, thin_stripes, thin_size) = opened("open-thin", 1);
12051 let (fat, fat_stripes, fat_size) = opened("open-fat", 1000);
12052 assert_eq!(
12053 thin_stripes, fat_stripes,
12054 "the same stripe count is what makes this a fair ask"
12055 );
12056 assert!(
12057 fat_size > thin_size * 50,
12058 "the fat file has to actually be larger, and it is {fat_size} against {thin_size}"
12059 );
12060
12061 assert_eq!(thin.opening.reads, fat.opening.reads, "the same reads either way");
12062 assert_eq!(thin.pages, 0, "opening read a page");
12063 assert_eq!(fat.pages, 0, "opening read a page");
12064 assert_eq!(thin.indexes, 0, "opening read an index");
12065 assert_eq!(fat.indexes, 0, "opening read an index");
12066 assert!(
12069 fat.opening.bytes < thin.opening.bytes * 2,
12070 "opening the thin file read {} bytes and the fat one read {}",
12071 thin.opening.bytes,
12072 fat.opening.bytes
12073 );
12074 }
12075
12076 #[test]
12084 fn two_opens_of_one_file_cost_the_same_and_the_second_is_not_cheaper() {
12085 let path = path("open-twice");
12086 let mut writer =
12087 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
12088 .expect("new file");
12089 for part in 0..STRIPE_PARTS * 3 {
12090 let chunk = Chunk::new(vec![
12091 Vector::from_values(LogicalType::Integer, &[Value::Integer(part as i32)])
12092 .expect("integers"),
12093 ])
12094 .expect("matching rows");
12095 writer.append(&chunk).expect("one part");
12096 }
12097 writer.finish().expect("commit");
12098
12099 let first = Reader::open(&path).expect("open");
12100 for part in 0..first.parts() {
12103 first.read(part, &[0]).expect("a part");
12104 }
12105 assert!(first.reads().pages > 0, "the scan has to have read something");
12106 let second = Reader::open(&path).expect("open again");
12107
12108 assert_eq!(first.reads().opening, second.reads().opening);
12109 assert_eq!(
12110 second.reads().pages,
12111 0,
12112 "the second open read a page off the back of the first"
12113 );
12114 assert_eq!(second.reads().indexes, 0, "the second open read an index it inherited");
12115 fs::remove_file(path).expect("remove scratch file");
12116 }
12117
12118 #[test]
12126 fn an_index_is_read_once_per_stripe_however_often_the_page_is_evicted() {
12127 let path = path("index-cache");
12128 let mut writer =
12129 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
12130 .expect("new file");
12131 let parts = STRIPE_PARTS * (CACHED_STRIPES_PER_COLUMN + 2);
12132 for part in 0..parts {
12133 let id = part as i32;
12134 let chunk = Chunk::new(vec![
12135 Vector::from_values(LogicalType::Integer, &[Value::Integer(id)]).expect("integers"),
12136 ])
12137 .expect("matching rows");
12138 writer.append(&chunk).expect("one part");
12139 }
12140 writer.finish().expect("commit");
12141
12142 let reader = Reader::open(&path).expect("reopen from disk");
12143 let stripes = reader.table().stripes().len();
12144 assert!(stripes > CACHED_STRIPES_PER_COLUMN, "the page cache has to be too small for this");
12145 for _ in 0..2 {
12147 for part in 0..parts {
12148 let chunk = reader.read(part, &[0]).expect("a part");
12149 assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
12150 }
12151 }
12152 assert_eq!(reader.indexes.load(Atomic::Relaxed), stripes, "one index read per stripe");
12153 assert!(
12154 reader.pages.load(Atomic::Relaxed) > stripes,
12155 "the pages are the ones that get read again, which is what makes the index count mean \
12156 something"
12157 );
12158 fs::remove_file(path).expect("remove scratch file");
12159 }
12160
12161 #[test]
12168 fn a_pool_keeps_pages_between_scans_and_gives_them_to_the_table_being_read() {
12169 let path = path("page-pool");
12170 let parts = STRIPE_PARTS * (CACHED_STRIPES_PER_COLUMN + 2);
12171 let fields = || vec![Field::required("id", LogicalType::Integer)];
12172 let mut writer = Writer::create(&path, "a", fields()).expect("new file");
12173 for table in ["a", "b"] {
12174 if table == "b" {
12175 writer = writer.next("b".to_string(), fields()).expect("a second table");
12176 }
12177 for part in 0..parts {
12178 let chunk = Chunk::new(vec![
12179 Vector::from_values(LogicalType::Integer, &[Value::Integer(part as i32)])
12180 .expect("integers"),
12181 ])
12182 .expect("matching rows");
12183 writer.append(&chunk).expect("one part");
12184 }
12185 }
12186 writer.finish().expect("commit");
12187
12188 let pool = PagePool::new(usize::MAX);
12189 let catalog = Catalog::open_in(&path, &pool).expect("the file opens");
12190 let (a, b) = (catalog.table("a").expect("a"), catalog.table("b").expect("b"));
12191 let stripes = a.table().stripes().len();
12192 assert!(stripes > CACHED_STRIPES_PER_COLUMN, "the floor has to be smaller than a table");
12193 let scan = |reader: &Reader| {
12194 for part in 0..parts {
12195 let chunk = reader.read(part, &[0]).expect("a part");
12196 assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
12197 }
12198 };
12199 scan(&a);
12200 scan(&a);
12201 assert_eq!(a.pages.load(Atomic::Relaxed), stripes, "the second scan reads nothing");
12202 let one = pool.bytes();
12203 assert!(one > 0, "the pool counts what the reader holds");
12204
12205 pool.budget.store(one, Atomic::Relaxed);
12207 scan(&b);
12208 assert_eq!(b.pages.load(Atomic::Relaxed), stripes, "a page is never let go while in use");
12209 assert_eq!(a.cache.held[0].load(Atomic::Relaxed), CACHED_STRIPES_PER_COLUMN);
12210 let held = a.cache.columns[0].lock().expect("the column").pages.iter().flatten().count();
12211 assert_eq!(held, CACHED_STRIPES_PER_COLUMN, "the count and the slots agree");
12212
12213 drop((a, b, catalog));
12215 let c = Catalog::open_in(&path, &pool).expect("again").table("a").expect("a");
12216 scan(&c);
12217 assert!(pool.bytes() <= one, "only what the live reader holds is counted");
12218 fs::remove_file(path).expect("remove scratch file");
12219 }
12220
12221 #[test]
12230 fn a_worker_per_stripe_reads_its_page_once_when_the_cache_was_told_to_expect_it() {
12231 let workers = CACHED_STRIPES_PER_COLUMN + 4;
12232 let path = path("stripe-per-worker");
12233 let mut writer =
12234 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
12235 .expect("new file");
12236 for part in 0..STRIPE_PARTS * workers {
12237 let chunk = Chunk::new(vec![
12238 Vector::from_values(LogicalType::Integer, &[Value::Integer(part as i32)])
12239 .expect("integers"),
12240 ])
12241 .expect("matching rows");
12242 writer.append(&chunk).expect("one part");
12243 }
12244 writer.finish().expect("commit");
12245
12246 let read = |told: bool| {
12247 let reader = Reader::open(&path).expect("reopen from disk");
12248 assert_eq!(reader.table().stripes().len(), workers, "a stripe per worker");
12249 if told {
12250 reader.keep_stripes(workers);
12251 }
12252 let barrier = std::sync::Barrier::new(workers);
12253 std::thread::scope(|scope| {
12254 for (worker, run) in reader.stripe_parts().into_iter().enumerate() {
12255 let reader = &reader;
12256 let barrier = &barrier;
12257 scope.spawn(move || {
12258 for part in run {
12259 barrier.wait();
12260 let chunk = reader.read(part, &[0]).expect("a part of my own stripe");
12261 assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
12262 }
12263 assert!(worker < workers);
12264 });
12265 }
12266 });
12267 reader.pages.load(Atomic::Relaxed)
12268 };
12269
12270 assert_eq!(read(true), workers, "one page read per stripe and no more");
12271 assert!(read(false) > workers, "a cache that small is read again on every part");
12272 fs::remove_file(path).expect("remove scratch file");
12273 }
12274
12275 #[test]
12280 fn a_damaged_index_page_is_an_error() {
12281 let path = path("damaged-index");
12282 let mut writer =
12283 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
12284 .expect("new file");
12285 writer.append(&sample_ids()).expect("first part");
12286 writer.append(&sample_ids()).expect("second part");
12287 writer.finish().expect("commit");
12288
12289 let reader = Reader::open(&path).expect("valid directory");
12290 let index = reader.table.stripes[0].index;
12291 let mut byte = [0; 1];
12292 read_at(&reader.file, index.offset, &mut byte).expect("the first part length");
12293 let mut file = OpenOptions::new().write(true).open(&path).expect("open index page");
12294 file.seek(SeekFrom::Start(index.offset)).expect("index start");
12295 file.write_all(&[!byte[0]]).expect("damage the first part length");
12296 let error = reader.read(1, &[0]).expect_err("a damaged index must not be used");
12297 assert!(error.message().contains("index page section checksum differs"), "{error}");
12298 fs::remove_file(path).expect("remove scratch file");
12299 }
12300
12301 #[test]
12308 fn every_integer_width_round_trips_through_a_page() {
12309 let path = path("integer-widths");
12310 let columns = [
12311 (LogicalType::TinyInt, vec![Value::TinyInt(i8::MIN), Value::TinyInt(i8::MAX)]),
12312 (LogicalType::UTinyInt, vec![Value::UTinyInt(0), Value::UTinyInt(u8::MAX)]),
12313 (LogicalType::SmallInt, vec![Value::SmallInt(i16::MIN), Value::SmallInt(i16::MAX)]),
12314 (LogicalType::USmallInt, vec![Value::USmallInt(0), Value::USmallInt(u16::MAX)]),
12315 (LogicalType::Integer, vec![Value::Integer(i32::MIN), Value::Integer(i32::MAX)]),
12316 (LogicalType::UInteger, vec![Value::UInteger(0), Value::UInteger(u32::MAX)]),
12317 (LogicalType::BigInt, vec![Value::BigInt(i64::MIN), Value::BigInt(i64::MAX)]),
12318 (LogicalType::UBigInt, vec![Value::UBigInt(0), Value::UBigInt(u64::MAX)]),
12319 ];
12320 let fields = columns
12321 .iter()
12322 .enumerate()
12323 .map(|(at, (ty, _))| Field::required(format!("c{at}"), ty.clone()))
12324 .collect::<Vec<_>>();
12325 let vectors = columns
12326 .iter()
12327 .map(|(ty, values)| Vector::from_values(ty.clone(), values).expect("a vector"))
12328 .collect::<Vec<_>>();
12329 let mut writer = Writer::create(&path, "widths", fields).expect("new file");
12330 writer.append(&Chunk::new(vectors).expect("matching rows")).expect("one stripe");
12331 writer.finish().expect("commit");
12332
12333 let reader = Reader::open(&path).expect("reopen from disk");
12334 let wanted = (0..columns.len()).collect::<Vec<_>>();
12335 let read = reader.read(0, &wanted).expect("every column");
12336 assert_eq!(read.len(), 2);
12337 for (at, (ty, values)) in columns.iter().enumerate() {
12339 assert_eq!(read.value_at(0, at), values[0], "the low end of {ty}");
12340 assert_eq!(read.value_at(1, at), values[1], "the high end of {ty}");
12341 }
12342 fs::remove_file(path).expect("remove scratch file");
12343 }
12344
12345 #[test]
12356 fn every_other_type_the_format_knows_round_trips_through_a_page() {
12357 let path = path("other-types");
12358 let columns = [
12359 (LogicalType::Float, vec![Value::Float(f32::MIN), Value::Float(-0.0)]),
12360 (LogicalType::Double, vec![Value::Double(f64::MIN), Value::Double(f64::MAX)]),
12361 (LogicalType::HugeInt, vec![Value::HugeInt(i128::MIN), Value::HugeInt(i128::MAX)]),
12362 (LogicalType::UHugeInt, vec![Value::UHugeInt(0), Value::UHugeInt(u128::MAX)]),
12363 (LogicalType::Time, vec![Value::Time(0), Value::Time(86_399_999_999)]),
12364 (LogicalType::TimeTz, vec![Value::TimeTz(-50_400_000_000), Value::TimeTz(0)]),
12365 (
12366 LogicalType::TimestampTz,
12367 vec![Value::TimestampTz(i64::MIN + 1), Value::TimestampTz(i64::MAX)],
12368 ),
12369 (
12370 LogicalType::Interval,
12371 vec![
12372 Value::Interval { months: i32::MIN, days: i32::MAX, micros: i64::MIN },
12373 Value::Interval { months: 13, days: -1, micros: 1 },
12374 ],
12375 ),
12376 (
12377 LogicalType::Blob,
12378 vec![Value::Blob(vec![0, 0xff, 0x80, 0xfe]), Value::Blob(Vec::new())],
12379 ),
12380 ];
12381 let fields = columns
12382 .iter()
12383 .enumerate()
12384 .map(|(at, (ty, _))| Field::required(format!("c{at}"), ty.clone()))
12385 .collect::<Vec<_>>();
12386 let vectors = columns
12387 .iter()
12388 .map(|(ty, values)| Vector::from_values(ty.clone(), values).expect("a vector"))
12389 .collect::<Vec<_>>();
12390 let mut writer = Writer::create(&path, "others", fields).expect("new file");
12391 writer.append(&Chunk::new(vectors).expect("matching rows")).expect("one stripe");
12392 writer.finish().expect("commit");
12393
12394 let reader = Reader::open(&path).expect("reopen from disk");
12395 let wanted = (0..columns.len()).collect::<Vec<_>>();
12396 let read = reader.read(0, &wanted).expect("every column");
12397 assert_eq!(read.len(), 2);
12398 for (at, (ty, values)) in columns.iter().enumerate() {
12399 assert_eq!(read.value_at(0, at), values[0], "the low end of {ty}");
12400 assert_eq!(read.value_at(1, at), values[1], "the high end of {ty}");
12401 }
12402 let Value::Float(zero) = read.value_at(1, 0) else { panic!("a float stays a float") };
12405 assert!(zero.is_sign_negative(), "a negative zero came back as {zero}");
12406
12407 fs::remove_file(path).expect("remove scratch file");
12408 }
12409
12410 #[test]
12416 fn a_nan_survives_being_written_down() {
12417 let path = path("nan");
12418 let nan = Vector::from_values(LogicalType::Double, &[Value::Double(f64::NAN)])
12419 .expect("a NaN vector");
12420 let mut writer =
12421 Writer::create(&path, "nan", vec![Field::required("d", LogicalType::Double)])
12422 .expect("new file");
12423 writer.append(&Chunk::new(vec![nan]).expect("one column")).expect("one stripe");
12424 writer.finish().expect("commit");
12425 let read = Reader::open(&path).expect("reopen").read(0, &[0]).expect("the column");
12426 let Value::Double(back) = read.value_at(0, 0) else { panic!("a double stays a double") };
12427 assert!(back.is_nan(), "a NaN came back as {back}");
12428 fs::remove_file(path).expect("remove scratch file");
12429 }
12430
12431 #[test]
12438 fn a_uuid_and_a_bit_string_come_back_as_the_bits_that_went_in() {
12439 let path = path("uuid-and-bit");
12440 let uuids = vec![0_i128, i128::MIN, -1];
12441 let mut bits = StringColumn::new();
12442 for value in [&b"\x02\xff"[..], &b""[..], &b"\x00\x01\x02\x03\x04\x05"[..]] {
12443 bits.push_bytes(value);
12444 }
12445 let expected = bits.clone();
12446 let fields =
12447 vec![Field::required("u", LogicalType::Uuid), Field::required("b", LogicalType::Bit)];
12448 let vectors = vec![
12449 Vector::flat(LogicalType::Uuid, Data::Int128(uuids.clone().into())).expect("uuids"),
12450 Vector::flat(LogicalType::Bit, Data::Varlen(bits)).expect("bit strings"),
12451 ];
12452 let mut writer = Writer::create(&path, "ids", fields).expect("new file");
12453 writer.append(&Chunk::new(vectors).expect("matching rows")).expect("one stripe");
12454 writer.finish().expect("commit");
12455
12456 let reader = Reader::open(&path).expect("reopen from disk");
12457 let read = reader.read(0, &[0, 1]).expect("both columns").flatten().expect("flat");
12458 let Some(Data::Int128(back)) = read.column(0).expect("the uuids").data() else {
12459 panic!("a uuid column is the 128 bit lane")
12460 };
12461 assert_eq!(back.as_slice(), uuids.as_slice());
12462 let Some(Data::Varlen(back)) = read.column(1).expect("the bits").data() else {
12463 panic!("a bit column is bytes")
12464 };
12465 for row in 0..expected.len() {
12466 assert_eq!(back.bytes(row), expected.bytes(row), "row {row} of the bit column");
12467 }
12468 fs::remove_file(path).expect("remove scratch file");
12469 }
12470
12471 #[test]
12472 fn numeric_frequency_candidates_keep_bounded_row_ordinals() {
12473 let path = path("frequency-ordinals");
12474 let mut writer =
12475 Writer::create(&path, "items", vec![Field::required("id", LogicalType::BigInt)])
12476 .expect("new file");
12477 let mut values = Vec::new();
12478 for leader in 0..10_i64 {
12479 values.extend(std::iter::repeat_n(leader, 100));
12480 }
12481 values.extend(1_000_i64..41_000);
12482 for part in values.chunks(1_024) {
12483 let vector = Vector::flat(LogicalType::BigInt, Data::Int64(part.to_vec().into()))
12484 .expect("big integers");
12485 writer.append(&Chunk::new(vec![vector]).expect("one column")).expect("one stripe");
12486 }
12487 writer.finish().expect("commit");
12488
12489 let reader = Reader::open(&path).expect("reopen from disk");
12490 let occurrences =
12491 reader.frequency_occurrences(0).expect("valid metadata").expect("bounded ordinals");
12492 assert!(occurrences.omitted_max < 100);
12493 assert!(occurrences.ordinals.len() <= FREQUENCY_ORDINALS);
12494 assert_eq!(occurrences.anchor_indices.len(), occurrences.ordinals.len());
12495 assert!(occurrences.ordinals.windows(2).all(|pair| pair[0] < pair[1]));
12496 assert_eq!(&occurrences.ordinals[..1_000], &(0_u64..1_000).collect::<Vec<_>>());
12497 assert_eq!(
12498 &occurrences.anchor_indices[..1_000]
12499 .iter()
12500 .map(|&entry| occurrences.anchors[entry as usize].clone())
12501 .collect::<Vec<_>>(),
12502 &(0_i64..10)
12503 .flat_map(|leader| std::iter::repeat_n(Value::BigInt(leader), 100))
12504 .collect::<Vec<_>>()
12505 );
12506 fs::remove_file(path).expect("remove scratch file");
12507 }
12508
12509 #[test]
12510 fn numeric_frequencies_count_nulls_and_values_past_the_top_of_bigint() {
12511 let path = path("frequency-bits");
12516 let mut writer = Writer::create(
12517 &path,
12518 "items",
12519 vec![Field::new("u", LogicalType::UBigInt), Field::new("s", LogicalType::BigInt)],
12520 )
12521 .expect("new file");
12522 let mut rows = Vec::new();
12523 let mut leaders = Vec::new();
12524 for leader in 0..10_u64 {
12525 let count = 300 - leader * 10;
12526 let (unsigned, signed) = if leader == 0 {
12527 (Value::Null, Value::Null)
12528 } else {
12529 (Value::UBigInt(u64::MAX - leader), Value::BigInt(-(leader as i64)))
12530 };
12531 rows.extend(std::iter::repeat_n((unsigned.clone(), signed.clone()), count as usize));
12532 leaders.push(((unsigned, count), (signed, count)));
12533 }
12534 rows.extend((1_000..41_000_u64).map(|id| (Value::UBigInt(id), Value::BigInt(id as i64))));
12535 for part in rows.chunks(1_024) {
12536 let unsigned = part.iter().map(|(value, _)| value.clone()).collect::<Vec<_>>();
12537 let signed = part.iter().map(|(_, value)| value.clone()).collect::<Vec<_>>();
12538 let chunk = Chunk::new(vec![
12539 Vector::from_values(LogicalType::UBigInt, &unsigned).expect("unsigned"),
12540 Vector::from_values(LogicalType::BigInt, &signed).expect("signed"),
12541 ])
12542 .expect("matching columns");
12543 writer.append(&chunk).expect("rows");
12544 }
12545 writer.finish().expect("commit");
12546
12547 let reader = Reader::open(&path).expect("reopen from disk");
12548 for column in 0..2 {
12549 let prefix =
12550 reader.frequency_prefix(column).expect("valid metadata").expect("a synopsis");
12551 let wanted = leaders
12552 .iter()
12553 .map(|(unsigned, signed)| if column == 0 { unsigned } else { signed })
12554 .cloned()
12555 .collect::<Vec<_>>();
12556 assert_eq!(&prefix.entries[..10], &wanted[..], "column {column}");
12557 assert!(prefix.omitted_max < 210, "column {column}");
12558 assert_eq!(
12559 reader.distinct_values(column).expect("valid metadata"),
12560 Some(9 + 40_000),
12561 "column {column}"
12562 );
12563 }
12564 fs::remove_file(path).expect("remove scratch file");
12565 }
12566
12567 #[test]
12568 fn narrow_nonzero_count_matches_the_full_reader_across_stripes() {
12569 let path = path("quick-nonzero");
12570 let mut writer = Writer::create(
12571 &path,
12572 "items",
12573 vec![Field::new("label", LogicalType::Varchar), Field::new("id", LogicalType::Integer)],
12574 )
12575 .expect("create");
12576 for ids in [
12577 &[Value::Integer(0), Value::Null, Value::Integer(3)][..],
12578 &[Value::Integer(0), Value::Integer(7), Value::Null][..],
12579 ] {
12580 let labels = vec![Value::Varchar("same".into()); ids.len()];
12581 writer
12582 .append(
12583 &Chunk::new(vec![
12584 Vector::from_values(LogicalType::Varchar, &labels).expect("labels"),
12585 Vector::from_values(LogicalType::Integer, ids).expect("ids"),
12586 ])
12587 .expect("chunk"),
12588 )
12589 .expect("append");
12590 }
12591 writer.finish().expect("finish");
12592 let catalog = Catalog::open(&path).expect("catalog");
12593 assert_eq!(catalog.entries[0].nonzero, vec![None, Some(2)]);
12594 assert_eq!(catalog.entries[0].aggregates, vec![None, Some((10, 4))]);
12595 assert_eq!(
12596 catalog.aggregate_sums("items", &[1]).expect("catalog sums"),
12597 Some(CertifiedSums { columns: vec![(10, 4)], rows: 6 })
12598 );
12599 assert_eq!(catalog.nonzero_count("items", 1).expect("quick count"), Some(2));
12600 assert_eq!(
12601 reader_nonzero_counts(&catalog.table("items").expect("reader")).expect("counts"),
12602 vec![None, Some(2)]
12603 );
12604 Writer::certify_counts(&path).expect("recertify");
12605 assert_eq!(
12606 Catalog::open(&path).expect("reopen").nonzero_count("items", 1).expect("count"),
12607 Some(2)
12608 );
12609 assert_eq!(
12610 Catalog::open(&path).expect("reopen").aggregate_sums("items", &[1]).expect("sums"),
12611 Some(CertifiedSums { columns: vec![(10, 4)], rows: 6 })
12612 );
12613 assert_eq!(catalog.table("items").expect("reader").null_count(1).expect("nulls"), 2);
12614 fs::remove_file(path).expect("remove scratch file");
12615 }
12616
12617 #[test]
12618 fn numeric_string_pair_leaders_are_certified_in_the_directory() {
12619 let path = path("pair-frequencies");
12620 let mut pairs = Vec::new();
12621 pairs.extend(std::iter::repeat_n((1_i64, "alpha".to_string()), 100));
12622 pairs.extend(std::iter::repeat_n((1_i64, "beta".to_string()), 50));
12623 pairs.extend(std::iter::repeat_n((2_i64, "gamma".to_string()), 40));
12624 pairs.extend((1_000_i64..1_600).map(|id| (id, format!("tail {id}"))));
12625 let mut writer = Writer::create(
12626 &path,
12627 "items",
12628 vec![
12629 Field::required("id", LogicalType::BigInt),
12630 Field::required("phrase", LogicalType::Varchar),
12631 ],
12632 )
12633 .expect("new file");
12634 for part in pairs.chunks(1_024) {
12635 let ids = part.iter().map(|(id, _)| Value::BigInt(*id)).collect::<Vec<_>>();
12636 let phrases =
12637 part.iter().map(|(_, phrase)| Value::Varchar(phrase.clone())).collect::<Vec<_>>();
12638 writer
12639 .append(
12640 &Chunk::new(vec![
12641 Vector::from_values(LogicalType::BigInt, &ids).expect("ids"),
12642 Vector::from_values(LogicalType::Varchar, &phrases).expect("phrases"),
12643 ])
12644 .expect("matching columns"),
12645 )
12646 .expect("rows");
12647 }
12648 writer.finish().expect("commit");
12649
12650 let reader = Reader::open(&path).expect("reopen from disk");
12651 let leaders = reader
12652 .top_pair_frequencies(0, 1, 2)
12653 .expect("valid pair metadata")
12654 .expect("the top two beat the omitted tail");
12655 assert!(
12656 leaders.contains(&(vec![Value::BigInt(1), Value::Varchar("alpha".to_string())], 100,))
12657 );
12658 assert!(
12659 leaders.contains(&(vec![Value::BigInt(1), Value::Varchar("beta".to_string())], 50,))
12660 );
12661 fs::remove_file(path).expect("remove scratch file");
12662 }
12663
12664 #[test]
12670 fn a_file_from_another_format_says_which_format_it_is() {
12671 let older = path("older-format");
12672 let mut writer =
12673 Writer::create(&older, "items", vec![Field::new("id", LogicalType::Integer)])
12674 .expect("new file");
12675 let chunk = Chunk::new(vec![
12676 Vector::flat(LogicalType::Integer, Data::Int32(vec![1, 2, 3].into()))
12677 .expect("integers"),
12678 ])
12679 .expect("chunk");
12680 writer.append(&chunk).expect("page written");
12681 writer.finish().expect("commit");
12682
12683 let unreadable =
12687 READABLE.iter().copied().min().expect("at least one format is readable") - 1;
12688 let mut file = OpenOptions::new().write(true).open(&older).expect("open for the header");
12689 file.seek(SeekFrom::Start(8)).expect("the version follows the magic");
12690 file.write_all(&unreadable.to_le_bytes()).expect("write an older version");
12691 drop(file);
12692 let complaint = Reader::open(&older).expect_err("an older format is refused").to_string();
12693 assert!(complaint.contains(&format!("format {unreadable}")), "{complaint}");
12694 assert!(complaint.contains(&format!("format {FORMAT}")), "{complaint}");
12695
12696 let mut file = OpenOptions::new().write(true).open(&older).expect("open for the header");
12697 file.seek(SeekFrom::Start(0)).expect("the magic is first");
12698 file.write_all(b"NOTRUDB!").expect("write another engine's magic");
12699 drop(file);
12700 let complaint = Reader::open(&older).expect_err("a foreign file is refused").to_string();
12701 assert!(complaint.contains("magic"), "{complaint}");
12702 assert!(!complaint.contains("format"), "a version has nothing to do with it: {complaint}");
12703 fs::remove_file(older).expect("remove scratch file");
12704 }
12705
12706 #[test]
12707 fn an_unfinished_or_damaged_file_does_not_answer_with_partial_rows() {
12708 let unfinished = path("unfinished");
12709 let mut writer =
12710 Writer::create(&unfinished, "items", vec![Field::new("id", LogicalType::Integer)])
12711 .expect("new file");
12712 let chunk = Chunk::new(vec![
12713 Vector::flat(LogicalType::Integer, Data::Int32(vec![1, 2, 3].into()))
12714 .expect("integers"),
12715 ])
12716 .expect("chunk");
12717 writer.append(&chunk).expect("page written");
12718 drop(writer);
12719 assert!(Reader::open(&unfinished).is_err(), "no directory was committed");
12720 fs::remove_file(unfinished).expect("remove scratch file");
12721
12722 let damaged = path("damaged");
12723 let mut writer =
12724 Writer::create(&damaged, "items", vec![Field::new("id", LogicalType::Integer)])
12725 .expect("new file");
12726 writer.append(&chunk).expect("page written");
12727 writer.finish().expect("commit");
12728 let reader = Reader::open(&damaged).expect("valid directory");
12729 let mut file =
12730 OpenOptions::new().write(true).open(&damaged).expect("open for a damaged page");
12731 file.seek(SeekFrom::Start(HEADER + 1)).expect("inside first page");
12732 file.write_all(&[255]).expect("damage one byte");
12733 assert!(reader.read(0, &[0]).is_err(), "page checksum rejects corruption");
12734 fs::remove_file(damaged).expect("remove scratch file");
12735 }
12736
12737 #[test]
12738 fn damaged_lazy_dictionary_payload_is_an_error() {
12739 let path = path("damaged-dictionary");
12740 let mut writer = Writer::create(
12741 &path,
12742 "items",
12743 vec![
12744 Field::required("id", LogicalType::Integer),
12745 Field::new("text", LogicalType::Varchar),
12746 ],
12747 )
12748 .expect("new file");
12749 writer.append(&sample()).expect("stripe written");
12750 writer.finish().expect("commit");
12751
12752 let reader = Reader::open(&path).expect("valid directory");
12753 let dictionary = reader.table.dictionaries[1].expect("string dictionary page");
12754 let mut header = [0; DICTIONARY_HEADER];
12757 read_at(&reader.file, dictionary.offset, &mut header).expect("dictionary header");
12758 let count = u32::from_le_bytes(header[0..4].try_into().expect("four bytes")) as usize;
12761 let width = u32::from_le_bytes(header[12..16].try_into().expect("four bytes"));
12762 assert_ne!(width & DICTIONARY_SCATTERED, 0, "the blocks say where they are");
12763 let bits = (width & !(DICTIONARY_SCATTERED | DICTIONARY_GRAMS)) as usize;
12764 let mut start = [0; 8];
12765 let at = dictionary.offset + (DICTIONARY_HEADER + offset_bytes(count, bits)) as u64;
12766 read_at(&reader.file, at, &mut start).expect("the first block's start");
12767 let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
12768 file.seek(SeekFrom::Start(u64::from_le_bytes(start))).expect("inside dictionary payload");
12769 file.write_all(&[255]).expect("damage dictionary payload");
12770
12771 let chunk = reader.read(0, &[1]).expect("code page and dictionary index remain valid");
12772 let error =
12773 chunk.validate_external().expect_err("payload corruption must reach the caller");
12774 assert!(error.message().contains("payload checksum differs"), "{error}");
12775 fs::remove_file(path).expect("remove scratch file");
12776 }
12777
12778 #[test]
12788 fn a_column_of_all_different_values_is_written_without_a_dictionary() {
12789 let path = path("dictionary-decide");
12790 let rows = 20_000;
12791 let unique =
12793 |row: usize| format!("{row:09} a value that appears exactly once in the table");
12794 let repeated = |row: usize| unique(row / 40);
12796 let mut writer = Writer::create(
12797 &path,
12798 "items",
12799 vec![
12800 Field::required("unique", LogicalType::Varchar),
12801 Field::required("repeated", LogicalType::Varchar),
12802 ],
12803 )
12804 .expect("new file");
12805 for part in (0..rows).step_by(1_000) {
12806 let span = part..(part + 1_000).min(rows);
12807 let left = span.clone().map(|row| Value::Varchar(unique(row))).collect::<Vec<_>>();
12808 let right = span.map(|row| Value::Varchar(repeated(row))).collect::<Vec<_>>();
12809 writer
12810 .append(
12811 &Chunk::new(vec![
12812 Vector::from_values(LogicalType::Varchar, &left).expect("strings"),
12813 Vector::from_values(LogicalType::Varchar, &right).expect("strings"),
12814 ])
12815 .expect("two columns"),
12816 )
12817 .expect("a part");
12818 }
12819 writer.finish().expect("commit");
12820
12821 let reader = Reader::open(&path).expect("reopen from disk");
12822 assert!(
12823 reader.table.dictionaries[0].is_none(),
12824 "a column with no repeats has nothing to say twice"
12825 );
12826 assert!(
12827 reader.table.dictionaries[1].is_some(),
12828 "a column whose values come round again keeps its dictionary"
12829 );
12830 let mut first = 0;
12831 for part in 0..reader.parts() {
12832 let chunk = reader.read(part, &[0, 1]).expect("a part");
12833 for row in 0..chunk.len() {
12834 assert_eq!(chunk.value_at(row, 0), Value::Varchar(unique(first + row)));
12835 assert_eq!(chunk.value_at(row, 1), Value::Varchar(repeated(first + row)));
12836 }
12837 first += chunk.len();
12838 }
12839 assert_eq!(first, rows, "every row was read back");
12840 let raw = (0..rows).map(|row| unique(row).len()).sum::<usize>();
12841 let size = fs::metadata(&path).expect("the file is there").len() as usize;
12842 assert!(size < raw, "a column without a dictionary is still encoded: {size} against {raw}");
12843 fs::remove_file(path).expect("remove scratch file");
12844 }
12845
12846 #[test]
12859 fn a_dictionary_over_many_blocks_checks_every_block_of_it() {
12860 let path = path("dictionary-blocks");
12861 let value = |row: usize| {
12862 let row = row.saturating_sub(8_000);
12863 format!("{row:07} a value long enough to be worth a payload block")
12864 };
12865 let parts = 40;
12866 let per_part = 1000;
12867 let mut writer =
12868 Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
12869 .expect("new file");
12870 for part in 0..parts {
12871 let values = (0..per_part)
12872 .map(|row| Value::Varchar(value(part * per_part + row)))
12873 .collect::<Vec<_>>();
12874 let chunk = Chunk::new(vec![
12875 Vector::from_values(LogicalType::Varchar, &values).expect("strings"),
12876 ])
12877 .expect("matching rows");
12878 writer.append(&chunk).expect("a part");
12879 }
12880 writer.finish().expect("commit");
12881
12882 let reader = Reader::open(&path).expect("reopen from disk");
12883 let dictionary = reader.table.dictionaries[0].expect("string dictionary page");
12884 assert!(
12885 parts * per_part > TEXT_PAYLOAD_VALUES * 4,
12886 "the dictionary has to be several blocks for this to be testing anything"
12887 );
12888 for part in [0, parts - 1] {
12889 let chunk = reader.read(part, &[0]).expect("a part");
12890 chunk.validate_external().expect("every payload block checks out");
12891 assert_eq!(chunk.value_at(0, 0), Value::Varchar(value(part * per_part)));
12892 }
12893
12894 let mut header = [0; DICTIONARY_HEADER];
12896 read_at(&reader.file, dictionary.offset, &mut header).expect("dictionary header");
12897 let count = u32::from_le_bytes(header[0..4].try_into().expect("four bytes")) as usize;
12898 let blocks = u32::from_le_bytes(header[8..12].try_into().expect("four bytes")) as usize;
12899 let width = u32::from_le_bytes(header[12..16].try_into().expect("four bytes"));
12900 let bits = (width & !(DICTIONARY_SCATTERED | DICTIONARY_GRAMS)) as usize;
12901 let mut place = [0; 16];
12902 let at = DICTIONARY_HEADER + offset_bytes(count, bits) + (blocks - 1) * 16;
12903 read_at(&reader.file, dictionary.offset + at as u64, &mut place).expect("its place");
12904 let start = u64::from_le_bytes(place[..8].try_into().expect("eight bytes"));
12905 let length = u64::from_le_bytes(place[8..].try_into().expect("eight bytes"));
12906 let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
12907 file.seek(SeekFrom::Start(start + length - 4)).expect("the last bytes of the last block");
12908 file.write_all(&[255]).expect("damage the last payload block");
12909 let reader = Reader::open(&path).expect("the directory and the index are untouched");
12910 let chunk = reader.read(parts - 1, &[0]).expect("the code page remains valid");
12911 let error = chunk.validate_external().expect_err("the damage must reach the caller");
12912 assert!(error.message().contains("payload checksum differs"), "{error}");
12913 fs::remove_file(path).expect("remove scratch file");
12914 }
12915
12916 #[test]
12930 fn values_of_different_lengths_read_back_out_of_packed_offsets() {
12931 let path = path("dictionary-offsets");
12932 let value = |row: usize| {
12933 let row = row % 5_000;
12934 if row % 511 == 3 { String::new() } else { "x".repeat(row % 97) + &format!("{row:05}") }
12935 };
12936 let rows = 6_000;
12937 let mut writer =
12938 Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
12939 .expect("new file");
12940 let values = (0..rows).map(|row| Value::Varchar(value(row))).collect::<Vec<_>>();
12941 for part in values.chunks(1_000) {
12942 let chunk =
12943 Chunk::new(vec![Vector::from_values(LogicalType::Varchar, part).expect("strings")])
12944 .expect("matching rows");
12945 writer.append(&chunk).expect("a part");
12946 }
12947 writer.finish().expect("commit");
12948
12949 let reader = Reader::open(&path).expect("reopen from disk");
12950 assert!(
12951 rows > TEXT_PAYLOAD_VALUES * 4,
12952 "the dictionary has to be several blocks for this to be testing anything"
12953 );
12954 for part in 0..rows / 1_000 {
12955 let chunk = reader.read(part, &[0]).expect("a part");
12956 for row in 0..1_000 {
12957 let row = part * 1_000 + row;
12958 assert_eq!(
12959 chunk.value_at(row % 1_000, 0),
12960 Value::Varchar(value(row)),
12961 "value {row}"
12962 );
12963 }
12964 }
12965 for _ in 0..2 {
12968 for part in 0..rows / 1_000 {
12969 let chunk = reader.read(part, &[0]).expect("a part");
12970 let mut lens = vec![0_i64; 1_000];
12971 let column = chunk.column(0).expect("one column");
12972 assert!(column.try_bytes_lens(&mut lens).expect("lengths"), "a stored column");
12973 for (row, &len) in lens.iter().enumerate() {
12974 let row = part * 1_000 + row;
12975 assert_eq!(len as usize, value(row).len(), "the length of value {row}");
12976 }
12977 }
12978 }
12979 fs::remove_file(path).expect("remove scratch file");
12980 }
12981
12982 #[test]
12984 fn lengths_restart_at_each_block_and_refuse_ends_that_go_backwards() {
12985 let mut ends: Vec<u32> = (1..=TEXT_PAYLOAD_VALUES as u32).map(|at| at * 2).collect();
12986 ends.extend([3, 3, 10]);
12987 let lens = lengths_of(&ends).expect("ordered ends");
12988 assert!(lens[..TEXT_PAYLOAD_VALUES].iter().all(|&len| len == 2));
12989 assert_eq!(&lens[TEXT_PAYLOAD_VALUES..], &[3, 0, 7]);
12990 ends.push(9);
12991 assert_eq!(lengths_of(&ends), None);
12992 }
12993
12994 #[test]
13006 fn a_global_dictionary_is_opened_once_however_many_workers_ask_at_once() {
13007 let path = path("dictionary-once");
13008 let parts = 8;
13009 let per_part = 500;
13010 let value =
13011 |row: usize| format!("{row:07} a value long enough to be worth a payload block");
13012 let mut writer =
13013 Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
13014 .expect("new file");
13015 for part in 0..parts {
13016 let values = (0..per_part)
13017 .map(|row| Value::Varchar(value(part * per_part + row)))
13018 .collect::<Vec<_>>();
13019 let chunk = Chunk::new(vec![
13020 Vector::from_values(LogicalType::Varchar, &values).expect("strings"),
13021 ])
13022 .expect("matching rows");
13023 writer.append(&chunk).expect("a part");
13024 }
13025 writer.finish().expect("commit");
13026
13027 let reader = Reader::open(&path).expect("reopen from disk");
13028 assert!(reader.table.dictionaries[0].is_some(), "the column has to have one to share");
13029 assert_eq!(reader.reads().dictionaries, 0, "opening the file does not open a dictionary");
13030
13031 let workers = 16;
13032 let gate = std::sync::Barrier::new(workers);
13033 std::thread::scope(|scope| {
13034 for worker in 0..workers {
13035 let reader = reader.clone();
13036 let gate = &gate;
13037 scope.spawn(move || {
13038 gate.wait();
13039 let chunk = reader.read(worker % parts, &[0]).expect("a part");
13040 assert_eq!(
13041 chunk.value_at(0, 0),
13042 Value::Varchar(value((worker % parts) * per_part))
13043 );
13044 });
13045 }
13046 });
13047
13048 assert_eq!(reader.reads().dictionaries, 1, "sixteen workers, one dictionary, one open");
13049 fs::remove_file(path).expect("remove scratch file");
13050 }
13051
13052 #[test]
13057 fn a_damaged_sorted_order_is_an_error() {
13058 let path = path("damaged-order");
13059 let mut writer = Writer::create(
13060 &path,
13061 "items",
13062 vec![
13063 Field::required("id", LogicalType::Integer),
13064 Field::new("text", LogicalType::Varchar),
13065 ],
13066 )
13067 .expect("new file");
13068 writer.append(&sample()).expect("stripe written");
13069 writer.finish().expect("commit");
13070
13071 let reader = Reader::open(&path).expect("valid directory");
13072 let page = reader.table.dictionaries[1].expect("string dictionary page");
13073 let mut header = [0; DICTIONARY_HEADER];
13074 read_at(&reader.file, page.offset, &mut header).expect("dictionary header");
13075 let index_len = dictionary_index_len(&header);
13076 let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
13077 file.seek(SeekFrom::Start(page.offset + index_len)).expect("the first head");
13078 file.write_all(&[255]).expect("damage the order");
13079
13080 let dictionary = reader.dictionary(1).expect("read").expect("a string column has one");
13081 let error = dictionary.compare_rank(0, b"anything").expect_err("a damaged order is caught");
13082 assert!(error.message().contains("rank checksum differs"), "{error}");
13083 fs::remove_file(path).expect("remove scratch file");
13084 }
13085
13086 #[test]
13090 fn a_global_dictionary_carries_the_sorted_order_of_its_values() {
13091 let spellings = ["overlong1z", "b", "", "overlong1a", "overlong", "ab", "a", "overlong1"];
13094 let path = path("dictionary-order");
13095 let mut writer =
13096 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
13097 .expect("new file");
13098 writer
13099 .append(
13100 &Chunk::new(vec![
13101 Vector::from_values(
13102 LogicalType::Varchar,
13103 &spellings.map(|text| Value::Varchar(text.into())),
13104 )
13105 .expect("strings"),
13106 ])
13107 .expect("one column"),
13108 )
13109 .expect("stripe written");
13110 writer.finish().expect("commit");
13111
13112 let reader = Reader::open(&path).expect("valid directory");
13113 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
13114 let count = dictionary.ranks().expect("a v10 file stores one");
13115 assert_eq!(count, spellings.len(), "every distinct value has a rank");
13116 let order = (0..count)
13117 .map(|rank| dictionary.code_at_rank(rank).expect("a code"))
13118 .collect::<Vec<_>>();
13119 let mut seen = order.clone();
13120 seen.sort_unstable();
13121 assert_eq!(seen, (0..spellings.len() as u32).collect::<Vec<_>>(), "a permutation of codes");
13122
13123 let ranked = order
13124 .iter()
13125 .map(|&code| {
13126 dictionary.try_bytes_at(code as usize).expect("read").expect("a value").to_vec()
13127 })
13128 .collect::<Vec<_>>();
13129 let mut expected = spellings.map(|text| text.as_bytes().to_vec()).to_vec();
13130 expected.sort();
13131 assert_eq!(ranked, expected, "rank order is value order");
13132
13133 for (rank, value) in expected.iter().enumerate() {
13136 assert_eq!(
13137 dictionary.compare_rank(rank, value).expect("compare"),
13138 Ordering::Equal,
13139 "rank {rank} is its own value"
13140 );
13141 if rank > 0 {
13142 assert_eq!(
13143 dictionary.compare_rank(rank - 1, value).expect("compare"),
13144 Ordering::Less,
13145 "rank {rank} follows the one before it"
13146 );
13147 }
13148 }
13149 fs::remove_file(path).expect("remove scratch file");
13150 }
13151
13152 #[test]
13160 fn a_large_dictionary_ranks_in_value_order() {
13161 let path = path("dictionary-large-rank");
13162 let value = |row: u64| {
13163 let mixed = row.wrapping_mul(0x9e37_79b9_7f4a_7c15) >> 40;
13164 match row % 3 {
13165 0 => format!("https://example.com/a/long/shared/path/{mixed:08}"),
13166 1 => format!("{mixed}"),
13167 _ => format!("x{}", row % 1000).repeat(1 + (row % 4) as usize) + &row.to_string(),
13168 }
13169 };
13170 let distinct = 70_000;
13171 let parts = 4 * distinct / 1000;
13172 let per_part = 1000;
13173 let mut writer =
13174 Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
13175 .expect("new file");
13176 for part in 0..parts {
13177 let values = (0..per_part)
13178 .map(|row| Value::Varchar(value((part * per_part + row) / 4)))
13179 .collect::<Vec<_>>();
13180 let chunk = Chunk::new(vec![
13181 Vector::from_values(LogicalType::Varchar, &values).expect("strings"),
13182 ])
13183 .expect("matching rows");
13184 writer.append(&chunk).expect("a part");
13185 }
13186 writer.finish().expect("commit");
13187
13188 let reader = Reader::open(&path).expect("reopen from disk");
13189 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
13190 let count = dictionary.ranks().expect("a ranked dictionary");
13191 assert_eq!(count, distinct as usize, "every distinct value has a rank");
13192 assert!(count >= PARALLEL_SORT_MIN, "too few values to be sorted on more than one thread");
13193 let ranked = (0..count)
13194 .map(|rank| {
13195 let code = dictionary.code_at_rank(rank).expect("a code");
13196 dictionary.try_bytes_at(code as usize).expect("read").expect("a value").to_vec()
13197 })
13198 .collect::<Vec<_>>();
13199 let mut expected = (0..distinct).map(|row| value(row).into_bytes()).collect::<Vec<_>>();
13200 expected.sort();
13201 assert_eq!(ranked, expected, "rank order is value order");
13202 fs::remove_file(path).expect("remove scratch file");
13203 }
13204
13205 #[test]
13218 fn a_directory_read_a_window_at_a_time_is_the_directory_read_whole() {
13219 let path = path("windowed-directory");
13220 let fields = vec![
13221 Field::required("id", LogicalType::BigInt),
13222 Field::required("word", LogicalType::Varchar),
13223 Field::new("score", LogicalType::Double),
13224 ];
13225 let mut writer = Writer::create(&path, "items", fields).expect("new file");
13226 for part in 0..70_i64 {
13227 let ids = (0..100).map(|row| Value::BigInt(part * 100 + row % 7)).collect::<Vec<_>>();
13228 let words = (0..100)
13229 .map(|row| Value::Varchar(format!("word {}", row % 13)))
13230 .collect::<Vec<_>>();
13231 let scores = (0..100)
13232 .map(|row| if row % 4 == 0 { Value::Null } else { Value::Double(row as f64) })
13233 .collect::<Vec<_>>();
13234 let chunk = Chunk::new(vec![
13235 Vector::from_values(LogicalType::BigInt, &ids).expect("integers"),
13236 Vector::from_values(LogicalType::Varchar, &words).expect("strings"),
13237 Vector::from_values(LogicalType::Double, &scores).expect("doubles"),
13238 ])
13239 .expect("three columns");
13240 writer.append(&chunk).expect("a part");
13241 }
13242 writer.finish().expect("commit");
13243
13244 let catalog = Catalog::open(&path).expect("reopen");
13245 let entry = catalog.entries.first().expect("one table").directory;
13246 let (offset, length) = (entry.offset, entry.length as usize);
13247 let mut bytes = vec![0; length];
13248 read_at(&catalog.file, offset, &mut bytes).expect("the directory");
13249 assert_eq!(file_checksum(&catalog.file, offset, length).expect("checksum"), entry.hash);
13250 let whole = decode_directory(&bytes, catalog.size).expect("whole");
13251 assert!(whole.stripes.len() > 1, "the table should span stripes");
13252 for size in [1, 7, 33, 4_096] {
13253 let mut cursor = Cursor::over(&catalog.file, offset, length);
13254 cursor.window.as_mut().expect("a window").size = size;
13255 let windowed = read_directory(cursor, catalog.size, Some(offset)).expect("windowed");
13256 assert_eq!(format!("{:?}", windowed.stripes), format!("{:?}", whole.stripes));
13257 assert_eq!(format!("{:?}", windowed.fields), format!("{:?}", whole.fields));
13258 let mut stored = 0;
13259 for (column, (left, held)) in
13260 windowed.frequencies.iter().zip(&whole.frequencies).enumerate()
13261 {
13262 match (left, held) {
13263 (None, None) => {}
13264 (
13265 Some(super::Frequencies::Stored { span, values }),
13266 Some(super::Frequencies::Held(summary)),
13267 ) => {
13268 let mut one = vec![0; span.length as usize];
13269 read_at(&catalog.file, span.offset, &mut one).expect("a synopsis");
13270 let read = decode_summary(
13271 &mut Cursor::new(&one),
13272 &whole.fields[column],
13273 whole.rows,
13274 *values,
13275 )
13276 .expect("a valid synopsis")
13277 .expect("one is there");
13278 assert_eq!(format!("{read:?}"), format!("{summary:?}"));
13279 stored += 1;
13280 }
13281 other => panic!("column {column} came back as {other:?}"),
13282 }
13283 }
13284 assert!(stored >= 2, "only {stored} synopses were left in the file");
13285 }
13286 let reader = catalog.table("items").expect("the table");
13287 assert!(reader.frequency_summaries[1].get().is_none());
13288 assert!(reader.top_frequencies(1, 1).expect("a readable synopsis").is_some());
13289 let first = reader.frequency_summaries[1].get().expect("decoded synopsis");
13290 let clone = reader.clone();
13291 assert!(clone.top_frequencies(1, 1).expect("cached synopsis").is_some());
13292 assert!(Arc::ptr_eq(first, clone.frequency_summaries[1].get().expect("same synopsis")));
13293 fs::remove_file(path).expect("remove scratch file");
13294 }
13295
13296 #[test]
13297 fn a_checksum_carried_across_reads_is_the_checksum_of_the_whole() {
13298 let path = path("file-checksum");
13299 let bytes = (0..200_000_u32)
13300 .map(|at| (at.wrapping_mul(2_654_435_761) >> 13) as u8)
13301 .collect::<Vec<_>>();
13302 fs::write(&path, &bytes).expect("scratch file");
13303 let file = File::open(&path).expect("open");
13304 for (offset, length) in [
13305 (0, 0),
13306 (3, 1),
13307 (5, 31),
13308 (0, 32),
13309 (9, 33),
13310 (1, 65_536),
13311 (7, 65_567),
13312 (0, 200_000),
13313 (11, 131_101),
13314 ] {
13315 let whole = checksum(&bytes[offset..offset + length]);
13316 assert_eq!(
13317 file_checksum(&file, offset as u64, length).expect("read"),
13318 whole,
13319 "{offset} {length}"
13320 );
13321 }
13322 fs::remove_file(path).expect("remove scratch file");
13323 }
13324
13325 #[test]
13326 fn a_string_synopsis_is_read_without_keeping_the_dictionary_blocks() {
13327 let path = path("synopsis-keeps-no-block");
13328 let spelled = |index: usize| Value::Varchar(format!("phrase {index:05}"));
13329 let mut values = (0..3_000).map(spelled).collect::<Vec<_>>();
13330 for _ in 0..3 {
13331 values.extend((0..3_000).step_by(5).map(spelled));
13332 }
13333 let mut writer =
13334 Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
13335 .expect("new file");
13336 for part in values.chunks(1_024) {
13337 writer
13338 .append(
13339 &Chunk::new(vec![
13340 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
13341 ])
13342 .expect("one column"),
13343 )
13344 .expect("a part");
13345 }
13346 writer.finish().expect("commit");
13347
13348 let reader = Reader::open(&path).expect("reopen from disk");
13349 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
13350 let resting = dictionary.footprint();
13351 let prefix = reader.frequency_prefix(0).expect("a readable synopsis").expect("one");
13352 assert_eq!(prefix.entries.len(), 512);
13353 for (value, count) in &prefix.entries {
13354 let Value::Varchar(text) = value else { panic!("a string column gave {value:?}") };
13355 let index = text["phrase ".len()..].parse::<usize>().expect("a spelled number");
13356 assert_eq!((index % 5, *count), (0, 4), "{text} came back with {count}");
13357 }
13358 assert_eq!(dictionary.footprint(), resting, "reading the synopsis kept a decoded block");
13359 let again = reader.frequency_prefix(0).expect("a readable synopsis").expect("one");
13360 assert_eq!(again.entries, prefix.entries);
13361 fs::remove_file(path).expect("remove scratch file");
13362 }
13363
13364 #[test]
13374 fn a_dictionary_sweep_reads_every_value_and_keeps_it_under_the_budget() {
13375 let path = path("dictionary-sweep");
13376 let spellings = (0..2_500)
13379 .map(|index| Value::Varchar(format!("value {index:08} {}", "x".repeat(index % 40))))
13380 .collect::<Vec<_>>();
13381 let mut writer =
13382 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
13383 .expect("new file");
13384 for part in spellings.chunks(1_024) {
13387 writer
13388 .append(
13389 &Chunk::new(vec![
13390 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
13391 ])
13392 .expect("one column"),
13393 )
13394 .expect("stripe written");
13395 }
13396 writer.finish().expect("commit");
13397
13398 let reader = Reader::open(&path).expect("valid directory");
13399 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
13400 assert_eq!(dictionary.len(), spellings.len(), "every value is distinct");
13401 for first in [0, TEXT_PAYLOAD_VALUES, TEXT_PAYLOAD_VALUES * 2] {
13402 assert!(dictionary.text_block_might_contain(first, b"value").expect("signature"));
13403 assert!(!dictionary.text_block_might_contain(first, b"google").expect("signature"));
13404 }
13405
13406 let resting = dictionary.footprint();
13407 let sweep = || {
13408 let mut swept: Vec<Vec<u8>> = Vec::new();
13409 let mut at = 0;
13410 let mut calls = 0;
13411 while at < dictionary.len() {
13412 let stopped = dictionary
13413 .sweep_text(at, dictionary.len(), &mut |index: usize, text: &[u8]| {
13414 assert_eq!(index, swept.len(), "a sweep hands its values over in order");
13415 swept.push(text.to_vec());
13416 Ok(())
13417 })
13418 .expect("a sweep reads");
13419 assert!(stopped > at, "a sweep moves");
13420 at = stopped;
13421 calls += 1;
13422 }
13423 assert_eq!(calls, 3, "a sweep hands over one block at a time");
13424 swept
13425 };
13426 let swept = sweep();
13427 assert_eq!(dictionary.footprint(), resting, "a first sweep keeps nothing it decoded");
13428 assert_eq!(sweep(), swept, "a second sweep reads what the first did");
13429 let after = dictionary.footprint();
13430 assert!(after > resting, "a second sweep under the budget keeps what it decoded");
13431
13432 let read = (0..dictionary.len())
13433 .map(|code| dictionary.try_bytes_at(code).expect("read").expect("a value").to_vec())
13434 .collect::<Vec<_>>();
13435 assert_eq!(swept, read, "a sweep answers what a point read answers");
13436 let grown = dictionary.footprint() - after;
13440 assert!(
13441 grown == 0 || grown == dictionary.len() * size_of::<u32>(),
13442 "a point read of a kept block decodes nothing, and {grown} bytes grew"
13443 );
13444 fs::remove_file(path).expect("remove scratch file");
13445 }
13446
13447 #[test]
13448 fn a_damaged_substring_signature_is_checked_only_when_used() {
13449 let path = path("damaged-substring-signature");
13450 let mut writer =
13451 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
13452 .expect("new file");
13453 let rows = [Value::Varchar("google".into()), Value::Varchar("example".into())];
13454 writer
13455 .append(
13456 &Chunk::new(vec![
13457 Vector::from_values(LogicalType::Varchar, &rows).expect("strings"),
13458 ])
13459 .expect("one column"),
13460 )
13461 .expect("stripe written");
13462 writer.finish().expect("commit");
13463
13464 let reader = Reader::open(&path).expect("valid directory");
13465 let page = reader.table.dictionaries[0].expect("string dictionary page");
13466 let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
13467 file.seek(SeekFrom::Start(page.offset + u64::from(page.length) - 1))
13468 .expect("last signature byte");
13469 file.write_all(&[255]).expect("damage signature");
13470 let reader = Reader::open(&path).expect("the directory is still valid");
13471 let dictionary = reader.dictionary(0).expect("index is still valid").expect("dictionary");
13472 let error = dictionary
13473 .text_block_might_contain(0, b"goog")
13474 .expect_err("a used signature checks its own checksum");
13475 assert!(error.message().contains("substring signatures checksum differs"), "{error}");
13476 fs::remove_file(path).expect("remove scratch file");
13477 }
13478
13479 #[test]
13490 fn a_sweep_over_a_block_with_a_short_second_run_reads_what_a_point_read_reads() {
13491 let path = path("dictionary-sweep-short-run");
13492 let spellings = (0..2_800)
13493 .map(|index| Value::Varchar(format!("value {index:08} {}", "x".repeat(index % 40))))
13494 .collect::<Vec<_>>();
13495 let mut writer =
13496 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
13497 .expect("new file");
13498 for part in spellings.chunks(1_024) {
13499 writer
13500 .append(
13501 &Chunk::new(vec![
13502 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
13503 ])
13504 .expect("one column"),
13505 )
13506 .expect("stripe written");
13507 }
13508 writer.finish().expect("commit");
13509
13510 let reader = Reader::open(&path).expect("valid directory");
13511 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
13512 assert_eq!(dictionary.len(), spellings.len(), "every value is distinct");
13513 let last = dictionary.len() % TEXT_PAYLOAD_VALUES;
13514 assert!(last > TEXT_OFFSET_RUN, "the last block has to reach into a second run of offsets");
13515 assert!(last < TEXT_PAYLOAD_VALUES, "and that second run has to be short of a whole one");
13516
13517 let mut swept: Vec<Vec<u8>> = Vec::new();
13518 let mut at = 0;
13519 while at < dictionary.len() {
13520 let stopped = dictionary
13521 .sweep_text(at, dictionary.len(), &mut |index: usize, text: &[u8]| {
13522 assert_eq!(index, swept.len(), "a sweep hands its values over in order");
13523 swept.push(text.to_vec());
13524 Ok(())
13525 })
13526 .expect("a sweep reads");
13527 assert!(stopped > at, "a sweep moves");
13528 at = stopped;
13529 }
13530 let read = (0..dictionary.len())
13531 .map(|code| dictionary.try_bytes_at(code).expect("read").expect("a value").to_vec())
13532 .collect::<Vec<_>>();
13533 assert_eq!(swept, read, "a sweep answers what a point read answers");
13534 fs::remove_file(path).expect("remove scratch file");
13535 }
13536
13537 #[test]
13546 fn the_unpacked_ends_answer_what_the_packed_ends_answer() {
13547 let path = path("dictionary-unpacked-ends");
13548 let spellings = (0..2_800)
13549 .map(|index| Value::Varchar(format!("value {index:08} {}", "x".repeat(index % 40))))
13550 .collect::<Vec<_>>();
13551 let mut writer =
13552 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
13553 .expect("new file");
13554 for part in spellings.chunks(1_024) {
13555 writer
13556 .append(
13557 &Chunk::new(vec![
13558 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
13559 ])
13560 .expect("one column"),
13561 )
13562 .expect("stripe written");
13563 }
13564 writer.finish().expect("commit");
13565
13566 let reader = Reader::open(&path).expect("valid directory");
13567 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
13568 assert_eq!(dictionary.len(), spellings.len(), "every value is distinct");
13569 let wanted = (0..spellings.len())
13570 .map(|index| format!("value {index:08} {}", "x".repeat(index % 40)).into_bytes())
13571 .collect::<Vec<_>>();
13572
13573 let pass = |what: &str| {
13574 for (index, value) in wanted.iter().enumerate() {
13575 let len = dictionary.try_bytes_len_at(index).expect("read").expect("a value");
13576 assert_eq!(len, value.len(), "{what} has the wrong length at {index}");
13577 let bytes = dictionary.try_bytes_at(index).expect("read").expect("a value");
13578 assert_eq!(bytes, value.as_slice(), "{what} has the wrong value at {index}");
13579 }
13580 };
13581 pass("the first pass");
13582 pass("the second pass");
13583
13584 let lens = wanted.iter().map(|value| value.len() as i64).collect::<Vec<_>>();
13588 let mut whole = vec![0i64; wanted.len()];
13589 assert!(dictionary.try_bytes_lens(&mut whole).expect("read"), "the text answers whole");
13590 assert_eq!(whole, lens, "a vector of lengths answers what a length at a time answers");
13591 let codes = (0..4_000_u32).map(|row| (7 * (4_000 - row)) % 2_800).collect::<Vec<_>>();
13592 let coded = Vector::dictionary_over(codes.clone(), dictionary).expect("codes in range");
13593 let mut through = vec![0i64; codes.len()];
13594 assert!(coded.try_bytes_lens(&mut through).expect("read"), "the codes answer whole");
13595 for (row, &code) in codes.iter().enumerate() {
13596 assert_eq!(through[row], lens[code as usize], "row {row} reads code {code}");
13597 let one = coded.try_bytes_len_at(row).expect("read").expect("a value");
13598 assert_eq!(through[row], one as i64, "row {row} a row at a time");
13599 }
13600
13601 let fresh = Reader::open(&path).expect("valid directory");
13604 let untouched = fresh.dictionary(0).expect("read").expect("a string column has one");
13605 let few = vec![2_799_u32, 0, 1_024, 1_023, 511, 512];
13606 let coded = Vector::dictionary_over(few.clone(), untouched).expect("in range");
13607 let mut short = vec![0i64; few.len()];
13608 assert!(coded.try_bytes_lens(&mut short).expect("read"), "the codes answer whole");
13609 let expected = few.iter().map(|&code| lens[code as usize]).collect::<Vec<_>>();
13610 assert_eq!(short, expected, "the packed ends answer what the table answers");
13611 fs::remove_file(path).expect("remove scratch file");
13612 }
13613
13614 #[test]
13624 fn narrowing_a_page_takes_what_fits_and_refuses_what_does_not() {
13625 assert_eq!(fit::<i8>(&[]).expect("an empty page fits anything"), Vec::<i8>::new());
13626 assert_eq!(fit::<i8>(&[-128, 0, 127]).expect("the edges fit"), vec![-128_i8, 0, 127]);
13627 fit::<i8>(&[128]).expect_err("one past the top does not fit");
13628 fit::<i8>(&[-129]).expect_err("one past the bottom does not fit");
13629 assert_eq!(fit::<u8>(&[0, 255]).expect("the edges fit"), vec![0_u8, 255]);
13630 fit::<u8>(&[256]).expect_err("one past the top does not fit");
13631 fit::<u8>(&[-1]).expect_err("a negative does not fit an unsigned page");
13632 assert_eq!(
13633 fit::<i16>(&[-32_768, 0, 32_767]).expect("the edges fit"),
13634 vec![-32_768_i16, 0, 32_767]
13635 );
13636 fit::<i16>(&[32_768]).expect_err("one past the top does not fit");
13637 fit::<i16>(&[-32_769]).expect_err("one past the bottom does not fit");
13638 assert_eq!(fit::<u16>(&[0, 65_535]).expect("the edges fit"), vec![0_u16, 65_535]);
13639 fit::<u16>(&[65_536]).expect_err("one past the top does not fit");
13640 fit::<u16>(&[-1]).expect_err("a negative does not fit an unsigned page");
13641 assert_eq!(
13642 fit::<i32>(&[i64::from(i32::MIN), 0, i64::from(i32::MAX)]).expect("the edges fit"),
13643 vec![i32::MIN, 0, i32::MAX]
13644 );
13645 fit::<i32>(&[i64::from(i32::MAX) + 1]).expect_err("one past the top does not fit");
13646 fit::<i32>(&[i64::from(i32::MIN) - 1]).expect_err("one past the bottom does not fit");
13647 assert_eq!(
13648 fit::<u32>(&[0, 4_294_967_295]).expect("the edges fit"),
13649 vec![0_u32, 4_294_967_295]
13650 );
13651 fit::<u32>(&[4_294_967_296]).expect_err("one past the top does not fit");
13652 fit::<u32>(&[-1]).expect_err("a negative does not fit an unsigned page");
13653
13654 fit::<i8>(&[0, 1, 2, 128, 3]).expect_err("one bad value spoils the page");
13657 }
13658
13659 #[test]
13666 fn the_residue_agrees_with_a_checked_conversion_everywhere() {
13667 for value in -70_000_i64..70_000 {
13668 assert_eq!(fit::<i8>(&[value]).is_ok(), i8::try_from(value).is_ok(), "{value} as i8");
13669 assert_eq!(fit::<u8>(&[value]).is_ok(), u8::try_from(value).is_ok(), "{value} as u8");
13670 assert_eq!(fit::<i16>(&[value]).is_ok(), i16::try_from(value).is_ok(), "{value} i16");
13671 assert_eq!(fit::<u16>(&[value]).is_ok(), u16::try_from(value).is_ok(), "{value} u16");
13672 }
13673 let wide = [i64::MIN, i64::MIN + 1, i64::from(i32::MIN), 0, i64::from(u32::MAX), i64::MAX];
13674 for edge in wide {
13675 for step in -2_i64..=2 {
13676 let value = edge.saturating_add(step);
13677 assert_eq!(
13678 fit::<i32>(&[value]).is_ok(),
13679 i32::try_from(value).is_ok(),
13680 "{value} as i32"
13681 );
13682 assert_eq!(
13683 fit::<u32>(&[value]).is_ok(),
13684 u32::try_from(value).is_ok(),
13685 "{value} as u32"
13686 );
13687 }
13688 }
13689 }
13690
13691 #[test]
13706 fn a_dictionary_reads_the_same_whether_its_blocks_say_where_they_are() {
13707 let spellings = (0..3_000)
13708 .map(|index| format!("value {index:08} {}", "y".repeat(index % 40)))
13709 .collect::<Vec<_>>();
13710 let mut read = Vec::new();
13711 for layout in ["outside", "inside", "behind"] {
13712 let mut dictionary = GlobalDictionary::new();
13713 for text in &spellings {
13714 dictionary.code(text).expect("a code for every spelling");
13715 }
13716 dictionary.finish_blocks().expect("the last block encodes");
13717 let order = dictionary.ranked(None).expect("a sorted order");
13718 let laid = |from: u64| {
13720 let mut at = from;
13721 dictionary
13722 .blocks
13723 .iter()
13724 .map(|block| {
13725 let place =
13726 Placed { start: at, length: block.len() as u64, hash: checksum(block) };
13727 at += block.len() as u64;
13728 place
13729 })
13730 .collect::<Vec<_>>()
13731 };
13732 let payload = dictionary.blocks.concat();
13733 let scattered = layout != "behind";
13734 let (bytes, encoded, offset, length) = if layout == "outside" {
13735 let mut bytes = vec![0; HEADER as usize];
13736 bytes.extend_from_slice(&payload);
13737 let encoded = encode_global_dictionary(&dictionary, &order, &laid(HEADER), true)
13738 .expect("an encoding");
13739 let offset = bytes.len() as u64;
13740 bytes.extend_from_slice(&encoded.index);
13741 bytes.extend_from_slice(&encoded.ranks);
13742 bytes.extend_from_slice(&encoded.grams);
13743 let length = encoded.index.len() + encoded.ranks.len() + encoded.grams.len();
13744 (bytes, encoded, offset, length)
13745 } else {
13746 let first = encode_global_dictionary(&dictionary, &order, &laid(0), scattered)
13749 .expect("an encoding");
13750 let body = (first.index.len() + first.ranks.len() + first.grams.len()) as u64;
13751 let encoded = encode_global_dictionary(&dictionary, &order, &laid(body), scattered)
13752 .expect("an encoding");
13753 let mut bytes = encoded.index.clone();
13754 bytes.extend_from_slice(&encoded.ranks);
13755 bytes.extend_from_slice(&encoded.grams);
13756 bytes.extend_from_slice(&payload);
13757 let length = bytes.len();
13758 (bytes, encoded, 0, length)
13759 };
13760 let path = path(&format!("blocks-{layout}"));
13761 fs::write(&path, &bytes).expect("the dictionary is written on its own");
13762 let file = Arc::new(File::open(&path).expect("it opens again"));
13763 let page = Page {
13764 offset,
13765 length: u32::try_from(length).expect("a test dictionary is small"),
13766 hash: checksum(&encoded.index),
13767 };
13768 let opened =
13769 open_global_dictionary(file, page, &LogicalType::Varchar, TEXT_KEEP_BUDGET)
13770 .expect("a dictionary laid out either way opens");
13771 let mut swept: Vec<Vec<u8>> = Vec::new();
13772 let mut at = 0;
13773 while at < opened.len() {
13774 at = opened
13775 .sweep_text(at, opened.len(), &mut |_index: usize, text: &[u8]| {
13776 swept.push(text.to_vec());
13777 Ok(())
13778 })
13779 .expect("a sweep reads");
13780 }
13781 fs::remove_file(&path).expect("clean up");
13782 read.push(swept);
13783 }
13784 let wanted =
13785 spellings.iter().map(|text| text.as_bytes().to_vec()).collect::<Vec<Vec<u8>>>();
13786 assert_eq!(read[0], wanted, "the blocks outside the page hold the values");
13787 assert_eq!(read[1], read[0], "the blocks inside the page hold the same values");
13788 assert_eq!(read[2], read[0], "the blocks behind one another hold the same values");
13789 }
13790
13791 #[test]
13799 fn a_dictionary_at_its_budget_sweeps_without_keeping() {
13800 let path = path("dictionary-budget");
13801 let spellings = (0..2_500)
13802 .map(|index| Value::Varchar(format!("value {index:08} {}", "y".repeat(index % 40))))
13803 .collect::<Vec<_>>();
13804 let mut writer =
13805 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
13806 .expect("new file");
13807 for part in spellings.chunks(1_024) {
13808 writer
13809 .append(
13810 &Chunk::new(vec![
13811 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
13812 ])
13813 .expect("one column"),
13814 )
13815 .expect("stripe written");
13816 }
13817 writer.finish().expect("commit");
13818
13819 let reader = Reader::open(&path).expect("valid directory");
13820 let page = reader.table.dictionaries[0].expect("a string column has one");
13821 let file = Arc::clone(&reader.file);
13822 let starved = open_global_dictionary(file, page, &LogicalType::Varchar, 0)
13823 .expect("a dictionary opens whatever it may keep");
13824
13825 let resting = starved.footprint();
13826 let mut swept: Vec<Vec<u8>> = Vec::new();
13827 let mut at = 0;
13828 while at < starved.len() {
13829 at = starved
13830 .sweep_text(at, starved.len(), &mut |_index: usize, text: &[u8]| {
13831 swept.push(text.to_vec());
13832 Ok(())
13833 })
13834 .expect("a sweep reads");
13835 }
13836 assert_eq!(swept.len(), spellings.len(), "a starved sweep still reads every value");
13837 assert_eq!(starved.footprint(), resting, "and keeps no block it decoded");
13838
13839 let generous = reader.dictionary(0).expect("read").expect("a string column has one");
13840 let read = (0..generous.len())
13841 .map(|code| generous.try_bytes_at(code).expect("read").expect("a value").to_vec())
13842 .collect::<Vec<_>>();
13843 assert_eq!(swept, read, "a starved sweep answers what a point read answers");
13844 fs::remove_file(path).expect("remove scratch file");
13845 }
13846
13847 #[test]
13848 fn damaged_membership_cannot_skip_a_string_page() {
13849 let path = path("damaged-membership");
13850 let mut writer = Writer::create(
13851 &path,
13852 "items",
13853 vec![
13854 Field::required("id", LogicalType::Integer),
13855 Field::new("text", LogicalType::Varchar),
13856 ],
13857 )
13858 .expect("new file");
13859 writer.append(&sample()).expect("stripe written");
13860 writer.finish().expect("commit");
13861
13862 let reader = Reader::open(&path).expect("valid directory");
13863 let membership = reader.table.stripes[0].memberships.get(1).expect("string membership");
13864 let mut file = OpenOptions::new().write(true).open(&path).expect("open membership page");
13865 file.seek(SeekFrom::Start(membership.offset)).expect("membership start");
13866 file.write_all(&[255]).expect("damage membership");
13867 let error = reader.skips_codes(0, 1, &[3]).expect_err("corruption must not skip rows");
13868 assert!(error.message().contains("membership page checksum differs"), "{error}");
13869 fs::remove_file(path).expect("remove scratch file");
13870 }
13871
13872 #[test]
13873 fn membership_delta_stream_is_sorted_exact_and_bounded() {
13874 let unique = unique_codes(&[900, 4, 4, 72, 9, u32::MAX]);
13875 assert_eq!(unique, [4, 9, 72, 900, u32::MAX]);
13876 let encoded = encode_membership(&unique);
13877 assert_eq!(
13878 decode_membership(&encoded).expect("valid membership"),
13879 [4, 9, 72, 900, u32::MAX]
13880 );
13881 let merged = merged_codes(vec![vec![4, 900], vec![9, 900, u32::MAX], vec![72]]);
13884 assert_eq!(merged, [4, 9, 72, 900, u32::MAX]);
13885 assert_eq!(
13886 decode_membership(&encode_membership(&merged)).expect("valid membership"),
13887 unique
13888 );
13889 assert!(decode_membership(&[1, 0x80]).is_err(), "a truncated varint is invalid");
13890 assert!(
13891 decode_membership(&[1, 0xff, 0xff, 0xff, 0xff, 0x10]).is_err(),
13892 "a value past u32 is invalid"
13893 );
13894 }
13895
13896 #[test]
13897 fn a_global_dictionary_may_be_larger_than_one_column_page() {
13898 let dictionary = Page {
13899 offset: HEADER,
13900 length: u32::try_from(MAX_PAGE + 1).expect("the page bound fits on disk"),
13901 hash: 0,
13902 };
13903 let table = Table {
13904 name: "items".to_owned(),
13905 fields: vec![Field::new("text", LogicalType::Varchar)],
13906 stripes: Vec::new(),
13907 rows: 0,
13908 dictionaries: vec![Some(dictionary)],
13909 dictionary_payloads: Vec::new(),
13910 distincts: vec![None],
13911 frequencies: vec![None],
13912 pair_frequencies: Vec::new(),
13913 frequency_texts: Vec::new(),
13914 host_groups: None,
13915 clustering: None,
13916 generation: 1,
13917 sections: Vec::new(),
13918 };
13919 let directory = encode_directory(&table).expect("directory");
13920 let file_size = dictionary.offset + u64::from(dictionary.length) + 1;
13921
13922 let decoded = decode_directory(&directory, file_size).expect("large lazy dictionary");
13923 assert_eq!(decoded.dictionaries[0].expect("dictionary").length, dictionary.length);
13924 }
13925
13926 #[test]
13927 fn a_column_with_one_value_everywhere_costs_almost_nothing_a_row() {
13928 let path = path("constant-codes");
13929 let mut writer =
13930 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
13931 .expect("new file");
13932 let empty = vec![Value::Varchar(String::new()); 1024];
13933 for _ in 0..4 {
13934 let column = Vector::from_values(LogicalType::Varchar, &empty).expect("strings");
13935 writer.append(&Chunk::new(vec![column]).expect("one column")).expect("a part");
13936 }
13937 writer.finish().expect("commit");
13938
13939 let reader = Reader::open(&path).expect("valid directory");
13940 let pages = reader.layout().columns.first().expect("one column").pages;
13941 assert!(pages < 256, "{pages} bytes of pages for 4,096 rows of one value");
13945 let read = reader.read(3, &[0]).expect("the last part back");
13946 assert_eq!(read.value_at(0, 0), Value::Varchar(String::new()));
13947 assert_eq!(read.value_at(1023, 0), Value::Varchar(String::new()));
13948 fs::remove_file(path).expect("remove scratch file");
13949 }
13950
13951 #[test]
13952 fn a_cascade_value_too_wide_for_its_column_is_refused_rather_than_cut() {
13953 let over = vec![i64::from(i32::MAX) + 1];
13956 let error = narrowed(&LogicalType::Integer, over).expect_err("a page that disagrees");
13957 assert!(format!("{error}").contains("not of its type"), "{error}");
13958 assert!(narrowed(&LogicalType::BigInt, vec![i64::MIN]).is_ok(), "bigint holds all of i64");
13959 assert!(narrowed(&LogicalType::Varchar, vec![0]).is_err(), "strings are not integers");
13960 }
13961
13962 #[test]
13963 fn a_code_stream_the_cascade_cannot_shrink_is_left_alone() {
13964 let mut state: u32 = 0x9e37_79b9;
13968 let spread: Vec<u32> = (0..1024)
13969 .map(|_| {
13970 state ^= state << 13;
13971 state ^= state >> 17;
13972 state ^= state << 5;
13973 state
13974 })
13975 .collect();
13976 assert_eq!(encoded_codes(&spread).expect("no failure"), None);
13977 let near: Vec<u32> = (0..1024).collect();
13978 let coded = encoded_codes(&near).expect("no failure").expect("counting up is packable");
13979 assert!(coded.len() < near.len() * 4, "{} bytes for a run of 1,024", coded.len());
13980 }
13981
13982 #[test]
13988 fn two_writes_of_the_same_rows_give_the_same_bytes() {
13989 fn written(path: &PathBuf) {
13990 let fields = (0..40)
13991 .map(|column| {
13992 let ty =
13993 if column % 4 == 0 { LogicalType::Varchar } else { LogicalType::BigInt };
13994 Field::new(format!("c{column}"), ty)
13995 })
13996 .collect::<Vec<_>>();
13997 let mut writer = Writer::create(path, "wide", fields).expect("new file");
13998 for part in 0..70_u64 {
13999 let columns = (0..40)
14000 .map(|column| {
14001 let values = (0..64_u64)
14002 .map(|row| {
14003 let seed = part.wrapping_mul(31).wrapping_add(row);
14004 if column % 4 == 0 {
14005 Value::Varchar(format!("v{}", seed % 17))
14006 } else {
14007 Value::BigInt(i64::try_from(seed % 97).expect("small"))
14008 }
14009 })
14010 .collect::<Vec<_>>();
14011 let ty = if column % 4 == 0 {
14012 LogicalType::Varchar
14013 } else {
14014 LogicalType::BigInt
14015 };
14016 Vector::from_values(ty, &values).expect("a column")
14017 })
14018 .collect::<Vec<_>>();
14019 writer.append(&Chunk::new(columns).expect("forty columns")).expect("a part");
14020 }
14021 writer.finish().expect("commit");
14022 }
14023
14024 let first = path("repeatable-one");
14025 let second = path("repeatable-two");
14026 written(&first);
14027 written(&second);
14028 let left = fs::read(&first).expect("the first file");
14029 let right = fs::read(&second).expect("the second file");
14030 assert_eq!(left.len(), right.len(), "two writes of the same rows differ in length");
14031 assert!(left == right, "two writes of the same rows differ in their bytes");
14032
14033 let reader = Reader::open(&first).expect("valid directory");
14036 assert_eq!(reader.table().rows(), 70 * 64);
14037 let read = reader.read(0, &[0, 1]).expect("the first part back");
14038 assert_eq!(read.value_at(0, 0), Value::Varchar("v0".to_owned()));
14039 assert_eq!(read.value_at(0, 1), Value::BigInt(0));
14040 fs::remove_file(first).expect("remove scratch file");
14041 fs::remove_file(second).expect("remove scratch file");
14042 }
14043
14044 fn three_tables(path: &PathBuf) {
14046 let writer = Writer::create(
14047 path,
14048 "region",
14049 vec![
14050 Field::new("r_key", LogicalType::Integer),
14051 Field::new("r_name", LogicalType::Varchar),
14052 ],
14053 )
14054 .expect("new file");
14055 let mut writer = writer;
14056 writer
14057 .append(
14058 &Chunk::new(vec![
14059 Vector::from_values(
14060 LogicalType::Integer,
14061 &[Value::Integer(0), Value::Integer(1)],
14062 )
14063 .expect("keys"),
14064 Vector::from_values(
14065 LogicalType::Varchar,
14066 &[Value::Varchar("AFRICA".to_owned()), Value::Varchar("ASIA".to_owned())],
14067 )
14068 .expect("names"),
14069 ])
14070 .expect("two columns"),
14071 )
14072 .expect("a part");
14073 let mut writer = writer
14074 .next("empty", vec![Field::new("nothing", LogicalType::BigInt)])
14075 .expect("a second table");
14076 writer
14077 .append(
14078 &Chunk::new(vec![
14079 Vector::from_values(LogicalType::BigInt, &[Value::BigInt(7)]).expect("a row"),
14080 ])
14081 .expect("one column"),
14082 )
14083 .expect("a part");
14084 let mut writer =
14085 writer.next("wide", vec![Field::new("n", LogicalType::BigInt)]).expect("a third table");
14086 for part in 0..70_i64 {
14087 let values = (0..64).map(|row| Value::BigInt(part * 64 + row)).collect::<Vec<_>>();
14088 writer
14089 .append(
14090 &Chunk::new(vec![
14091 Vector::from_values(LogicalType::BigInt, &values).expect("a column"),
14092 ])
14093 .expect("one column"),
14094 )
14095 .expect("a part");
14096 }
14097 writer.finish().expect("commit");
14098 }
14099
14100 #[test]
14101 fn three_tables_in_one_file_read_back_by_name() {
14102 let file = path("three-tables");
14103 three_tables(&file);
14104 let catalog = Catalog::open(&file).expect("a committed catalog");
14105 assert_eq!(catalog.names().collect::<Vec<_>>(), ["region", "empty", "wide"]);
14106
14107 let region = catalog.table("region").expect("the first table");
14108 assert_eq!(region.table().rows(), 2);
14109 assert_eq!(
14110 region.read(0, &[1]).expect("names").value_at(1, 0),
14111 Value::Varchar("ASIA".to_owned())
14112 );
14113
14114 let wide = catalog.table("wide").expect("the third table");
14115 assert_eq!(wide.table().rows(), 70 * 64);
14116 assert_eq!(wide.read(0, &[0]).expect("the first part").value_at(0, 0), Value::BigInt(0));
14117
14118 let empty = catalog.table("empty").expect("the second table");
14121 assert_eq!(empty.table().rows(), 1);
14122 assert_eq!(empty.read(0, &[0]).expect("the row").value_at(0, 0), Value::BigInt(7));
14123
14124 fs::remove_file(file).expect("remove scratch file");
14125 }
14126
14127 #[test]
14128 fn a_name_the_file_does_not_hold_is_an_error_rather_than_the_first_table() {
14129 let file = path("three-tables-missing");
14130 three_tables(&file);
14131 let catalog = Catalog::open(&file).expect("a committed catalog");
14132 let error = catalog.table("nation").expect_err("no such table");
14133 assert!(error.message().contains("nation"), "{}", error.message());
14134 fs::remove_file(file).expect("remove scratch file");
14135 }
14136
14137 #[test]
14138 fn a_file_of_three_tables_will_not_open_as_one() {
14139 let file = path("three-tables-unnamed");
14140 three_tables(&file);
14141 let error = Reader::open(&file).expect_err("more than one table");
14142 assert!(error.message().contains("more than one table"), "{}", error.message());
14143 fs::remove_file(file).expect("remove scratch file");
14144 }
14145
14146 #[test]
14148 fn decimals_of_every_storage_width_round_trip() {
14149 let file = path("decimals");
14150 let widths = [(4_u8, 2_u8), (9, 2), (18, 4), (38, 6)];
14151 let fields = widths
14152 .iter()
14153 .enumerate()
14154 .map(|(index, (width, scale))| {
14155 Field::new(
14156 format!("d{index}"),
14157 LogicalType::decimal(*width, *scale).expect("a decimal type"),
14158 )
14159 })
14160 .collect::<Vec<_>>();
14161 let mut writer = Writer::create(&file, "money", fields).expect("new file");
14162 let rows: [i128; 3] = [-1234, 0, 999];
14163 let columns = widths
14164 .iter()
14165 .map(|(width, scale)| {
14166 let values = rows
14167 .iter()
14168 .map(|unscaled| Value::Decimal {
14169 unscaled: *unscaled,
14170 width: *width,
14171 scale: *scale,
14172 })
14173 .collect::<Vec<_>>();
14174 Vector::from_values(
14175 LogicalType::decimal(*width, *scale).expect("a decimal type"),
14176 &values,
14177 )
14178 .expect("a decimal column")
14179 })
14180 .collect::<Vec<_>>();
14181 writer.append(&Chunk::new(columns).expect("four columns")).expect("a part");
14182 writer.finish().expect("commit");
14183
14184 let reader = Reader::open(&file).expect("a committed file");
14185 for (index, (width, scale)) in widths.iter().enumerate() {
14186 assert_eq!(
14187 reader.table().fields()[index].ty,
14188 LogicalType::decimal(*width, *scale).expect("a decimal type"),
14189 "column {index} came back as another type"
14190 );
14191 let column = reader.read(0, &[index]).expect("the column");
14192 for (row, unscaled) in rows.iter().enumerate() {
14193 assert_eq!(
14194 column.value_at(row, 0),
14195 Value::Decimal { unscaled: *unscaled, width: *width, scale: *scale },
14196 "column {index} row {row}"
14197 );
14198 }
14199 }
14200 fs::remove_file(file).expect("remove scratch file");
14201 }
14202
14203 #[test]
14204 fn two_tables_of_one_name_are_refused_before_anything_is_committed() {
14205 let file = path("two-of-a-name");
14206 let writer = Writer::create(&file, "t", vec![Field::new("a", LogicalType::BigInt)])
14207 .expect("new file");
14208 let error = writer
14209 .next("t", vec![Field::new("a", LogicalType::BigInt)])
14210 .expect_err("the same name twice");
14211 assert!(error.message().contains("same name"), "{}", error.message());
14212 fs::remove_file(file).expect("remove scratch file");
14213 }
14214
14215 #[test]
14216 fn opening_the_catalog_reads_no_table_directory() {
14217 let file = path("catalog-only");
14218 three_tables(&file);
14219 let catalog = Catalog::open(&file).expect("a committed catalog");
14220 assert_eq!(catalog.opening.reads, 2, "opening the catalog read more than the slot");
14223 assert_eq!(catalog.names().len(), 3);
14224 fs::remove_file(file).expect("remove scratch file");
14225 }
14226
14227 #[test]
14238 fn the_checksum_answers_what_it_has_always_answered() {
14239 let bytes: Vec<u8> =
14240 (0..1000_u32).map(|at| (at.wrapping_mul(31).wrapping_add(7) % 251) as u8).collect();
14241 for (length, expected) in [
14242 (0, 0xef46_db37_51d8_e999),
14243 (1, 0xa96c_7f0c_e858_bbb7),
14244 (3, 0x56e6_9576_32a4_87f9),
14245 (4, 0xc60d_15b1_e3ff_8f04),
14246 (5, 0x8088_1585_8624_dd4e),
14247 (7, 0xafbe_fc3d_6c6f_9a8e),
14248 (8, 0x3da5_c7aa_2696_83e0),
14249 (9, 0x465e_c429_b13c_3892),
14250 (15, 0xdee8_9d8a_065a_6233),
14251 (16, 0x1330_489a_7767_9c80),
14252 (31, 0x3391_303d_485e_846e),
14253 (32, 0x40b7_aff7_5d45_bbc8),
14254 (33, 0x4997_cae4_951c_17a5),
14255 (39, 0x5807_28fd_5c14_5739),
14256 (40, 0xf95c_f6f5_c08a_3d3b),
14257 (63, 0x2944_b4da_fc69_b206),
14258 (64, 0xbb76_f6ef_19bd_5a1b),
14259 (65, 0x814e_0c65_4a9f_d640),
14260 (127, 0x00de_aab1_31cf_f89b),
14261 (1000, 0x9e33_00c1_cde3_c58d),
14262 ] {
14263 assert_eq!(checksum(&bytes[..length]), expected, "the checksum of {length} bytes");
14264 }
14265 assert_eq!(checksum(b"the quick brown fox jumps over the lazy dog"), 0xed71_4233_c5a9_a792);
14266 }
14267 #[test]
14274 fn a_declared_order_comes_back_out_of_the_file() {
14275 let path = path("clustered");
14276 let shipped = vec![
14277 Field::new("key", LogicalType::BigInt),
14278 Field::new("line", LogicalType::Integer),
14279 Field::new("shipdate", LogicalType::Date),
14280 ];
14281 let plain = vec![Field::new("a", LogicalType::Integer)];
14282 let stage_zero = Clustering::new(vec![2, 0, 1], Width::Month, &shipped).expect("valid");
14283
14284 let mut writer = Writer::create(&path, "lineitem", shipped)
14285 .expect("new file")
14286 .declare(stage_zero.clone())
14287 .expect("the columns are the table's");
14288 let column = |ty: LogicalType, values: &[Value]| {
14289 Vector::from_values(ty, values).expect("the values match the type")
14290 };
14291 writer
14292 .append(
14293 &Chunk::new(vec![
14294 column(
14295 LogicalType::BigInt,
14296 &[Value::BigInt(0), Value::BigInt(1), Value::BigInt(2), Value::BigInt(3)],
14297 ),
14298 column(
14299 LogicalType::Integer,
14300 &[
14301 Value::Integer(1),
14302 Value::Integer(1),
14303 Value::Integer(1),
14304 Value::Integer(1),
14305 ],
14306 ),
14307 column(
14308 LogicalType::Date,
14309 &[Value::Date(0), Value::Date(1), Value::Date(2), Value::Date(3)],
14310 ),
14311 ])
14312 .expect("three columns"),
14313 )
14314 .expect("four rows");
14315 let mut writer = writer.next("nation", plain).expect("a second table");
14316 writer
14317 .append(
14318 &Chunk::new(vec![column(LogicalType::Integer, &[Value::Integer(7)])])
14319 .expect("one column"),
14320 )
14321 .expect("one row");
14322 writer.finish().expect("commit");
14323
14324 let catalog = Catalog::open(&path).expect("reopen");
14325 let lineitem = catalog.table("lineitem").expect("the clustered table");
14326 assert_eq!(lineitem.table().clustering(), Some(&stage_zero));
14327 let nation = catalog.table("nation").expect("the plain table");
14328 assert_eq!(nation.table().clustering(), None, "nobody declared one here");
14329
14330 assert_eq!(lineitem.table().rows(), 4);
14333 assert_eq!(nation.table().rows(), 1);
14334 fs::remove_file(&path).ok();
14335 }
14336
14337 #[test]
14339 fn a_declaration_off_the_end_of_the_table_never_reaches_the_file() {
14340 let path = path("clustered-bad");
14341 let writer = Writer::create(&path, "items", vec![Field::new("a", LogicalType::Integer)])
14342 .expect("new file");
14343 let four =
14344 (0..4).map(|at| Field::new(format!("c{at}"), LogicalType::Integer)).collect::<Vec<_>>();
14345 let wrong = Clustering::new(vec![3], Width::Exact, &four).expect("valid against four");
14346 assert!(writer.declare(wrong).is_err(), "the table has one column, not four");
14347 fs::remove_file(&path).ok();
14348 }
14349
14350 #[test]
14358 fn the_dictionary_order_is_the_byte_order_however_deep_the_values_agree() {
14359 let mut values = vec![String::new(), "http://".to_owned()];
14360 for host in 0..7 {
14361 for path in 0..30 {
14362 values.push(format!("http://example{host}.test/page/{path:04}/index.html"));
14363 values.push(format!("http://example{host}.test/page/{path:04}"));
14364 }
14365 }
14366 values.push("http://example0.test/page/0000/index.htmlx".to_owned());
14367
14368 let mut dictionary = GlobalDictionary::new();
14369 for value in &values {
14370 dictionary.code(value).expect("a code for every value");
14371 }
14372 dictionary.finish_blocks().expect("the last block encodes");
14373 let ranked = dictionary.ranked(None).expect("a sorted order");
14374 assert_eq!(ranked.len(), values.len(), "one entry a distinct value");
14375
14376 let spellings = dictionary_values(&dictionary);
14377 let seen = ranked
14378 .iter()
14379 .map(|&(_, code)| {
14380 String::from_utf8(spellings[code as usize].clone()).expect("text in, text out")
14381 })
14382 .collect::<Vec<_>>();
14383 let mut wanted = values.clone();
14384 wanted.sort_unstable();
14385 assert_eq!(seen, wanted, "the order is the order the bytes give");
14386
14387 for &(carried, code) in &ranked {
14388 let value = &spellings[code as usize];
14389 assert_eq!(carried, head(value), "the head belongs to the value it is filed with");
14390 }
14391 }
14392
14393 #[test]
14398 fn the_commonest_entries_are_the_ones_a_full_sort_would_have_kept() {
14399 let entry =
14400 |value: u32, count: u64| FrequencyEntry { value: FrequencyValue::Code(value), count };
14401 let mut all = (0..FREQUENCY_ENTRIES as u32 * 3)
14402 .map(|code| entry(code, u64::from(code % 7) + 1))
14403 .collect::<Vec<_>>();
14404 all.push(FrequencyEntry { value: FrequencyValue::Null, count: 4 });
14405
14406 let mut sorted = all.clone();
14407 sorted.sort_unstable_by(|left, right| {
14408 right.count.cmp(&left.count).then_with(|| frequency_order(left.value, right.value))
14409 });
14410 let wanted_omitted = sorted[FREQUENCY_ENTRIES].count;
14411 sorted.truncate(FREQUENCY_ENTRIES);
14412
14413 let mut picked = all.clone();
14414 let omitted = keep_most_frequent(&mut picked);
14415 assert_eq!(omitted, wanted_omitted, "the largest count that did not make the cut");
14416 assert_eq!(picked.len(), FREQUENCY_ENTRIES, "the cut is where it says it is");
14417 assert!(
14418 picked
14419 .iter()
14420 .zip(&sorted)
14421 .all(|(one, two)| one.value == two.value && one.count == two.count),
14422 "the same entries in the same order"
14423 );
14424
14425 let mut short = all[..FREQUENCY_ENTRIES - 1].to_vec();
14426 let omitted = keep_most_frequent(&mut short);
14427 assert_eq!(omitted, 0, "nothing is omitted when everything fits");
14428 assert!(short.windows(2).all(|pair| pair[0].count >= pair[1].count), "still in order");
14429 }
14430
14431 #[test]
14433 fn a_short_dictionary_sorts_without_a_bucketing_pass() {
14434 let empty = GlobalDictionary::new();
14435 assert!(empty.ranked(None).expect("an empty order").is_empty(), "nothing in, nothing out");
14436
14437 let mut dictionary = GlobalDictionary::new();
14438 for value in ["pear", "apple", "", "apples", "app"] {
14439 dictionary.code(value).expect("a code for every value");
14440 }
14441 dictionary.finish_blocks().expect("the one block encodes");
14442 let spellings = dictionary_values(&dictionary);
14443 let seen = dictionary
14444 .ranked(None)
14445 .expect("a sorted order")
14446 .iter()
14447 .map(|&(_, code)| spellings[code as usize].clone())
14448 .collect::<Vec<_>>();
14449 let wanted: Vec<Vec<u8>> =
14450 [&b""[..], b"app", b"apple", b"apples", b"pear"].iter().map(|v| v.to_vec()).collect();
14451 assert_eq!(seen, wanted, "shorter first where one runs out inside another");
14452 }
14453}