1#![forbid(unsafe_code)]
34
35use std::borrow::Cow;
36use std::cmp::{Ordering, Reverse};
37use std::collections::{BTreeMap, HashMap, VecDeque};
38use std::fs::File;
39use std::mem::{size_of, size_of_val};
40use std::path::Path;
41use std::slice;
42use std::sync::atomic::{AtomicBool, AtomicUsize, Ordering as Atomic};
43use std::sync::{Arc, Condvar, Mutex, OnceLock, PoisonError, Weak};
44
45use rudb_common::bounds::{self, Bound, Op, scaled_as};
46use rudb_common::{Clustering, Error, Field, LogicalType, PhysicalType, Result, Value, Width};
47use rudb_encoding::sequence::Sequence;
48use rudb_encoding::{bitpack, chooser, integer, string};
49use rudb_io::{Filesystem, OpenMode, RealFilesystem};
50use rudb_metrics::{LoadProfile, Stage};
51use rudb_storage::sieve::Sieve;
52use rudb_storage::{Probe, Range, Zone};
53use rudb_vector::string::StringColumn;
54use rudb_vector::validity::Validity;
55use rudb_vector::{Buffer, Chunk, Data, Packed, TextSource, Vector, search_below};
56
57mod distinct;
58pub mod graph;
59pub mod host;
60mod prepare;
61mod projection;
62mod run_projection;
63use prepare::Lent;
64pub mod section;
65pub mod stats;
66mod zones;
67
68pub use prepare::{Building, DICTIONARY_CAP_BYTES, Merged, Merger, Paged, Prepared, Preparer};
69pub use projection::build_sorted_projection;
70pub use run_projection::{RunProjectionPart, RunProjectionScan, build_run_projection};
71pub use section::Section;
72pub use zones::{Common, Stripes, ascending, distincts, widths};
73
74const MAGIC: &[u8; 8] = b"RUDBNV10";
75const DIRECTORY: &[u8; 8] = b"RUDBDI10";
76const CATALOG: &[u8; 8] = b"RUDBCA10";
77const NONZERO_COUNTS: &[u8; 8] = b"RUDBNZ10";
78const AGGREGATE_SUMS: &[u8; 8] = b"RUDBAG10";
79const DISTINCT_COUNTS: &[u8; 8] = b"RUDBDC10";
80const INTEGER_EXTREMES: &[u8; 8] = b"RUDBEX10";
81const COMPLETE_FREQUENCIES: &[u8; 8] = b"RUDBFQ10";
82const MAX_CATALOG_FREQUENCIES: usize = 64;
83const FORMAT: u32 = 29;
84
85const READABLE: &[u32] = &[22, 23, 24, 25, 26, 27, 28, FORMAT];
116
117const HEADER: u64 = 80;
118const SLOT_BYTES: usize = 28;
119const MAX_PAGE: usize = 256 * 1024 * 1024;
120const MAX_DIRECTORY: usize = 128 * 1024 * 1024;
121const FREQUENCIES_V2: &[u8; 8] = b"RUDBFQ2\0";
122const FREQUENCIES: &[u8; 8] = b"RUDBFQ3\0";
123const FREQUENCIES_SPANS: &[u8; 8] = b"RUDBFQ4\0";
124const FREQUENCY_TEXTS: &[u8; 8] = b"RUDBFT1\0";
132const HOST_GROUPS: &[u8; 8] = b"RUDBHG1\0";
134const PAIR_FREQUENCIES: &[u8; 8] = b"RUDBPF1\0";
140const CLUSTERING: &[u8; 8] = b"RUDBCL1\0";
155const DEMOTED: &[u8; 8] = b"RUDBDM1\0";
175const SECTIONS: &[u8; 8] = b"RUDBSE1\0";
183const KEYS: &[u8; 8] = b"RUDBKY1\0";
190const DICTIONARY_PAYLOADS: &[u8; 8] = b"RUDBDP1\0";
198
199const MAX_SECTIONS: usize = 4096;
206const FREQUENCY_CANDIDATES: usize = 32_768;
207const FREQUENCY_ENTRIES: usize = 512;
208const FREQUENCY_BUILD_RANK: usize = 10;
209const FREQUENCY_ORDINALS: usize = 131_072;
210const MAX_PAIR_FREQUENCIES: usize = 1024;
211const FREQUENCY_TEXT_BUDGET: usize = 1024 * 1024;
216const MAX_FREQUENCY_WORKERS: usize = 32;
223
224fn close_workers() -> usize {
226 std::thread::available_parallelism().map_or(1, usize::from).min(MAX_FREQUENCY_WORKERS)
227}
228
229const CLOSE_BYTES: usize = 1 << 30;
240
241const NUMERIC_CLOSE_BYTES: usize = 4 << 20;
244
245const MAX_ENCODE_WORKERS: usize = 32;
252
253const WRITEBACK_STRETCH: u64 = 32 << 20;
261
262const SIEVE_BUDGET: usize = 8 * 1024;
270
271const PART_BOUND_BYTES: usize = 24;
280
281fn io(error: std::io::Error) -> Error {
282 Error::io(error.to_string())
283}
284
285fn invalid(message: &str) -> Error {
286 Error::invalid_input(format!("invalid rudb native file: {message}"))
287}
288
289fn sum(counts: impl Iterator<Item = u64>) -> u64 {
291 counts.fold(0, u64::saturating_add)
292}
293
294fn span_bytes(spans: &[Span], at: usize) -> u64 {
296 spans.get(at).map_or(0, |span| u64::from(span.length))
297}
298
299fn page_bytes(pages: &[Option<Page>], at: usize) -> u64 {
301 pages.get(at).and_then(Option::as_ref).map_or(0, Page::bytes)
302}
303
304fn dictionary_bytes(table: &Table, at: usize) -> u64 {
306 page_bytes(&table.dictionaries, at)
307 .saturating_add(table.dictionary_payloads.get(at).copied().unwrap_or(0))
308}
309
310fn checksum(bytes: &[u8]) -> u64 {
320 seeded_checksum(bytes, 0)
321}
322
323#[must_use]
330pub fn content_name(bytes: &[u8]) -> u128 {
331 let seed = u64::from(FORMAT);
332 u128::from(seeded_checksum(bytes, seed)) << 64 | u128::from(seeded_checksum(bytes, !seed))
333}
334
335#[derive(Debug, Clone)]
341pub struct ContentNamer {
342 seeds: [u64; 2],
343 lanes: [[u64; 4]; 2],
344 held: [u8; 32],
345 filled: usize,
346 length: u64,
347}
348
349impl Default for ContentNamer {
350 fn default() -> Self {
351 let seed = u64::from(FORMAT);
352 let seeds = [seed, !seed];
353 let lanes = seeds.map(|seed| {
354 [
355 seed.wrapping_add(XXH_P1).wrapping_add(XXH_P2),
356 seed.wrapping_add(XXH_P2),
357 seed,
358 seed.wrapping_sub(XXH_P1),
359 ]
360 });
361 Self { seeds, lanes, held: [0; 32], filled: 0, length: 0 }
362 }
363}
364
365impl ContentNamer {
366 pub fn update(&mut self, mut bytes: &[u8]) {
368 self.length += bytes.len() as u64;
369 if self.filled > 0 {
370 let take = (32 - self.filled).min(bytes.len());
371 self.held[self.filled..self.filled + take].copy_from_slice(&bytes[..take]);
372 self.filled += take;
373 bytes = &bytes[take..];
374 if self.filled < 32 {
375 return;
376 }
377 let block = self.held;
378 self.lanes.iter_mut().for_each(|lanes| checksum_block(lanes, &block));
379 self.filled = 0;
380 }
381 let mut blocks = bytes.chunks_exact(32);
382 for block in blocks.by_ref() {
383 self.lanes.iter_mut().for_each(|lanes| checksum_block(lanes, block));
384 }
385 let rest = blocks.remainder();
386 self.held[..rest.len()].copy_from_slice(rest);
387 self.filled = rest.len();
388 }
389
390 #[must_use]
392 pub fn finish(&self) -> u128 {
393 let rest = &self.held[..self.filled];
394 let [first, second] = [0, 1].map(|at| {
395 if self.length < 32 {
396 checksum_tail(self.seeds[at].wrapping_add(XXH_P5).wrapping_add(self.length), rest)
397 } else {
398 finish_checksum(self.lanes[at], rest, self.length)
399 }
400 });
401 u128::from(first) << 64 | u128::from(second)
402 }
403}
404
405fn seeded_checksum(bytes: &[u8], seed: u64) -> u64 {
414 let mut blocks = bytes.chunks_exact(32);
417 let rest = blocks.remainder();
418 if bytes.len() < 32 {
419 return checksum_tail(seed.wrapping_add(XXH_P5).wrapping_add(bytes.len() as u64), rest);
420 }
421 let mut lanes = [
422 seed.wrapping_add(XXH_P1).wrapping_add(XXH_P2),
423 seed.wrapping_add(XXH_P2),
424 seed,
425 seed.wrapping_sub(XXH_P1),
426 ];
427 for block in blocks.by_ref() {
428 checksum_block(&mut lanes, block);
429 }
430 finish_checksum(lanes, rest, bytes.len() as u64)
431}
432
433const XXH_P1: u64 = 11_400_714_785_074_694_791;
434const XXH_P2: u64 = 14_029_467_366_897_019_727;
435const XXH_P3: u64 = 1_609_587_929_392_839_161;
436const XXH_P4: u64 = 9_650_029_242_287_828_579;
437const XXH_P5: u64 = 2_870_177_450_012_600_261;
438
439fn checksum_round(state: u64, word: u64) -> u64 {
440 state.wrapping_add(word.wrapping_mul(XXH_P2)).rotate_left(31).wrapping_mul(XXH_P1)
441}
442
443fn checksum_word(chunk: &[u8]) -> u64 {
444 u64::from_le_bytes(chunk.try_into().expect("eight checksum bytes"))
445}
446
447fn checksum_block(lanes: &mut [u64; 4], block: &[u8]) {
449 for (lane, chunk) in lanes.iter_mut().zip(block.chunks_exact(8)) {
450 *lane = checksum_round(*lane, checksum_word(chunk));
451 }
452}
453
454fn finish_checksum(lanes: [u64; 4], rest: &[u8], length: u64) -> u64 {
456 let merge = |state: u64, lane: u64| {
457 (state ^ checksum_round(0, lane)).wrapping_mul(XXH_P1).wrapping_add(XXH_P4)
458 };
459 let [one, two, three, four] = lanes;
460 let combined = one
461 .rotate_left(1)
462 .wrapping_add(two.rotate_left(7))
463 .wrapping_add(three.rotate_left(12))
464 .wrapping_add(four.rotate_left(18));
465 let hash = merge(merge(merge(merge(combined, one), two), three), four);
466 checksum_tail(hash.wrapping_add(length), rest)
467}
468
469fn checksum_tail(mut hash: u64, mut rest: &[u8]) -> u64 {
471 let mut words = rest.chunks_exact(8);
472 for chunk in words.by_ref() {
473 hash ^= checksum_round(0, checksum_word(chunk));
474 hash = hash.rotate_left(27).wrapping_mul(XXH_P1).wrapping_add(XXH_P4);
475 }
476 rest = words.remainder();
477 if rest.len() >= 4 {
478 let (head, tail) = rest.split_at(4);
479 let quarter = u32::from_le_bytes(head.try_into().expect("four checksum bytes"));
480 hash ^= u64::from(quarter).wrapping_mul(XXH_P1);
481 hash = hash.rotate_left(23).wrapping_mul(XXH_P2).wrapping_add(XXH_P3);
482 rest = tail;
483 }
484 for &byte in rest {
485 hash ^= u64::from(byte).wrapping_mul(XXH_P5);
486 hash = hash.rotate_left(11).wrapping_mul(XXH_P1);
487 }
488 hash ^= hash >> 33;
489 hash = hash.wrapping_mul(XXH_P2);
490 hash ^= hash >> 29;
491 hash = hash.wrapping_mul(XXH_P3);
492 hash ^ (hash >> 32)
493}
494
495fn file_checksum(file: &File, offset: u64, length: usize) -> Result<u64> {
501 walk_checksummed(file, offset, length, DIRECTORY_WINDOW, |_| Ok(()))
502}
503
504fn walk_checksummed(
510 file: &File,
511 offset: u64,
512 length: usize,
513 window: usize,
514 mut each: impl FnMut(&[u8]) -> Result<()>,
515) -> Result<u64> {
516 debug_assert!(window % 32 == 0 && window > 0, "a window is whole blocks of the hash");
517 if length < 32 {
518 let mut bytes = vec![0; length];
519 read_at(file, offset, &mut bytes)?;
520 each(&bytes)?;
521 return Ok(checksum(&bytes));
522 }
523 let mut lanes = [XXH_P1.wrapping_add(XXH_P2), XXH_P2, 0, 0_u64.wrapping_sub(XXH_P1)];
524 let mut buffer = vec![0; window.min(length)];
525 let mut read = 0;
526 let (mut whole, mut filled) = (0, 0);
527 while read < length {
528 filled = buffer.len().min(length - read);
529 read_at(file, offset + read as u64, &mut buffer[..filled])?;
530 read += filled;
531 each(&buffer[..filled])?;
532 whole = filled / 32 * 32;
533 for block in buffer[..whole].chunks_exact(32) {
534 checksum_block(&mut lanes, block);
535 }
536 }
537 Ok(finish_checksum(lanes, &buffer[whole..filled], length as u64))
538}
539
540#[derive(Debug, Clone, Copy)]
541struct Slot {
542 offset: u64,
543 length: u32,
544 generation: u64,
545 hash: u64,
546}
547
548impl Slot {
549 fn bytes(self) -> [u8; SLOT_BYTES] {
550 let mut result = [0; SLOT_BYTES];
551 result[..8].copy_from_slice(&self.offset.to_le_bytes());
552 result[8..12].copy_from_slice(&self.length.to_le_bytes());
553 result[12..20].copy_from_slice(&self.generation.to_le_bytes());
554 result[20..28].copy_from_slice(&self.hash.to_le_bytes());
555 result
556 }
557
558 fn read(bytes: &[u8]) -> Self {
559 Self {
560 offset: u64::from_le_bytes(bytes[..8].try_into().expect("eight bytes")),
561 length: u32::from_le_bytes(bytes[8..12].try_into().expect("four bytes")),
562 generation: u64::from_le_bytes(bytes[12..20].try_into().expect("eight bytes")),
563 hash: u64::from_le_bytes(bytes[20..28].try_into().expect("eight bytes")),
564 }
565 }
566}
567
568#[derive(Debug, Clone, Copy)]
569struct Page {
570 offset: u64,
571 length: u32,
572 hash: u64,
573}
574
575impl Page {
576 fn bytes(&self) -> u64 {
578 u64::from(self.length)
579 }
580}
581
582#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
583enum FrequencyValue {
584 Null,
585 Integer(i128),
586 Code(u32),
587}
588
589type FrequencyMap<V> = HashMap<u64, V, Spread>;
595
596#[derive(Debug)]
610struct Candidates {
611 slots: Vec<Candidate>,
614 held: usize,
615 nulls: u32,
616 decrements: u64,
617 survivors: Vec<Candidate>,
619}
620
621#[derive(Debug, Default, Clone, Copy)]
623struct Candidate {
624 bits: u64,
625 count: u32,
626}
627
628const FIRST_CANDIDATE_SLOTS: usize = 64;
630
631impl Default for Candidates {
632 fn default() -> Self {
633 Self {
634 slots: vec![Candidate::default(); FIRST_CANDIDATE_SLOTS],
635 held: 0,
636 nulls: 0,
637 decrements: 0,
638 survivors: Vec::new(),
639 }
640 }
641}
642
643impl Candidates {
644 fn add(&mut self, bits: Option<u64>, mut times: u32) {
651 while times > 0 {
652 let room = self.held + usize::from(self.nulls != 0) < FREQUENCY_CANDIDATES;
653 match bits {
654 Some(bits) => {
655 let (at, found) = self.find(bits);
656 if found {
657 self.slots[at].count = self.slots[at].count.saturating_add(times);
658 return;
659 }
660 if room {
661 self.place(at, bits, times);
662 return;
663 }
664 }
665 None if self.nulls != 0 => {
666 self.nulls = self.nulls.saturating_add(times);
667 return;
668 }
669 None if room => {
670 self.nulls = times;
671 return;
672 }
673 None => {}
674 }
675 self.decrement();
676 times -= 1;
677 }
678 }
679
680 fn find(&self, bits: u64) -> (usize, bool) {
682 let mask = self.slots.len() - 1;
683 let mut at = home(bits, self.slots.len());
684 loop {
685 let slot = self.slots[at];
686 if slot.count == 0 {
687 return (at, false);
688 }
689 if slot.bits == bits {
690 return (at, true);
691 }
692 at = (at + 1) & mask;
693 }
694 }
695
696 fn position(&self, bits: u64) -> Option<usize> {
698 match self.find(bits) {
699 (at, true) => Some(at),
700 (_, false) => None,
701 }
702 }
703
704 fn place(&mut self, at: usize, bits: u64, count: u32) {
707 let at = if (self.held + 1) * 2 > self.slots.len() {
708 let wider = self.slots.len() * 2;
709 let old = std::mem::replace(&mut self.slots, vec![Candidate::default(); wider]);
710 for slot in old.into_iter().filter(|slot| slot.count != 0) {
711 let (to, _) = self.find(slot.bits);
712 self.slots[to] = slot;
713 }
714 self.find(bits).0
715 } else {
716 at
717 };
718 self.slots[at] = Candidate { bits, count };
719 self.held += 1;
720 }
721
722 fn decrement(&mut self) {
724 let mut survivors = std::mem::take(&mut self.survivors);
725 survivors.clear();
726 survivors.extend(
727 self.slots
728 .iter()
729 .filter(|slot| slot.count > 1)
730 .map(|slot| Candidate { bits: slot.bits, count: slot.count - 1 }),
731 );
732 self.slots.fill(Candidate::default());
733 self.held = survivors.len();
734 for &slot in &survivors {
735 let (at, _) = self.find(slot.bits);
736 self.slots[at] = slot;
737 }
738 self.survivors = survivors;
739 self.nulls = self.nulls.saturating_sub(1);
740 self.decrements = self.decrements.saturating_add(1);
741 }
742
743 fn pairs(&self) -> impl Iterator<Item = (u64, u32)> + '_ {
745 self.slots.iter().filter(|slot| slot.count != 0).map(|slot| (slot.bits, slot.count))
746 }
747}
748
749fn home(bits: u64, slots: usize) -> usize {
754 (bits.wrapping_mul(0x9E37_79B9_7F4A_7C15) >> (64 - slots.trailing_zeros())) as usize
755}
756
757#[derive(Debug, Default)]
759struct Run {
760 bits: Option<u64>,
761 times: u32,
762}
763
764impl Run {
765 fn push(&mut self, bits: Option<u64>) -> Option<(Option<u64>, u32)> {
767 if self.times != 0 && self.bits == bits && self.times < u32::MAX {
768 self.times += 1;
769 return None;
770 }
771 let ended = self.take();
772 self.bits = bits;
773 self.times = 1;
774 ended
775 }
776
777 fn take(&mut self) -> Option<(Option<u64>, u32)> {
779 let times = std::mem::take(&mut self.times);
780 (times != 0).then_some((self.bits, times))
781 }
782}
783
784#[derive(Debug, Default, Clone, Copy)]
786struct Spread;
787
788impl std::hash::BuildHasher for Spread {
789 type Hasher = SpreadHasher;
790
791 fn build_hasher(&self) -> SpreadHasher {
792 SpreadHasher(0)
793 }
794}
795
796#[derive(Debug)]
803struct SpreadHasher(u64);
804
805impl SpreadHasher {
806 fn mix(&mut self, word: u64) {
807 let product = u128::from(self.0 ^ word) * 0x9E37_79B9_7F4A_7C15_u128;
808 self.0 = (product as u64) ^ ((product >> 64) as u64);
809 }
810}
811
812impl std::hash::Hasher for SpreadHasher {
813 fn write(&mut self, bytes: &[u8]) {
814 for part in bytes.chunks(8) {
815 let mut word = [0; 8];
816 word[..part.len()].copy_from_slice(part);
817 self.mix(u64::from_le_bytes(word));
818 }
819 }
820
821 fn write_u32(&mut self, value: u32) {
822 self.mix(u64::from(value));
823 }
824
825 fn write_u64(&mut self, value: u64) {
826 self.mix(value);
827 }
828
829 fn write_i128(&mut self, value: i128) {
830 self.mix(value as u64);
831 self.mix((value >> 64) as u64);
832 }
833
834 fn write_isize(&mut self, value: isize) {
835 self.mix(value as u64);
836 }
837
838 fn finish(&self) -> u64 {
839 self.0
840 }
841}
842
843#[derive(Debug, Clone)]
844struct FrequencyEntry {
845 value: FrequencyValue,
846 count: u64,
847}
848
849#[derive(Debug, Clone)]
854struct FrequencySummary {
855 entries: Vec<FrequencyEntry>,
856 omitted_max: u64,
857 ordinals: Vec<u64>,
858 ordinal_entries: Vec<u16>,
859}
860
861#[derive(Debug, Clone)]
862struct PairFrequencyEntry {
863 first_entry: u16,
864 second: Option<u32>,
865 count: u64,
866}
867
868#[derive(Debug, Clone)]
874struct PairFrequencySummary {
875 first: u16,
876 second: u16,
877 entries: Vec<PairFrequencyEntry>,
878 omitted_max: u64,
879}
880
881#[derive(Debug, Clone)]
889enum Frequencies {
890 Held(FrequencySummary),
891 Stored {
894 span: Span,
895 values: bool,
896 entries: usize,
897 },
898}
899
900#[derive(Debug, Clone)]
905pub struct FrequencyPrefix {
906 pub entries: Vec<(Value, u64)>,
908 pub omitted_max: u64,
910}
911
912#[derive(Debug, Clone, PartialEq)]
914pub struct FrequencyOccurrences {
915 pub omitted_max: u64,
917 pub ordinals: Vec<u64>,
919 pub anchors: Vec<Value>,
921 pub anchor_indices: Vec<u16>,
923}
924
925pub type PairFrequencyCounts = Vec<(Vec<Value>, u64)>;
927
928#[derive(Debug, Clone, Copy, Default)]
935struct Span {
936 offset: u64,
937 length: u32,
938}
939
940#[derive(Debug, Clone, Default)]
948struct Pages {
949 columns: usize,
950 held: Box<[StripePage]>,
951}
952
953#[derive(Debug, Clone, Copy)]
955struct StripePage {
956 offset: u64,
957 hash: u64,
958 length: u32,
959 column: u32,
960}
961
962impl Pages {
963 fn from_slots(slots: Vec<Option<Page>>) -> Result<Self> {
965 let mut held = Vec::with_capacity(slots.iter().flatten().count());
966 for (column, page) in slots.iter().enumerate() {
967 if let Some(page) = page {
968 let column =
969 u32::try_from(column).map_err(|_| invalid("too many columns for a page"))?;
970 held.push(StripePage {
971 offset: page.offset,
972 hash: page.hash,
973 length: page.length,
974 column,
975 });
976 }
977 }
978 Ok(Self { columns: slots.len(), held: held.into_boxed_slice() })
979 }
980
981 fn get(&self, column: usize) -> Option<Page> {
983 let at = self.held.binary_search_by_key(&column, |placed| placed.column as usize).ok()?;
984 let placed = self.held[at];
985 Some(Page { offset: placed.offset, length: placed.length, hash: placed.hash })
986 }
987
988 fn slots(&self) -> impl Iterator<Item = Option<Page>> + '_ {
990 (0..self.columns).map(|column| self.get(column))
991 }
992
993 fn bytes(&self, column: usize) -> u64 {
995 self.get(column).map_or(0, |page| page.bytes())
996 }
997}
998
999#[derive(Debug, Clone)]
1001pub struct Stripe {
1002 rows: usize,
1003 parts: Vec<u32>,
1006 index: Span,
1010 pages: Vec<Span>,
1011 memberships: Pages,
1012 sieves: Pages,
1015 part_ranges: Pages,
1026 zone: Zone,
1027}
1028
1029impl Stripe {
1030 #[must_use]
1032 pub fn rows(&self) -> usize {
1033 self.rows
1034 }
1035
1036 #[must_use]
1038 pub fn parts(&self) -> usize {
1039 self.parts.len()
1040 }
1041
1042 #[must_use]
1048 pub fn zone(&self) -> &Zone {
1049 &self.zone
1050 }
1051}
1052
1053#[derive(Debug, Clone)]
1055pub struct Table {
1056 name: String,
1057 fields: Vec<Field>,
1058 stripes: Vec<Stripe>,
1059 rows: usize,
1060 dictionaries: Vec<Option<Page>>,
1061 dictionary_payloads: Vec<u64>,
1067 demoted: Vec<bool>,
1073 frequencies: Vec<Option<Frequencies>>,
1074 pair_frequencies: Vec<PairFrequencySummary>,
1075 frequency_texts: Vec<Vec<Option<Vec<u8>>>>,
1080 host_groups: Option<host::HostSummary>,
1082 distincts: Vec<Option<u64>>,
1092 clustering: Option<Clustering>,
1100 generation: u64,
1114 sections: Vec<Section>,
1121 constraints: Constraints,
1124}
1125
1126#[derive(Debug, Clone, Default, PartialEq, Eq)]
1131pub struct Constraints {
1132 pub keys: Vec<(Vec<u16>, bool)>,
1134 pub foreign: Vec<StoredForeign>,
1136}
1137
1138impl Constraints {
1139 #[must_use]
1141 pub fn is_empty(&self) -> bool {
1142 self.keys.is_empty() && self.foreign.is_empty()
1143 }
1144}
1145
1146#[derive(Debug, Clone, PartialEq, Eq)]
1148pub struct StoredForeign {
1149 pub columns: Vec<u16>,
1151 pub table: String,
1153 pub referenced: Vec<u16>,
1155}
1156
1157impl Table {
1158 #[must_use]
1160 pub fn name(&self) -> &str {
1161 &self.name
1162 }
1163
1164 #[must_use]
1166 pub fn fields(&self) -> &[Field] {
1167 &self.fields
1168 }
1169
1170 #[must_use]
1172 pub fn rows(&self) -> usize {
1173 self.rows
1174 }
1175
1176 #[must_use]
1178 pub fn stripes(&self) -> &[Stripe] {
1179 &self.stripes
1180 }
1181
1182 #[must_use]
1184 pub fn clustering(&self) -> Option<&Clustering> {
1185 self.clustering.as_ref()
1186 }
1187
1188 #[must_use]
1190 pub fn constraints(&self) -> &Constraints {
1191 &self.constraints
1192 }
1193
1194 #[must_use]
1199 pub fn generation(&self) -> u64 {
1200 self.generation
1201 }
1202
1203 #[must_use]
1210 pub fn sections(&self) -> &[Section] {
1211 &self.sections
1212 }
1213}
1214
1215#[derive(Debug, Clone)]
1227struct Entry {
1228 name: String,
1229 fields: Vec<Field>,
1230 rows: usize,
1231 directory: Page,
1233 nonzero: Vec<Option<u64>>,
1236 aggregates: Vec<Option<(i128, u64)>>,
1238 distincts: Vec<Option<u64>>,
1240 extremes: Vec<StoredIntegerExtremes>,
1242 frequencies: Vec<StoredNumericFrequencies>,
1244}
1245
1246type StoredIntegerExtremes = Option<Option<(i128, i128)>>;
1247type StoredNumericFrequencies = Option<NumericFrequencies>;
1248
1249#[derive(Debug, Clone, PartialEq, Eq)]
1262pub struct ViewEntry {
1263 pub name: String,
1265 pub sql: String,
1267 pub statement: String,
1269 pub aliases: Vec<String>,
1271 pub columns: Vec<Field>,
1273}
1274
1275#[derive(Debug, Clone)]
1277pub struct ColumnLayout {
1278 pub name: String,
1280 pub kind: String,
1282 pub pages: u64,
1284 pub memberships: u64,
1286 pub sieves: u64,
1288 pub part_ranges: u64,
1290 pub dictionary: u64,
1292}
1293
1294impl ColumnLayout {
1295 #[must_use]
1297 pub fn total(&self) -> u64 {
1298 self.pages
1299 .saturating_add(self.memberships)
1300 .saturating_add(self.sieves)
1301 .saturating_add(self.part_ranges)
1302 .saturating_add(self.dictionary)
1303 }
1304}
1305
1306#[derive(Debug, Clone)]
1317pub struct Layout {
1318 pub file: u64,
1320 pub rows: usize,
1322 pub stripes: usize,
1324 pub parts: usize,
1326 pub columns: Vec<ColumnLayout>,
1328 pub indexes: u64,
1331 pub directory: u64,
1333 pub header: u64,
1335}
1336
1337impl Layout {
1338 #[must_use]
1340 pub fn columns_total(&self) -> u64 {
1341 self.columns.iter().map(ColumnLayout::total).fold(0, u64::saturating_add)
1342 }
1343
1344 #[must_use]
1350 pub fn unaccounted(&self) -> u64 {
1351 self.file
1352 .saturating_sub(self.columns_total())
1353 .saturating_sub(self.indexes)
1354 .saturating_sub(self.directory)
1355 .saturating_sub(self.header)
1356 }
1357}
1358
1359#[derive(Debug, Clone)]
1370pub struct StoredPart {
1371 pub stripe: usize,
1373 pub part: usize,
1375 pub row: usize,
1377 pub rows: usize,
1379 pub encoding: String,
1381 pub bytes: u64,
1383 pub page: u64,
1385 pub offset: u64,
1387 pub low: Option<Value>,
1389 pub high: Option<Value>,
1391 pub nulls: Option<usize>,
1393}
1394
1395const DICTIONARY_CHECK_SEED: u64 = 11_400_714_819_323_198_485;
1402
1403#[derive(Debug)]
1428struct GlobalDictionary {
1429 primary: HashMap<u64, u32, Spread>,
1433 collisions: HashMap<u64, Vec<u32>, Spread>,
1434 checks: Vec<u64>,
1436 ends: Vec<u32>,
1438 counts: Vec<u64>,
1439 nulls: u64,
1440 filling: Vec<u8>,
1442 grams: Vec<[u8; TEXT_GRAM_BYTES]>,
1448 waiting: Vec<(usize, Vec<u8>)>,
1453 sample: Vec<(usize, Vec<u8>)>,
1459 stride: usize,
1461 shape: Option<chooser::Settled>,
1463 settled: usize,
1465 blocks: Vec<Vec<u8>>,
1470 early: BTreeMap<usize, EncodedBlock>,
1476 placed: Vec<Placed>,
1478 charged: u64,
1481 demoted: bool,
1483}
1484
1485#[derive(Debug, Clone, Copy)]
1487struct Placed {
1488 start: u64,
1489 length: u64,
1490 hash: u64,
1491}
1492
1493type RankedDictionary = (Vec<(u64, u32)>, Vec<u8>, Vec<u64>);
1495
1496impl GlobalDictionary {
1497 fn new() -> Self {
1498 Self {
1499 primary: HashMap::default(),
1500 collisions: HashMap::default(),
1501 checks: Vec::new(),
1502 ends: Vec::new(),
1503 counts: Vec::new(),
1504 nulls: 0,
1505 filling: Vec::new(),
1506 grams: Vec::new(),
1507 waiting: Vec::new(),
1508 sample: Vec::new(),
1509 stride: 1,
1510 shape: None,
1511 settled: 0,
1512 blocks: Vec::new(),
1513 early: BTreeMap::new(),
1514 placed: Vec::new(),
1515 charged: 0,
1516 demoted: false,
1517 }
1518 }
1519
1520 fn values(&self) -> usize {
1522 self.ends.len()
1523 }
1524
1525 fn closing_bytes(&self) -> usize {
1528 let values = self.values();
1529 let decoded = (0..values.div_ceil(TEXT_PAYLOAD_VALUES))
1530 .map(|block| self.ends[((block + 1) * TEXT_PAYLOAD_VALUES).min(values) - 1] as usize)
1531 .sum::<usize>();
1532 decoded.saturating_add(values.saturating_mul(size_of::<(u64, u32)>() + size_of::<u32>()))
1533 }
1534
1535 fn held_bytes(&self) -> u64 {
1541 fn table<K, V, S>(map: &HashMap<K, V, S>) -> usize {
1542 (map.capacity() * 8 / 7).next_power_of_two() * (size_of::<(K, V)>() + 1)
1543 }
1544 fn spilled<T>(values: &Vec<T>) -> usize {
1545 values.capacity() * size_of::<T>()
1546 }
1547 let raw = |blocks: &Vec<(usize, Vec<u8>)>| {
1548 spilled(blocks) + blocks.iter().map(|(_, block)| block.capacity()).sum::<usize>()
1549 };
1550 let bytes = table(&self.primary)
1551 + table(&self.collisions)
1552 + self.collisions.values().map(spilled).sum::<usize>()
1553 + spilled(&self.checks)
1554 + spilled(&self.ends)
1555 + spilled(&self.counts)
1556 + self.filling.capacity()
1557 + spilled(&self.grams)
1558 + raw(&self.waiting)
1559 + raw(&self.sample)
1560 + self.blocks.iter().map(Vec::capacity).sum::<usize>()
1561 + spilled(&self.placed);
1562 bytes as u64
1563 }
1564
1565 fn recharge(&mut self, profile: Option<&LoadProfile>) -> (u64, u64) {
1568 let before = self.charged;
1569 let now = self.held_bytes();
1570 if let Some(profile) = profile {
1571 if now >= before {
1572 profile.hold(now - before);
1573 } else {
1574 profile.release(before - now);
1575 }
1576 }
1577 self.charged = now;
1578 (before, now)
1579 }
1580
1581 fn demote(&mut self) {
1589 if self.demoted {
1590 return;
1591 }
1592 self.seal_rest();
1593 self.release_lookup();
1594 self.demoted = true;
1595 }
1596
1597 fn release_lookup(&mut self) {
1604 self.primary = HashMap::default();
1605 self.collisions = HashMap::default();
1606 self.checks = Vec::new();
1607 self.sample = Vec::new();
1608 self.filling = Vec::new();
1609 }
1610
1611 fn encoded(&self) -> usize {
1613 self.placed.len() + self.blocks.len()
1614 }
1615
1616 #[cfg(test)]
1617 fn code(&mut self, text: &str) -> Result<u32> {
1618 let bytes = text.as_bytes();
1619 self.code_hashed(bytes, checksum(bytes), seeded_checksum(bytes, DICTIONARY_CHECK_SEED))
1620 }
1621
1622 fn code_hashed(&mut self, text: &[u8], hash: u64, check: u64) -> Result<u32> {
1628 if let Some(&code) = self.primary.get(&hash) {
1629 if self.checks.get(code as usize) == Some(&check) {
1630 return Ok(code);
1631 }
1632 if let Some(codes) = self.collisions.get(&hash) {
1633 if let Some(code) =
1634 codes.iter().copied().find(|&code| self.checks[code as usize] == check)
1635 {
1636 return Ok(code);
1637 }
1638 }
1639 let code = self.insert(text, check)?;
1640 self.collisions.entry(hash).or_default().push(code);
1641 return Ok(code);
1642 }
1643 let code = self.insert(text, check)?;
1644 self.primary.insert(hash, code);
1645 Ok(code)
1646 }
1647
1648 fn insert(&mut self, text: &[u8], check: u64) -> Result<u32> {
1649 if self.demoted {
1650 return Err(Error::internal("a value was coded against a demoted dictionary"));
1651 }
1652 let code = u32::try_from(self.ends.len())
1653 .map_err(|_| invalid("global dictionary has too many values"))?;
1654 self.filling.extend_from_slice(text);
1655 self.ends.push(
1656 u32::try_from(self.filling.len())
1657 .map_err(|_| invalid("a global dictionary value exceeds 4 GiB"))?,
1658 );
1659 self.checks.push(check);
1660 self.counts.push(0);
1661 if self.ends.len() % TEXT_PAYLOAD_VALUES == 0 {
1662 self.seal();
1663 }
1664 Ok(code)
1665 }
1666
1667 fn seal(&mut self) {
1673 let at = self.ends.len().div_ceil(TEXT_PAYLOAD_VALUES) - 1;
1674 let bytes = std::mem::take(&mut self.filling);
1675 if at % self.stride == 0 {
1676 self.sample.push((at, bytes.clone()));
1677 if self.sample.len() > PAYLOAD_SAMPLE_BLOCKS {
1678 self.stride *= 2;
1679 let stride = self.stride;
1680 self.sample.retain(|(at, _)| at % stride == 0);
1681 }
1682 }
1683 self.waiting.push((at, bytes));
1684 }
1685
1686 fn slices<'a>(&self, at: usize, bytes: &'a [u8]) -> Vec<&'a [u8]> {
1688 block_values(self.block_ends(at), bytes)
1689 }
1690
1691 fn block_ends(&self, at: usize) -> &[u32] {
1693 let first = (at * TEXT_PAYLOAD_VALUES).min(self.ends.len());
1694 let last = (first + TEXT_PAYLOAD_VALUES).min(self.ends.len());
1695 &self.ends[first..last]
1696 }
1697
1698 fn hand_out(&mut self, column: usize) -> Vec<Unencoded> {
1705 let Some(shape) = &self.shape else { return Vec::new() };
1706 let waiting = std::mem::take(&mut self.waiting);
1707 waiting
1708 .into_iter()
1709 .map(|(at, bytes)| Unencoded {
1710 column,
1711 at,
1712 ends: self.block_ends(at).to_vec(),
1713 bytes,
1714 shape: shape.clone(),
1715 })
1716 .collect()
1717 }
1718
1719 fn take_back(&mut self, at: usize, block: EncodedBlock) -> Result<()> {
1722 if at < self.encoded() || self.early.insert(at, block).is_some() {
1723 return Err(Error::internal("a dictionary block came back twice"));
1724 }
1725 while let Some(block) = self.early.remove(&self.encoded()) {
1726 self.push_block(block);
1727 }
1728 Ok(())
1729 }
1730
1731 fn push_block(&mut self, (bytes, grams): EncodedBlock) {
1733 self.blocks.push(bytes);
1734 self.grams.push(*grams);
1735 }
1736
1737 fn settle(&mut self) -> Result<()> {
1745 if self.sample.len() < PAYLOAD_SAMPLE_BLOCKS {
1746 return Ok(());
1747 }
1748 self.settle_on_sample()
1749 }
1750
1751 fn settle_rest(&mut self) -> Result<()> {
1759 if self.shape.is_some() || self.sample.is_empty() {
1760 return Ok(());
1761 }
1762 self.settle_on_sample()
1763 }
1764
1765 fn settle_on_sample(&mut self) -> Result<()> {
1766 let complete = self.ends.len() / TEXT_PAYLOAD_VALUES;
1767 if self.shape.is_some() && complete < self.settled.saturating_mul(4) {
1768 return Ok(());
1769 }
1770 let sample =
1771 self.sample.iter().map(|(at, bytes)| self.slices(*at, bytes)).collect::<Vec<_>>();
1772 self.shape = Some(string::with_symbols(settle_shape(&sample)?, &sample));
1773 self.settled = complete;
1774 Ok(())
1775 }
1776
1777 fn seal_rest(&mut self) {
1779 if !self.demoted && self.ends.len() % TEXT_PAYLOAD_VALUES != 0 {
1783 self.seal();
1784 }
1785 }
1786
1787 fn encode_waiting(&self, at: usize) -> Result<EncodedBlock> {
1790 let (block, bytes) = &self.waiting[at];
1791 let values = self.slices(*block, bytes);
1792 let encoded = match &self.shape {
1793 Some(shape) => string::encode_with(&values, shape)?,
1794 None => string::encode(&values)?,
1795 };
1796 Ok((encoded, block_grams(&values)))
1797 }
1798
1799 #[cfg(test)]
1801 fn finish_blocks(&mut self) -> Result<()> {
1802 self.seal_rest();
1803 let made = (0..self.waiting.len())
1804 .map(|at| self.encode_waiting(at))
1805 .collect::<Result<Vec<_>>>()?;
1806 for ((at, _), block) in std::mem::take(&mut self.waiting).into_iter().zip(made) {
1807 if self.encoded() != at {
1808 return Err(Error::internal("a dictionary block was encoded out of order"));
1809 }
1810 self.push_block(block);
1811 }
1812 Ok(())
1813 }
1814
1815 fn decoded(&self, file: Option<&dyn rudb_io::File>) -> Result<(Vec<u8>, Vec<u64>)> {
1833 let count = self.placed.len() + self.blocks.len();
1834 if count != self.values().div_ceil(TEXT_PAYLOAD_VALUES) {
1835 return Err(invalid("global dictionary blocks do not cover its values"));
1836 }
1837 let mut bases = Vec::with_capacity(count);
1838 let mut total = 0_usize;
1839 for block in 0..count {
1840 bases.push(total as u64);
1841 let last = ((block + 1) * TEXT_PAYLOAD_VALUES).min(self.values()) - 1;
1842 total = total
1843 .checked_add(self.ends[last] as usize)
1844 .ok_or_else(|| invalid("global dictionary does not fit in memory"))?;
1845 }
1846 let mut flat = vec![0_u8; total];
1847 let mut outs = Vec::with_capacity(count);
1848 let mut rest = flat.as_mut_slice();
1849 for block in 0..count {
1850 let end = bases.get(block + 1).map_or(total, |&base| base as usize);
1851 let (out, after) = rest.split_at_mut(end - bases[block] as usize);
1852 outs.push((block, out));
1853 rest = after;
1854 }
1855 let one = |run: &mut [(usize, &mut [u8])]| -> Result<()> {
1856 let mut stored = Vec::new();
1857 for (block, out) in run {
1858 let encoded = match self.placed.get(*block) {
1859 Some(place) => {
1860 let file = file.ok_or_else(|| {
1861 Error::internal("a written dictionary block has no file")
1862 })?;
1863 let length = usize::try_from(place.length).map_err(|_| {
1864 invalid("global dictionary block does not fit in memory")
1865 })?;
1866 stored.resize(length, 0);
1867 read_at(file, place.start, &mut stored)?;
1868 if checksum(&stored) != place.hash {
1869 return Err(invalid(
1870 "a global dictionary block did not read back as written",
1871 ));
1872 }
1873 stored.as_slice()
1874 }
1875 None => &self.blocks[*block - self.placed.len()],
1876 };
1877 let decoded = string::decode_flat(encoded)?;
1878 if decoded.bytes().len() != out.len() {
1879 return Err(invalid(
1880 "a global dictionary block is not the length its ends say",
1881 ));
1882 }
1883 out.copy_from_slice(decoded.bytes());
1884 }
1885 Ok(())
1886 };
1887 let workers = close_workers().min(count / 16).max(1);
1890 if workers <= 1 {
1891 one(&mut outs)?;
1892 } else {
1893 let per = count.div_ceil(workers);
1894 std::thread::scope(|scope| {
1895 outs.chunks_mut(per)
1896 .map(|run| scope.spawn(|| one(run)))
1897 .collect::<Vec<_>>()
1898 .into_iter()
1899 .try_for_each(|handle| {
1900 handle.join().map_err(|_| {
1901 Error::internal("a global dictionary decode worker panicked")
1902 })?
1903 })
1904 })?;
1905 }
1906 drop(outs);
1907 Ok((flat, bases))
1908 }
1909
1910 fn value_span(ends: &[u32], bases: &[u64], code: usize) -> (usize, usize) {
1915 let Some(&base) = bases.get(code / TEXT_PAYLOAD_VALUES) else { return (0, 0) };
1916 let Some(&end) = ends.get(code) else { return (0, 0) };
1917 let base = base as usize;
1918 let from = if code % TEXT_PAYLOAD_VALUES == 0 { 0 } else { ends[code - 1] as usize };
1919 (base + from, base + end as usize)
1920 }
1921
1922 fn ranked_with_values(&self, file: Option<&dyn rudb_io::File>) -> Result<RankedDictionary> {
1942 let (flat, bases) = self.decoded(file)?;
1943 let value = |code: u32| {
1944 let (from, to) = Self::value_span(&self.ends, &bases, code as usize);
1945 flat.get(from..to).unwrap_or_default()
1946 };
1947 let mut codes = (0..self.values() as u32).collect::<Vec<_>>();
1948 sort_by_value_across(&mut codes, value, close_workers());
1949 let order = codes.into_iter().map(|code| (head(value(code)), code)).collect();
1950 Ok((order, flat, bases))
1951 }
1952
1953 #[cfg(test)]
1954 fn ranked(&self, file: Option<&dyn rudb_io::File>) -> Result<Vec<(u64, u32)>> {
1955 self.ranked_with_values(file).map(|(order, _, _)| order)
1956 }
1957}
1958
1959#[derive(Debug)]
1967pub struct Writer {
1968 file: Box<dyn rudb_io::File>,
1971 at: u64,
1979 written_back: u64,
1981 table: Table,
1982 generation: u64,
1983 order: Vec<((u64, u64), (u64, u64))>,
1986 next_order: u64,
1987 dictionaries: Vec<Option<GlobalDictionary>>,
1988 coded: Arc<prepare::Coding>,
1991 gathers: Vec<Option<stats::Gather>>,
1997 lent: Option<Arc<Lent>>,
2000 pending: Vec<PendingChunk>,
2001 closed: Vec<Entry>,
2003 views: Vec<ViewEntry>,
2008 profile: Option<Arc<LoadProfile>>,
2014}
2015
2016#[derive(Debug)]
2024struct PendingChunk {
2025 order: (u64, u64),
2026 chunk: Chunk,
2027}
2028
2029#[derive(Debug, Clone, Copy)]
2035struct Part {
2036 order: (u64, u64),
2037 rows: usize,
2038 footprint: usize,
2039}
2040
2041impl Part {
2042 fn of(pending: &PendingChunk) -> Self {
2043 Self {
2044 order: pending.order,
2045 rows: pending.chunk.len(),
2046 footprint: pending.chunk.footprint(),
2047 }
2048 }
2049}
2050
2051#[derive(Debug, Default)]
2057struct ColumnStripe {
2058 pages: Vec<Vec<u8>>,
2059 sums: Vec<u64>,
2062 codes: Vec<Option<Vec<u32>>>,
2063 sieves: Vec<Option<Sieve>>,
2064 ranges: Vec<Range>,
2065}
2066
2067fn coded_type(ty: &LogicalType) -> bool {
2075 matches!(ty, LogicalType::Varchar | LogicalType::Blob)
2076}
2077
2078fn dictionary_tag(ty: &LogicalType) -> u8 {
2085 if ty == &LogicalType::Blob { 2 } else { 1 }
2086}
2087
2088fn weight(ty: &LogicalType) -> usize {
2096 match ty {
2097 LogicalType::Varchar | LogicalType::Blob | LogicalType::Bit => 64,
2098 LogicalType::HugeInt
2099 | LogicalType::UHugeInt
2100 | LogicalType::Uuid
2101 | LogicalType::Interval => 16,
2102 LogicalType::BigInt
2103 | LogicalType::UBigInt
2104 | LogicalType::Timestamp
2105 | LogicalType::Time
2106 | LogicalType::TimeTz
2107 | LogicalType::TimestampTz
2108 | LogicalType::TimestampS
2109 | LogicalType::TimestampMs
2110 | LogicalType::TimestampNs
2111 | LogicalType::Double
2112 | LogicalType::Decimal { .. } => 8,
2113 LogicalType::Integer | LogicalType::UInteger | LogicalType::Date | LogicalType::Float => 4,
2114 LogicalType::SmallInt | LogicalType::USmallInt => 2,
2115 _ => 1,
2116 }
2117}
2118
2119pub const STRIPE_PARTS: usize = 64;
2126
2127const DICTIONARY_DECIDE_ROWS: usize = 4_096;
2135
2136const DICTIONARY_DISTINCT_IN_TEN: usize = 9;
2152
2153const INDEX_ENTRY: usize = size_of::<u32>() + size_of::<u64>();
2155
2156fn index_section(parts: usize) -> Result<usize> {
2158 parts
2159 .checked_mul(INDEX_ENTRY)
2160 .and_then(|bytes| bytes.checked_add(size_of::<u64>()))
2161 .ok_or_else(|| invalid("index page length overflow"))
2162}
2163
2164impl Writer {
2165 pub fn open(
2184 path: impl AsRef<Path>,
2185 name: impl Into<String>,
2186 fields: Vec<Field>,
2187 ) -> Result<Self> {
2188 Self::open_in(&RealFilesystem::new(), path, name, fields)
2189 }
2190
2191 pub fn open_in(
2198 fs: &dyn Filesystem,
2199 path: impl AsRef<Path>,
2200 name: impl Into<String>,
2201 fields: Vec<Field>,
2202 ) -> Result<Self> {
2203 for field in &fields {
2204 type_tag(&field.ty)?;
2205 }
2206 let name = name.into();
2207 let file = fs.open(path.as_ref(), OpenMode::ReadWrite)?;
2208 let size = file.len()?;
2209 let (slot, bytes, _) = committed_slot(&*file, size)?;
2210 let (mut closed, views) = decode_catalog(&bytes, size)?;
2211 if let Some(at) = closed.iter().position(|held| held.name == name) {
2222 if closed[at].rows > 0 {
2223 return Err(invalid("two tables in one native file have the same name"));
2224 }
2225 closed.remove(at);
2226 }
2227 let generation = slot
2232 .generation
2233 .checked_add(1)
2234 .ok_or_else(|| invalid("native file generation overflow"))?;
2235 Ok(Self {
2236 file,
2237 at: size,
2240 written_back: size,
2241 dictionaries: fields
2242 .iter()
2243 .map(|field| coded_type(&field.ty).then(GlobalDictionary::new))
2244 .collect(),
2245 coded: Arc::new(prepare::Coding::new(fields.iter().map(|field| coded_type(&field.ty)))),
2246 gathers: fields.iter().map(|field| stats::Gather::new(&field.ty, generation)).collect(),
2247 lent: None,
2248 table: Table {
2249 name,
2250 dictionaries: vec![None; fields.len()],
2251 dictionary_payloads: Vec::new(),
2252 demoted: Vec::new(),
2253 distincts: vec![None; fields.len()],
2254 fields,
2255 stripes: Vec::new(),
2256 rows: 0,
2257 frequencies: Vec::new(),
2258 pair_frequencies: Vec::new(),
2259 frequency_texts: Vec::new(),
2260 host_groups: None,
2261 clustering: None,
2262 constraints: Constraints::default(),
2263 generation,
2264 sections: Vec::new(),
2265 },
2266 generation,
2267 order: Vec::new(),
2268 next_order: 0,
2269 pending: Vec::with_capacity(STRIPE_PARTS),
2270 closed,
2271 views,
2272 profile: None,
2273 })
2274 }
2275
2276 pub fn create(
2282 path: impl AsRef<Path>,
2283 name: impl Into<String>,
2284 fields: Vec<Field>,
2285 ) -> Result<Self> {
2286 Self::create_in(&RealFilesystem::new(), path, name, fields)
2287 }
2288
2289 pub fn create_in(
2299 fs: &dyn Filesystem,
2300 path: impl AsRef<Path>,
2301 name: impl Into<String>,
2302 fields: Vec<Field>,
2303 ) -> Result<Self> {
2304 for field in &fields {
2305 type_tag(&field.ty)?;
2306 }
2307 let file = fs.open(path.as_ref(), OpenMode::CreateNew)?;
2308 let mut header = [0; HEADER as usize];
2309 header[..8].copy_from_slice(MAGIC);
2310 header[8..12].copy_from_slice(&FORMAT.to_le_bytes());
2311 file.write_at(0, &header)?;
2312 Ok(Self {
2313 file,
2314 at: HEADER,
2315 written_back: HEADER,
2316 dictionaries: fields
2317 .iter()
2318 .map(|field| coded_type(&field.ty).then(GlobalDictionary::new))
2319 .collect(),
2320 coded: Arc::new(prepare::Coding::new(fields.iter().map(|field| coded_type(&field.ty)))),
2321 gathers: fields.iter().map(|field| stats::Gather::new(&field.ty, 1)).collect(),
2322 lent: None,
2323 table: Table {
2324 name: name.into(),
2325 dictionaries: vec![None; fields.len()],
2326 dictionary_payloads: Vec::new(),
2327 demoted: Vec::new(),
2328 distincts: vec![None; fields.len()],
2329 fields,
2330 stripes: Vec::new(),
2331 rows: 0,
2332 frequencies: Vec::new(),
2333 pair_frequencies: Vec::new(),
2334 frequency_texts: Vec::new(),
2335 host_groups: None,
2336 clustering: None,
2337 constraints: Constraints::default(),
2338 generation: 1,
2339 sections: Vec::new(),
2340 },
2341 generation: 1,
2342 order: Vec::new(),
2343 next_order: 0,
2344 pending: Vec::with_capacity(STRIPE_PARTS),
2345 closed: Vec::new(),
2346 views: Vec::new(),
2347 profile: None,
2348 })
2349 }
2350
2351 pub fn empty(path: impl AsRef<Path>, views: &[ViewEntry]) -> Result<()> {
2373 let file = RealFilesystem::new().open(path.as_ref(), OpenMode::CreateNew)?;
2374 let mut header = [0; HEADER as usize];
2375 header[..8].copy_from_slice(MAGIC);
2376 header[8..12].copy_from_slice(&FORMAT.to_le_bytes());
2377 file.write_at(0, &header)?;
2378 let catalog = encode_catalog(&[], views)?;
2379 file.write_at(HEADER, &catalog)?;
2380 file.sync()?;
2384 let slot = Slot {
2385 offset: HEADER,
2386 length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
2387 generation: 1,
2388 hash: checksum(&catalog),
2389 };
2390 file.write_at(slot_offset(1), &slot.bytes())?;
2391 file.sync()?;
2392 Ok(())
2393 }
2394
2395 pub fn next(mut self, name: impl Into<String>, fields: Vec<Field>) -> Result<Self> {
2406 for field in &fields {
2407 type_tag(&field.ty)?;
2408 }
2409 let name = name.into();
2410 let entry = self.close()?;
2411 if self.closed.iter().chain(std::iter::once(&entry)).any(|held| held.name == name) {
2412 return Err(invalid("two tables in one native file have the same name"));
2413 }
2414 let Self { file, at, generation, mut closed, views, .. } = self;
2415 closed.push(entry);
2416 Ok(Self {
2417 file,
2418 written_back: at,
2419 at,
2420 generation,
2421 closed,
2422 views,
2423 profile: None,
2424 dictionaries: fields
2425 .iter()
2426 .map(|field| coded_type(&field.ty).then(GlobalDictionary::new))
2427 .collect(),
2428 coded: Arc::new(prepare::Coding::new(fields.iter().map(|field| coded_type(&field.ty)))),
2429 gathers: fields.iter().map(|field| stats::Gather::new(&field.ty, generation)).collect(),
2430 lent: None,
2431 table: Table {
2432 name,
2433 dictionaries: vec![None; fields.len()],
2434 dictionary_payloads: Vec::new(),
2435 demoted: Vec::new(),
2436 distincts: vec![None; fields.len()],
2437 fields,
2438 stripes: Vec::new(),
2439 rows: 0,
2440 frequencies: Vec::new(),
2441 pair_frequencies: Vec::new(),
2442 frequency_texts: Vec::new(),
2443 host_groups: None,
2444 clustering: None,
2445 constraints: Constraints::default(),
2446 generation,
2447 sections: Vec::new(),
2448 },
2449 order: Vec::new(),
2450 next_order: 0,
2451 pending: Vec::with_capacity(STRIPE_PARTS),
2452 })
2453 }
2454
2455 #[must_use]
2465 pub fn with_views(mut self, views: Vec<ViewEntry>) -> Self {
2466 self.views = views;
2467 self
2468 }
2469
2470 #[must_use]
2476 pub fn with_profile(mut self, profile: Arc<LoadProfile>) -> Self {
2477 self.profile = Some(profile);
2478 self
2479 }
2480
2481 #[must_use]
2485 pub fn with_dictionary_cap(self, bytes: u64) -> Self {
2486 self.coded.cap(bytes);
2487 self
2488 }
2489
2490 pub fn declare(mut self, clustering: Clustering) -> Result<Self> {
2505 self.table.clustering = Some(Clustering::new(
2508 clustering.columns().to_vec(),
2509 clustering.width(),
2510 &self.table.fields,
2511 )?);
2512 Ok(self)
2513 }
2514
2515 pub fn constrain(mut self, constraints: Constraints) -> Result<Self> {
2523 let width = self.table.fields.len();
2524 let fits = |columns: &[u16]| {
2525 !columns.is_empty() && columns.iter().all(|&column| usize::from(column) < width)
2526 };
2527 if !constraints.keys.iter().all(|(columns, _)| fits(columns))
2528 || !constraints.foreign.iter().all(|foreign| {
2529 fits(&foreign.columns) && foreign.referenced.len() == foreign.columns.len()
2530 })
2531 {
2532 return Err(invalid("a constraint names a column the table does not have"));
2533 }
2534 self.table.constraints = constraints;
2535 Ok(self)
2536 }
2537
2538 fn put(&mut self, bytes: &[u8]) -> Result<()> {
2543 self.file.write_at(self.at, bytes)?;
2544 self.at = self
2545 .at
2546 .checked_add(bytes.len() as u64)
2547 .ok_or_else(|| invalid("native file length overflow"))?;
2548 if self.at - self.written_back >= WRITEBACK_STRETCH {
2549 self.file.start_writeback(self.written_back, self.at - self.written_back);
2550 self.written_back = self.at;
2551 }
2552 Ok(())
2553 }
2554
2555 pub fn append(&mut self, chunk: &Chunk) -> Result<()> {
2561 let order = (self.next_order, 0);
2562 self.next_order = self.next_order.saturating_add(1);
2563 self.append_at(order, chunk)
2564 }
2565
2566 pub fn append_at(&mut self, order: (u64, u64), chunk: &Chunk) -> Result<()> {
2577 if chunk.is_empty() {
2578 return Ok(());
2579 }
2580 self.admit(chunk)?;
2581 if self.pending.last().is_some_and(|last| last.order > order) {
2582 self.flush_pending()?;
2583 }
2584 self.pending.push(PendingChunk { order, chunk: chunk.clone() });
2589 if self.pending.len() == STRIPE_PARTS {
2590 self.flush_pending()?;
2591 }
2592 Ok(())
2593 }
2594
2595 pub fn append_stripe(&mut self, parts: Vec<((u64, u64), Chunk)>) -> Result<()> {
2611 if parts.len() > STRIPE_PARTS {
2612 return Err(invalid("a stripe was handed more parts than it holds"));
2613 }
2614 self.flush_pending()?;
2617 for (order, chunk) in parts {
2618 if chunk.is_empty() {
2619 continue;
2620 }
2621 self.admit(&chunk)?;
2622 self.pending.push(PendingChunk { order, chunk });
2623 }
2624 self.flush_pending()
2625 }
2626
2627 fn admit(&mut self, chunk: &Chunk) -> Result<()> {
2629 if chunk.width() != self.table.fields.len() {
2630 return Err(invalid("chunk width differs from table schema"));
2631 }
2632 for (index, field) in self.table.fields.iter().enumerate() {
2633 if chunk.column(index)?.logical_type() != &field.ty {
2634 return Err(invalid("chunk type differs from table schema"));
2635 }
2636 }
2637 self.table.rows = self
2638 .table
2639 .rows
2640 .checked_add(chunk.len())
2641 .ok_or_else(|| invalid("row count overflow"))?;
2642 Ok(())
2643 }
2644
2645 fn encode_pages(columns: &[&Vector]) -> Result<ColumnStripe> {
2647 let mut stripe = ColumnStripe {
2648 pages: Vec::with_capacity(columns.len()),
2649 sums: Vec::with_capacity(columns.len()),
2650 codes: Vec::with_capacity(columns.len()),
2651 sieves: Vec::with_capacity(columns.len()),
2652 ranges: Vec::with_capacity(columns.len()),
2653 };
2654 let mut settling = Settling::default();
2655 for &column in columns {
2656 Self::encode_page(&mut stripe, &mut settling, column)?;
2657 }
2658 Ok(stripe)
2659 }
2660
2661 fn encode_page(
2664 stripe: &mut ColumnStripe,
2665 settling: &mut Settling,
2666 column: &Vector,
2667 ) -> Result<()> {
2668 let bytes = encode(column, settling)?;
2669 if bytes.len() > MAX_PAGE {
2670 return Err(invalid("column page exceeds the configured bound"));
2671 }
2672 let range = Range::of(column);
2675 let sieve =
2686 Sieve::of(column, &range, SIEVE_BUDGET).filter(|sieve| sieve.len() < bytes.len());
2687 stripe.sums.push(checksum(&bytes));
2688 stripe.pages.push(bytes);
2689 stripe.codes.push(None);
2690 stripe.sieves.push(sieve);
2691 stripe.ranges.push(range);
2692 Ok(())
2693 }
2694
2695 fn place_blocks(&mut self) -> Result<()> {
2700 if let Some(lent) = self.lent.clone() {
2701 return self.place_lent_blocks(&lent);
2702 }
2703 let mut dictionaries = std::mem::take(&mut self.dictionaries);
2704 let placed = dictionaries.iter_mut().flatten().try_for_each(|dictionary| {
2705 for block in std::mem::take(&mut dictionary.blocks) {
2706 let start = self.at;
2707 self.put(&block)?;
2708 dictionary.placed.push(Placed {
2709 start,
2710 length: block.len() as u64,
2711 hash: checksum(&block),
2712 });
2713 }
2714 Ok(())
2715 });
2716 self.dictionaries = dictionaries;
2717 placed
2718 }
2719
2720 fn place_lent_blocks(&mut self, lent: &Lent) -> Result<()> {
2726 for column in lent.columns() {
2727 let Ok(mut held) = column.try_lock() else { continue };
2728 let Some(dictionary) = held.dictionary.as_mut() else { continue };
2729 for block in std::mem::take(&mut dictionary.blocks) {
2730 let start = self.at;
2731 self.put(&block)?;
2732 dictionary.placed.push(Placed {
2733 start,
2734 length: block.len() as u64,
2735 hash: checksum(&block),
2736 });
2737 }
2738 }
2739 Ok(())
2740 }
2741
2742 fn reclaim(&mut self) -> Result<()> {
2746 let Some(lent) = self.lent.take() else { return Ok(()) };
2747 let (dictionaries, gathers) = lent.reclaim()?;
2748 self.dictionaries = dictionaries;
2749 self.gathers = gathers;
2750 Ok(())
2751 }
2752
2753 fn flush_pending(&mut self) -> Result<()> {
2758 if self.pending.is_empty() {
2759 return Ok(());
2760 }
2761 let held = std::mem::take(&mut self.pending);
2762 let prepared = self.preparer().prepare_held(held)?;
2763 let merged = self.merge_held(prepared)?;
2764 let paged = merged.pages()?;
2765 self.write_paged(paged)
2766 }
2767
2768 fn write_stripe(&mut self, held: &[Part], encoded: Vec<ColumnStripe>) -> Result<()> {
2770 let width = self.table.fields.len();
2771 let parts = held.len();
2772 if encoded.len() != width {
2773 return Err(Error::internal("a stripe came to the writer with the wrong columns"));
2774 }
2775 let profile = self.profile.clone();
2776 if let Some(profile) = &profile {
2777 let rows = held.iter().map(|part| part.rows as u64).sum();
2778 let raw = held.iter().map(|part| part.footprint as u64).sum();
2779 let pages =
2780 encoded.iter().flat_map(|stripe| &stripe.pages).map(|page| page.len() as u64).sum();
2781 profile.moved(Stage::Pages, raw, pages, rows);
2782 }
2783 let timing = profile.as_deref().map(|profile| profile.span(Stage::Dictionary));
2786 let before = self.at;
2787 self.place_blocks()?;
2788 drop(timing);
2789 if let Some(profile) = &profile {
2790 profile.moved(Stage::Dictionary, 0, self.at - before, 0);
2791 }
2792 let timing = profile.as_deref().map(|profile| profile.span(Stage::Write));
2793 let before = self.at;
2794 let mut pages = Vec::with_capacity(width);
2795 let mut memberships = vec![None; width];
2796 let mut ranges = Vec::with_capacity(width);
2797 let mut index = Vec::with_capacity(width.saturating_mul(index_section(parts)?));
2798 let start = self.at;
2801 let mut out = Vec::with_capacity(width.saturating_mul(parts));
2802 for stripe in &encoded {
2803 let offset = self.at;
2804 let section = index.len();
2805 let mut length = 0_usize;
2806 if stripe.sums.len() != stripe.pages.len() {
2807 return Err(Error::internal("a stripe's pages came without their checksums"));
2808 }
2809 for (bytes, &sum) in stripe.pages.iter().zip(&stripe.sums) {
2810 put_u32(
2811 &mut index,
2812 u32::try_from(bytes.len()).map_err(|_| invalid("part length overflow"))?,
2813 );
2814 put_u64(&mut index, sum);
2815 out.push(bytes.as_slice());
2816 length = length
2817 .checked_add(bytes.len())
2818 .ok_or_else(|| invalid("column page length overflow"))?;
2819 }
2820 let hash = checksum(&index[section..]);
2821 put_u64(&mut index, hash);
2822 if length > MAX_PAGE {
2823 return Err(invalid("column page exceeds the configured bound"));
2824 }
2825 self.at = self
2826 .at
2827 .checked_add(length as u64)
2828 .ok_or_else(|| invalid("native file length overflow"))?;
2829 pages.push(Span {
2830 offset,
2831 length: u32::try_from(length).map_err(|_| invalid("page length overflow"))?,
2832 });
2833 ranges.push(merged_range(stripe.ranges.iter().cloned()));
2834 }
2835 self.file.write_parts_at(start, &out)?;
2836 drop(out);
2837 for (membership, stripe) in memberships.iter_mut().zip(&encoded) {
2838 if stripe.codes.iter().all(Option::is_none) {
2839 continue;
2840 }
2841 let lists = stripe
2842 .codes
2843 .iter()
2844 .map(|codes| codes.clone().unwrap_or_default())
2845 .collect::<Vec<_>>();
2846 let bytes = encode_membership(&merged_codes(lists));
2847 let offset = self.at;
2848 self.put(&bytes)?;
2849 *membership = Some(Page {
2850 offset,
2851 length: u32::try_from(bytes.len())
2852 .map_err(|_| invalid("membership page length overflow"))?,
2853 hash: checksum(&bytes),
2854 });
2855 }
2856 let mut sieves = vec![None; width];
2857 for (page, stripe) in sieves.iter_mut().zip(&encoded) {
2858 if stripe.sieves.iter().all(Option::is_none) {
2859 continue;
2860 }
2861 let bytes = encode_sieves(stripe.sieves.iter())?;
2862 let offset = self.at;
2863 self.put(&bytes)?;
2864 *page = Some(Page {
2865 offset,
2866 length: u32::try_from(bytes.len())
2867 .map_err(|_| invalid("sieve page length overflow"))?,
2868 hash: checksum(&bytes),
2869 });
2870 }
2871 let mut part_ranges = vec![None; width];
2877 if parts > 1 {
2878 for ((page, stripe), span) in part_ranges.iter_mut().zip(&encoded).zip(&pages) {
2879 let bytes = encode_part_ranges(&stripe.ranges)?;
2880 if bytes.len() >= span.length as usize {
2881 continue;
2882 }
2883 let offset = self.at;
2884 self.put(&bytes)?;
2885 *page = Some(Page {
2886 offset,
2887 length: u32::try_from(bytes.len())
2888 .map_err(|_| invalid("part range page length overflow"))?,
2889 hash: checksum(&bytes),
2890 });
2891 }
2892 }
2893 let offset = self.at;
2894 self.put(&index)?;
2895 let index = Span {
2896 offset,
2897 length: u32::try_from(index.len())
2898 .map_err(|_| invalid("index page length overflow"))?,
2899 };
2900 let mut rows = 0_usize;
2901 let mut lengths = Vec::with_capacity(parts);
2902 let mut span = None;
2903 for part in held {
2904 rows = rows.checked_add(part.rows).ok_or_else(|| invalid("row count overflow"))?;
2905 lengths.push(u32::try_from(part.rows).map_err(|_| invalid("part row count overflow"))?);
2906 span = Some(span.map_or((part.order, part.order), |(first, _)| (first, part.order)));
2907 }
2908 self.order.push(span.ok_or_else(|| invalid("a stripe was flushed with no parts"))?);
2909 self.table.stripes.push(Stripe {
2910 rows,
2911 parts: lengths,
2912 index,
2913 pages,
2914 memberships: Pages::from_slots(memberships)?,
2915 sieves: Pages::from_slots(sieves)?,
2916 part_ranges: Pages::from_slots(part_ranges)?,
2917 zone: Zone::from_ranges(ranges),
2918 });
2919 drop(timing);
2920 if let Some(profile) = &profile {
2921 profile.moved(Stage::Write, 0, self.at - before, rows as u64);
2922 }
2923 Ok(())
2924 }
2925
2926 fn numeric_frequency(
2946 &self,
2947 column: usize,
2948 counted: bool,
2949 dense: Option<(u64, usize)>,
2950 ) -> Result<(Option<FrequencySummary>, Option<u64>)> {
2951 let signed = match self.table.fields[column].ty {
2952 LogicalType::TinyInt
2953 | LogicalType::SmallInt
2954 | LogicalType::Integer
2955 | LogicalType::BigInt
2956 | LogicalType::Date
2957 | LogicalType::Timestamp => true,
2958 LogicalType::UTinyInt
2959 | LogicalType::USmallInt
2960 | LogicalType::UInteger
2961 | LogicalType::UBigInt => false,
2962 _ => return Ok((None, None)),
2963 };
2964 let value_of = |bits: Option<u64>| match bits {
2965 None => FrequencyValue::Null,
2966 Some(bits) => integer_value(bits, signed),
2967 };
2968 let tallied = self
2973 .gathers
2974 .get(column)
2975 .and_then(Option::as_ref)
2976 .filter(|gather| gather.rows() == self.table.rows as u64)
2977 .and_then(stats::Gather::frequencies)
2978 .and_then(|(values, nulls)| {
2979 let entries = values
2980 .iter()
2981 .map(|(value, count)| {
2982 let value = value_of(Some(frequency_bits(value)?));
2983 Some(FrequencyEntry { value, count: *count })
2984 })
2985 .chain((nulls != 0).then_some(Some(FrequencyEntry {
2986 value: FrequencyValue::Null,
2987 count: nulls,
2988 })))
2989 .collect::<Option<Vec<_>>>()?;
2990 Some((entries, values.len() as u64))
2991 });
2992 let exact = match (&tallied, counted) {
2996 (None, true) => self.exact_frequency(column, signed, dense)?,
2997 _ => None,
2998 };
2999 let (mut entries, decrements, distinct_count) = match (tallied, exact) {
3000 (Some((entries, distinct)), _) => (entries, 0, Some(distinct)),
3001 (None, Some((Some(entries), distinct))) => (entries, 0, Some(distinct)),
3002 (None, Some((None, distinct))) => return Ok((None, Some(distinct))),
3003 (None, None) => {
3004 let mut first = Candidates::default();
3008 let mut run = Run::default();
3009 self.visit_numeric(column, signed, |_, bits| {
3010 if let Some((ended, times)) = run.push(bits) {
3011 first.add(ended, times);
3012 }
3013 })?;
3014 if let Some((bits, times)) = run.take() {
3015 first.add(bits, times);
3016 }
3017 let (nulls, decrements) = (first.nulls, first.decrements);
3020 let distinct_count = (decrements == 0).then_some(first.held as u64);
3021 let (exact, null_count) = if decrements == 0 {
3022 let exact = first
3023 .pairs()
3024 .map(|(bits, count)| (bits, u64::from(count)))
3025 .collect::<FrequencyMap<_>>();
3026 (exact, (nulls != 0).then_some(u64::from(nulls)))
3027 } else {
3028 let mut lower = first.pairs().map(|(_, count)| count).collect::<Vec<_>>();
3029 if nulls != 0 {
3030 lower.push(nulls);
3031 }
3032 lower.sort_unstable_by(|left, right| right.cmp(left));
3033 if lower.len() < FREQUENCY_BUILD_RANK
3034 || u64::from(lower[FREQUENCY_BUILD_RANK - 1]) <= decrements
3035 {
3036 return Ok((None, distinct_count));
3037 }
3038 let mut recounts = vec![0_u64; first.slots.len()];
3041 let mut null_count = (nulls != 0).then_some(0_u64);
3042 let mut recount = |bits: Option<u64>, times: u32| {
3043 let held = match bits {
3044 Some(bits) => first.position(bits).map(|at| &mut recounts[at]),
3045 None => null_count.as_mut(),
3046 };
3047 if let Some(count) = held {
3048 *count = count.saturating_add(u64::from(times));
3049 }
3050 };
3051 let mut run = Run::default();
3052 self.visit_numeric(column, signed, |_, bits| {
3053 if let Some((bits, times)) = run.push(bits) {
3054 recount(bits, times);
3055 }
3056 })?;
3057 if let Some((bits, times)) = run.take() {
3058 recount(bits, times);
3059 }
3060 let exact = first
3061 .slots
3062 .iter()
3063 .zip(&recounts)
3064 .filter(|(slot, _)| slot.count != 0)
3065 .map(|(slot, &count)| (slot.bits, count))
3066 .collect::<FrequencyMap<_>>();
3067 (exact, null_count)
3068 };
3069 let entries = exact
3070 .into_iter()
3071 .map(|(bits, count)| FrequencyEntry { value: value_of(Some(bits)), count })
3072 .chain(
3073 null_count
3074 .map(|count| FrequencyEntry { value: FrequencyValue::Null, count }),
3075 )
3076 .collect::<Vec<_>>();
3077 (entries, decrements, distinct_count)
3078 }
3079 };
3080 let mut omitted_max = keep_most_frequent(&mut entries).max(decrements);
3081 if omitted_max == 0 && entries.len() > 1 {
3085 let retained = entries.len().saturating_sub(1).min(2);
3086 omitted_max = entries[retained].count;
3087 entries.truncate(retained);
3088 }
3089 let kept_rows = entries.iter().try_fold(0_u64, |total, entry| {
3090 total.checked_add(entry.count).filter(|&total| total <= FREQUENCY_ORDINALS as u64)
3091 });
3092 let mut ordinals = Vec::new();
3093 let mut ordinal_entries = Vec::new();
3094 if let Some(kept_rows) = kept_rows {
3095 let mut kept = FrequencyMap::default();
3096 let mut null_kept = None;
3097 for (at, entry) in entries.iter().enumerate() {
3098 let at = u16::try_from(at)
3099 .map_err(|_| invalid("too many retained frequency entries"))?;
3100 match entry.value {
3101 FrequencyValue::Integer(value) => {
3102 kept.insert(value as u64, at);
3103 }
3104 FrequencyValue::Null => null_kept = Some(at),
3105 FrequencyValue::Code(_) => {}
3106 }
3107 }
3108 ordinals.reserve(usize::try_from(kept_rows).unwrap_or(FREQUENCY_ORDINALS));
3109 ordinal_entries.reserve(usize::try_from(kept_rows).unwrap_or(FREQUENCY_ORDINALS));
3110 self.visit_numeric(column, signed, |ordinal, bits| {
3111 let held = match bits {
3112 Some(bits) => kept.get(&bits).copied(),
3113 None => null_kept,
3114 };
3115 if let Some(entry) = held {
3116 ordinals.push(ordinal);
3117 ordinal_entries.push(entry);
3118 }
3119 })?;
3120 }
3121 Ok((
3122 Some(FrequencySummary { entries, omitted_max, ordinals, ordinal_entries }),
3123 distinct_count,
3124 ))
3125 }
3126
3127 fn dense_range(&self, gather: &stats::Gather, set: usize) -> Option<(u64, usize)> {
3134 let rows = self.table.rows;
3135 if gather.rows() != rows as u64 || u32::try_from(rows).is_err() {
3136 return None;
3137 }
3138 let (low, high) = gather.span()?;
3139 let len = usize::try_from(high.checked_sub(low)?.checked_add(1)?).ok()?;
3140 #[allow(clippy::cast_sign_loss, clippy::cast_possible_truncation)]
3141 let bits = low as u64;
3142 (len.checked_mul(size_of::<u32>())? <= set.max(1 << 20)).then_some((bits, len))
3143 }
3144
3145 fn exact_frequency(
3159 &self,
3160 column: usize,
3161 signed: bool,
3162 dense: Option<(u64, usize)>,
3163 ) -> Result<Option<(Option<Vec<FrequencyEntry>>, u64)>> {
3164 if let Some((low, len)) = dense {
3167 let mut counts = distinct::DenseCounts::new(low, len);
3168 let nulls =
3169 self.count_numeric(column, signed, |bits, times| counts.insert(bits, times))?;
3170 if let Some(distinct) = counts.count() {
3171 let Some(distinct) = distinct else { return Ok(None) };
3172 return Ok(Some(self.frequent_entries(signed, distinct, nulls, |visit| {
3173 counts.visit(visit);
3174 })));
3175 }
3176 }
3177 let mut set = distinct::ExactCounts::new();
3178 let nulls = self.count_numeric(column, signed, |bits, times| set.insert(bits, times))?;
3179 let Some(distinct) = set.count() else {
3180 return Ok(None);
3181 };
3182 Ok(Some(self.frequent_entries(signed, distinct, nulls, |visit| {
3183 set.visit(visit);
3184 })))
3185 }
3186
3187 fn count_numeric(
3190 &self,
3191 column: usize,
3192 signed: bool,
3193 mut add: impl FnMut(u64, u32),
3194 ) -> Result<u64> {
3195 let mut nulls = 0_u64;
3196 let mut run = Run::default();
3197 let mut take = |bits: Option<u64>, times: u32| match bits {
3198 Some(bits) => add(bits, times),
3199 None => nulls += u64::from(times),
3200 };
3201 self.visit_numeric(column, signed, |_, bits| {
3202 if let Some((bits, times)) = run.push(bits) {
3203 take(bits, times);
3204 }
3205 })?;
3206 if let Some((bits, times)) = run.take() {
3207 take(bits, times);
3208 }
3209 Ok(nulls)
3210 }
3211
3212 fn frequent_entries(
3215 &self,
3216 signed: bool,
3217 distinct: u64,
3218 nulls: u64,
3219 mut visit: impl FnMut(&mut dyn FnMut(u64, u64)),
3220 ) -> (Option<Vec<FrequencyEntry>>, u64) {
3221 let mut top = std::collections::BinaryHeap::with_capacity(FREQUENCY_ENTRIES + 2);
3223 let mut rank = |count: u64| {
3224 if top.len() <= FREQUENCY_ENTRIES {
3225 top.push(Reverse(count));
3226 } else if top.peek().is_some_and(|&Reverse(least)| count > least) {
3227 top.pop();
3228 top.push(Reverse(count));
3229 }
3230 };
3231 visit(&mut |_, count| rank(count));
3232 if nulls != 0 {
3233 rank(nulls);
3234 }
3235 let top = top.into_sorted_vec();
3236 let values = distinct + u64::from(nulls != 0);
3237 if values > FREQUENCY_CANDIDATES as u64 {
3238 let bound = self.table.rows as u64 / (FREQUENCY_CANDIDATES as u64 + 1);
3239 if top.get(FREQUENCY_BUILD_RANK - 1).is_none_or(|&Reverse(count)| count <= bound) {
3240 return (None, distinct);
3241 }
3242 }
3243 let least = top.get(FREQUENCY_ENTRIES).map_or(0, |&Reverse(count)| count);
3244 let mut entries = Vec::with_capacity(FREQUENCY_ENTRIES + 1);
3245 visit(&mut |bits, count| {
3246 if count >= least {
3247 entries.push(FrequencyEntry { value: integer_value(bits, signed), count });
3248 }
3249 });
3250 if nulls != 0 && nulls >= least {
3251 entries.push(FrequencyEntry { value: FrequencyValue::Null, count: nulls });
3252 }
3253 (Some(entries), distinct)
3254 }
3255
3256 fn visit_numeric(
3263 &self,
3264 column: usize,
3265 signed: bool,
3266 mut visit: impl FnMut(u64, Option<u64>),
3267 ) -> Result<()> {
3268 let ty = &self.table.fields[column].ty;
3269 let mut start = 0_u64;
3270 let mut block = Vec::new();
3271 for stripe in &self.table.stripes {
3272 let spans = read_index(&self.file, stripe, column)?;
3273 let page = stripe.pages[column];
3274 let mut bytes = vec![0; page.length as usize];
3275 read_at(&self.file, page.offset, &mut bytes)?;
3276 for (span, &rows) in spans.iter().zip(&stripe.parts) {
3277 let part = part_bytes(&bytes, *span)?;
3278 if checksum(part) != span.hash {
3279 return Err(invalid("column page checksum differs while building frequencies"));
3280 }
3281 let rows = rows as usize;
3282 let vector = decode(ty, rows, part, None)?;
3283 if signed && vector.signed_block(&mut block) && block.len() == rows {
3287 if vector.none_null() {
3288 for (row, &value) in block.iter().enumerate() {
3289 visit(start.saturating_add(row as u64), Some(value as u64));
3290 }
3291 } else {
3292 for (row, &value) in block.iter().enumerate() {
3293 let bits = (!vector.is_null_at(row)).then_some(value as u64);
3294 visit(start.saturating_add(row as u64), bits);
3295 }
3296 }
3297 start = start.saturating_add(rows as u64);
3298 continue;
3299 }
3300 for row in 0..rows {
3302 let bits = if vector.is_null_at(row) {
3303 None
3304 } else {
3305 let widened = match vector.signed_at(row) {
3309 Some(value) => Some(value as u64),
3310 None => match vector.value_at(row) {
3311 Value::UTinyInt(value) => Some(u64::from(value)),
3312 Value::USmallInt(value) => Some(u64::from(value)),
3313 Value::UInteger(value) => Some(u64::from(value)),
3314 Value::UBigInt(value) => Some(value),
3315 _ => None,
3316 },
3317 };
3318 Some(widened.ok_or_else(|| {
3319 invalid("numeric frequency page did not contain an integer value")
3320 })?)
3321 };
3322 visit(start.saturating_add(row as u64), bits);
3323 }
3324 start = start.saturating_add(rows as u64);
3325 }
3326 }
3327 Ok(())
3328 }
3329
3330 fn numeric_columns(&self) -> Vec<usize> {
3332 self.table
3333 .fields
3334 .iter()
3335 .enumerate()
3336 .filter_map(|(column, field)| {
3337 matches!(
3338 field.ty,
3339 LogicalType::TinyInt
3340 | LogicalType::SmallInt
3341 | LogicalType::Integer
3342 | LogicalType::BigInt
3343 | LogicalType::UTinyInt
3344 | LogicalType::USmallInt
3345 | LogicalType::UInteger
3346 | LogicalType::UBigInt
3347 | LogicalType::Date
3348 | LogicalType::Timestamp
3349 )
3350 .then_some(column)
3351 })
3352 .collect()
3353 }
3354
3355 #[allow(dead_code)]
3357 fn stable_codes_at(&self, column: usize, ordinals: &[u64]) -> Result<Option<Vec<Option<u32>>>> {
3358 if self.dictionaries.get(column).and_then(Option::as_ref).is_none() {
3359 return Ok(None);
3360 }
3361 if ordinals.windows(2).any(|pair| pair[0] >= pair[1]) {
3362 return Err(invalid("frequency ordinals are not sorted and unique"));
3363 }
3364 let mut out = Vec::with_capacity(ordinals.len());
3365 let mut wanted = 0;
3366 let mut stripe_start = 0_u64;
3367 for stripe in &self.table.stripes {
3368 let stripe_end = stripe_start.saturating_add(stripe.rows as u64);
3369 if wanted == ordinals.len() || ordinals[wanted] >= stripe_end {
3370 stripe_start = stripe_end;
3371 continue;
3372 }
3373 let spans = read_index(&self.file, stripe, column)?;
3374 let page = stripe.pages[column];
3375 let mut bytes = vec![0; page.length as usize];
3376 read_at(&self.file, page.offset, &mut bytes)?;
3377 let mut part_start = stripe_start;
3378 for (span, &rows) in spans.iter().zip(&stripe.parts) {
3379 let part_end = part_start.saturating_add(u64::from(rows));
3380 if wanted < ordinals.len() && ordinals[wanted] < part_end {
3381 let part = part_bytes(&bytes, *span)?;
3382 if checksum(part) != span.hash {
3383 return Err(invalid(
3384 "column page checksum differs while building pair frequencies",
3385 ));
3386 }
3387 let upto = ordinals.partition_point(|&ordinal| ordinal < part_end);
3388 let positions = ordinals[wanted..upto]
3389 .iter()
3390 .map(|&ordinal| {
3391 usize::try_from(ordinal.saturating_sub(part_start))
3392 .map_err(|_| invalid("frequency row offset does not fit in memory"))
3393 })
3394 .collect::<Result<Vec<_>>>()?;
3395 if !decode_selected_stable_codes(rows as usize, part, &positions, &mut out)? {
3396 return Ok(None);
3397 }
3398 wanted = upto;
3399 }
3400 part_start = part_end;
3401 }
3402 stripe_start = stripe_end;
3403 }
3404 if wanted != ordinals.len() {
3405 return Err(invalid("frequency ordinal is outside the table"));
3406 }
3407 Ok(Some(out))
3408 }
3409
3410 #[allow(dead_code)]
3412 fn pair_frequencies(
3413 &self,
3414 frequencies: &[Option<Frequencies>],
3415 ) -> Result<Vec<PairFrequencySummary>> {
3416 let anchors = frequencies
3417 .iter()
3418 .enumerate()
3419 .filter_map(|(column, summary)| {
3420 match summary {
3422 Some(Frequencies::Held(summary)) => Some(summary),
3423 _ => None,
3424 }
3425 .filter(|summary| {
3426 !summary.ordinals.is_empty()
3427 && summary.ordinal_entries.len() == summary.ordinals.len()
3428 })
3429 .cloned()
3430 .map(|summary| (column, summary))
3431 })
3432 .collect::<Vec<_>>();
3433 let strings = self
3434 .dictionaries
3435 .iter()
3436 .enumerate()
3437 .filter_map(|(column, dictionary)| dictionary.as_ref().map(|_| column))
3438 .collect::<Vec<_>>();
3439 let mut summaries = Vec::new();
3440 for (first, anchors) in anchors {
3441 for &second in &strings {
3442 if summaries.len() == MAX_PAIR_FREQUENCIES {
3443 return Ok(summaries);
3444 }
3445 let Some(codes) = self.stable_codes_at(second, &anchors.ordinals)? else {
3446 continue;
3447 };
3448 if codes.len() != anchors.ordinal_entries.len() {
3449 return Err(invalid("pair frequency columns have different lengths"));
3450 }
3451 let mut counts = HashMap::<(u16, Option<u32>), u64>::new();
3452 for (&anchor, code) in anchors.ordinal_entries.iter().zip(codes) {
3453 *counts.entry((anchor, code)).or_default() += 1;
3454 }
3455 let mut entries = counts
3456 .into_iter()
3457 .map(|((first_entry, second), count)| PairFrequencyEntry {
3458 first_entry,
3459 second,
3460 count,
3461 })
3462 .collect::<Vec<_>>();
3463 entries.sort_unstable_by(|left, right| {
3464 right
3465 .count
3466 .cmp(&left.count)
3467 .then_with(|| left.first_entry.cmp(&right.first_entry))
3468 .then_with(|| left.second.cmp(&right.second))
3469 });
3470 let pair_omitted = entries.get(FREQUENCY_ENTRIES).map_or(0, |entry| entry.count);
3471 entries.truncate(FREQUENCY_ENTRIES);
3472 summaries.push(PairFrequencySummary {
3473 first: u16::try_from(first)
3474 .map_err(|_| invalid("pair frequency column index overflows"))?,
3475 second: u16::try_from(second)
3476 .map_err(|_| invalid("pair frequency column index overflows"))?,
3477 entries,
3478 omitted_max: anchors.omitted_max.max(pair_omitted),
3479 });
3480 }
3481 }
3482 Ok(summaries)
3483 }
3484
3485 fn close(&mut self) -> Result<Entry> {
3496 self.reclaim()?;
3497 self.flush_pending()?;
3498 let profile = self.profile.clone();
3502 let timing = profile.as_deref().map(|profile| profile.span(Stage::Publish));
3503 let before = self.at;
3504 let mut stripes = std::mem::take(&mut self.order)
3505 .into_iter()
3506 .zip(std::mem::take(&mut self.table.stripes))
3507 .collect::<Vec<_>>();
3508 stripes.sort_by_key(|(order, _)| order.0);
3509 let mut previous: Option<(u64, u64)> = None;
3510 for ((first, last), _) in &stripes {
3511 if previous.is_some_and(|previous| previous >= *first) {
3512 return Err(invalid("chunks did not arrive in source order"));
3513 }
3514 previous = Some(*last);
3515 }
3516 self.table.stripes = stripes.into_iter().map(|(_, stripe)| stripe).collect();
3517 drop(timing);
3518 let timing = profile.as_deref().map(|profile| profile.span(Stage::Dictionary));
3519 let placing = self.at;
3520 finish_dictionaries(&mut self.dictionaries)?;
3521 self.place_blocks()?;
3522 for dictionary in self.dictionaries.iter_mut().flatten() {
3523 dictionary.release_lookup();
3524 dictionary.recharge(profile.as_deref());
3525 }
3526 let (numeric, closed) = self.close_columns()?;
3527 let (frequencies, distincts): (Vec<Option<FrequencySummary>>, Vec<_>) =
3528 numeric.into_iter().unzip();
3529 let frequencies =
3530 frequencies.into_iter().map(|held| held.map(Frequencies::Held)).collect::<Vec<_>>();
3531 let pairs = Vec::new();
3533 self.table.frequencies = frequencies;
3534 self.table.distincts = distincts;
3535 self.table.pair_frequencies = pairs;
3536 if let Some(profile) = &profile {
3537 profile.release(self.dictionaries.iter().flatten().map(|held| held.charged).sum());
3538 }
3539 self.table.demoted = self
3540 .dictionaries
3541 .iter()
3542 .map(|dictionary| dictionary.as_ref().is_some_and(|held| held.demoted))
3543 .collect();
3544 if !self.table.demoted.contains(&true) {
3545 self.table.demoted = Vec::new();
3546 }
3547 self.dictionaries = Vec::new();
3548 self.table.dictionary_payloads = vec![0; self.table.fields.len()];
3549 self.table.frequency_texts = vec![Vec::new(); self.table.fields.len()];
3550 self.table.host_groups = None;
3551 for (index, closed) in closed.into_iter().enumerate() {
3552 let Some(closed) = closed else { continue };
3553 let ClosedDictionary { distinct, frequencies, texts, hosts, encoded, payload } = closed;
3554 self.table.distincts[index] = distinct;
3555 self.table.frequencies[index] = frequencies.map(Frequencies::Held);
3556 self.table.frequency_texts[index] = texts;
3557 if hosts.is_some() {
3558 self.table.host_groups = hosts;
3559 }
3560 let offset = self.at;
3561 self.put(&encoded.index)?;
3562 self.put(&encoded.ranks)?;
3563 self.put(&encoded.grams)?;
3564 self.table.dictionary_payloads[index] = payload;
3565 let length = encoded
3566 .index
3567 .len()
3568 .checked_add(encoded.ranks.len())
3569 .and_then(|len| len.checked_add(encoded.grams.len()))
3570 .ok_or_else(|| invalid("dictionary page length overflow"))?;
3571 self.table.dictionaries[index] = Some(Page {
3572 offset,
3573 length: u32::try_from(length)
3574 .map_err(|_| invalid("dictionary page length overflow"))?,
3575 hash: checksum(&encoded.index),
3576 });
3577 }
3578 drop(timing);
3579 let timing = profile.as_deref().map(|profile| profile.span(Stage::Publish));
3580 let placed = self.at - placing;
3581 self.write_stats()?;
3582 let directory = encode_directory(&self.table)?;
3583 if directory.len() > MAX_DIRECTORY {
3584 return Err(invalid("directory exceeds the configured bound"));
3585 }
3586 let offset = self.at;
3587 self.put(&directory)?;
3588 drop(timing);
3589 if let Some(profile) = &profile {
3590 profile.moved(Stage::Dictionary, 0, placed, 0);
3591 profile.moved(Stage::Publish, 0, self.at - before - placed, 0);
3592 }
3593 Ok(Entry {
3594 name: self.table.name.clone(),
3595 fields: self.table.fields.clone(),
3596 rows: self.table.rows,
3597 nonzero: vec![None; self.table.fields.len()],
3598 aggregates: table_aggregate_sums(&self.table),
3599 distincts: self.table.distincts.clone(),
3600 extremes: table_integer_extremes(&self.table),
3601 frequencies: table_complete_numeric_frequencies(&self.table),
3602 directory: Page {
3603 offset,
3604 length: u32::try_from(directory.len())
3605 .map_err(|_| invalid("directory length overflow"))?,
3606 hash: checksum(&directory),
3607 },
3608 })
3609 }
3610
3611 #[allow(clippy::type_complexity)]
3628 fn close_columns(
3629 &self,
3630 ) -> Result<(Vec<(Option<FrequencySummary>, Option<u64>)>, Vec<Option<ClosedDictionary>>)> {
3631 let numeric = self.numeric_columns().into_iter().map(|column| {
3632 let gather = self.gathers.get(column).and_then(Option::as_ref);
3633 let estimate = gather.and_then(stats::Gather::distinct);
3634 let counted = !estimate.is_some_and(distinct::beyond);
3635 let set =
3636 if counted { distinct::bytes_for(estimate.unwrap_or(f64::INFINITY)) } else { 0 };
3637 let dense = gather.filter(|_| counted).and_then(|gather| self.dense_range(gather, set));
3638 let set = dense.map_or(set, |(_, len)| len * size_of::<u32>());
3639 let cost = self.table.rows.saturating_mul(weight(&self.table.fields[column].ty));
3640 (Closing::Numeric { column, counted, dense }, NUMERIC_CLOSE_BYTES + set, cost)
3641 });
3642 let dictionaries =
3643 self.dictionaries.iter().enumerate().filter_map(|(index, dictionary)| {
3644 let dictionary = dictionary.as_ref()?;
3645 let bytes = dictionary.closing_bytes();
3646 Some((Closing::Dictionary { index, dictionary }, bytes, bytes))
3647 });
3648 let mut jobs = numeric.chain(dictionaries).collect::<Vec<_>>();
3649 jobs.sort_by_key(|&(_, _, cost)| cost);
3650 let columns = self.table.fields.len();
3651 let mut frequencies = vec![(None, None); columns];
3652 let mut closed = (0..columns).map(|_| None).collect::<Vec<_>>();
3653 let profile = self.profile.as_deref();
3654 let run = |job: Closing<'_>, bytes: usize| -> Result<Closed> {
3655 let _holding = profile.map(|profile| profile.holding(bytes as u64));
3656 let closed = match job {
3657 Closing::Numeric { column, counted, dense } => {
3658 let _timing = profile.map(|profile| profile.span(Stage::Publish));
3659 Closed::Numeric(column, self.numeric_frequency(column, counted, dense)?)
3660 }
3661 Closing::Dictionary { index, dictionary } => {
3662 let _timing = profile.map(|profile| profile.span(Stage::Dictionary));
3663 Closed::Dictionary(index, self.close_dictionary(index, dictionary)?)
3664 }
3665 };
3666 rudb_common::heap::release();
3669 Ok(closed)
3670 };
3671 let workers = close_workers().min(jobs.len());
3672 let pieces = if workers <= 1 {
3673 jobs.into_iter().map(|(job, bytes, _)| run(job, bytes)).collect::<Result<Vec<_>>>()?
3674 } else {
3675 let state = Mutex::new((jobs, 0_usize));
3677 let finished = Condvar::new();
3678 std::thread::scope(|scope| {
3679 (0..workers)
3680 .map(|_| {
3681 scope.spawn(|| {
3682 let mut mine = Vec::new();
3683 loop {
3684 let mut held = state.lock().map_err(|_| {
3685 Error::internal("a native close worker panicked")
3686 })?;
3687 let (job, bytes) = loop {
3688 let (jobs, busy) = &mut *held;
3689 if jobs.is_empty() {
3690 return Ok(mine);
3691 }
3692 let fits = jobs.iter().rposition(|&(_, bytes, _)| {
3693 *busy == 0 || busy.saturating_add(bytes) <= CLOSE_BYTES
3694 });
3695 if let Some(at) = fits {
3696 let (job, bytes, _) = jobs.remove(at);
3697 *busy += bytes;
3698 break (job, bytes);
3699 }
3700 held = finished.wait(held).map_err(|_| {
3701 Error::internal("a native close worker panicked")
3702 })?;
3703 };
3704 drop(held);
3705 let _room = Room { state: &state, finished: &finished, bytes };
3708 mine.push(run(job, bytes)?);
3709 }
3710 })
3711 })
3712 .collect::<Vec<_>>()
3713 .into_iter()
3714 .map(|handle| {
3715 handle
3716 .join()
3717 .map_err(|_| Error::internal("a native close worker panicked"))?
3718 })
3719 .collect::<Result<Vec<_>>>()
3720 })?
3721 .into_iter()
3722 .flatten()
3723 .collect()
3724 };
3725 for piece in pieces {
3726 match piece {
3727 Closed::Numeric(column, summary) => frequencies[column] = summary,
3728 Closed::Dictionary(index, one) => closed[index] = Some(one),
3729 }
3730 }
3731 Ok((frequencies, closed))
3732 }
3733
3734 fn close_dictionary(
3741 &self,
3742 _index: usize,
3743 dictionary: &GlobalDictionary,
3744 ) -> Result<ClosedDictionary> {
3745 let (order, flat, bases) = dictionary.ranked_with_values(Some(&*self.file))?;
3746 let (distinct, frequencies, texts) = if dictionary.demoted {
3751 (None, None, Vec::new())
3752 } else {
3753 let distinct = dictionary.counts.iter().filter(|count| **count != 0).count() as u64;
3754 let (frequencies, texts) = code_frequency(dictionary, &flat, &bases)?;
3755 (Some(distinct), Some(frequencies), texts)
3756 };
3757 let hosts = None;
3759 drop(flat);
3760 drop(bases);
3761 let encoded = encode_global_dictionary(dictionary, &order, &dictionary.placed, true)?;
3762 let payload = dictionary
3763 .placed
3764 .iter()
3765 .try_fold(0_u64, |sum, place| sum.checked_add(place.length))
3766 .ok_or_else(|| invalid("global dictionary payload overflow"))?;
3767 Ok(ClosedDictionary { distinct, frequencies, texts, hosts, encoded, payload })
3768 }
3769
3770 fn write_stats(&mut self) -> Result<()> {
3782 let gathers = std::mem::take(&mut self.gathers);
3783 let rows = self.table.rows as u64;
3784 let mut payloads = Vec::new();
3785 for (column, gather) in gathers.into_iter().enumerate() {
3786 let Some(gather) = gather else { continue };
3787 if gather.rows() != rows {
3793 continue;
3794 }
3795 let Some(stats) = gather.finish() else { continue };
3796 let mut summary = Vec::new();
3797 stats.summary.encode(&mut summary)?;
3798 let mut sketches = Vec::new();
3799 stats.sketches.encode(&mut sketches)?;
3800 payloads.push((column, summary, sketches));
3801 }
3802 if payloads.is_empty() {
3803 return Ok(());
3804 }
3805 let summaries = payloads.iter().map(|(_, summary, _)| summary.len()).collect::<Vec<_>>();
3806 let sketches = payloads.iter().map(|(_, _, sketches)| sketches.len()).collect::<Vec<_>>();
3807 let allowance = stats::allowance(stats::column_bytes(&self.table), stats::BUDGET_SHARE);
3808 let keep = stats::kept(&summaries, &sketches, allowance, 0);
3811 for ((column, summary, sketches), &(built, sketched)) in payloads.iter().zip(&keep) {
3812 if !built {
3813 continue;
3814 }
3815 let id = u64::try_from(*column).map_err(|_| invalid("column index overflow"))?;
3816 let sections = [
3817 (*section::SUMMARY, summary, summary.len() as u32),
3820 (*section::SKETCHES, sketches, rudb_stats::sketches::HEADER_BYTES),
3821 ];
3822 let wanted = 1 + usize::from(sketched);
3823 for (kind, bytes, header_bytes) in sections.into_iter().take(wanted) {
3824 let written = write_section(
3825 &*self.file,
3826 &mut self.at,
3827 §ion::Attachment { kind, id, flags: 0, header_bytes, bytes },
3828 self.generation,
3829 )?;
3830 self.table.sections.push(written);
3831 }
3832 }
3833 if self.table.sections.len() > MAX_SECTIONS {
3834 return Err(invalid("the table would name more sections than the bound allows"));
3835 }
3836 Ok(())
3837 }
3838
3839 pub fn finish(mut self) -> Result<Table> {
3849 let entry = self.close()?;
3850 let profile = self.profile.take();
3851 let _timing = profile.as_deref().map(|profile| profile.span(Stage::Publish));
3852 let mut tables = std::mem::take(&mut self.closed);
3853 tables.push(entry);
3854 let catalog = encode_catalog(&tables, &self.views)?;
3855 if catalog.len() > MAX_DIRECTORY {
3856 return Err(invalid("catalog exceeds the configured bound"));
3857 }
3858 let offset = self.at;
3859 self.put(&catalog)?;
3860 if let Some(profile) = &profile {
3861 profile.moved(Stage::Publish, 0, catalog.len() as u64, 0);
3862 }
3863 synced(&*self.file, profile.as_deref())?;
3867 let slot = Slot {
3868 offset,
3869 length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
3870 generation: self.generation,
3871 hash: checksum(&catalog),
3872 };
3873 self.file.write_at(slot_offset(self.generation), &slot.bytes())?;
3878 synced(&*self.file, profile.as_deref())?;
3879 Ok(self.table)
3880 }
3881
3882 pub fn restate(path: impl AsRef<Path>, views: &[ViewEntry]) -> Result<()> {
3899 let file = RealFilesystem::new().open(path.as_ref(), OpenMode::ReadWrite)?;
3900 let size = file.len()?;
3901 let (slot, bytes, _) = committed_slot(&*file, size)?;
3902 let (closed, _) = decode_catalog(&bytes, size)?;
3903 let generation = slot
3904 .generation
3905 .checked_add(1)
3906 .ok_or_else(|| invalid("native file generation overflow"))?;
3907 let catalog = encode_catalog(&closed, views)?;
3908 if catalog.len() > MAX_DIRECTORY {
3909 return Err(invalid("catalog exceeds the configured bound"));
3910 }
3911 file.write_at(size, &catalog)?;
3912 file.sync()?;
3913 let slot = Slot {
3914 offset: size,
3915 length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
3916 generation,
3917 hash: checksum(&catalog),
3918 };
3919 file.write_at(slot_offset(generation), &slot.bytes())?;
3920 file.sync()?;
3921 Ok(())
3922 }
3923
3924 pub fn certify_summaries(path: impl AsRef<Path>) -> Result<()> {
3927 let path = path.as_ref();
3928 let (_, size, slot, bytes, _) = slot_bytes(path)?;
3929 let (mut entries, views) = decode_catalog(&bytes, size)?;
3930 let native = Catalog::open(path)?;
3931 for entry in &mut entries {
3932 let reader = native.table(&entry.name)?;
3933 entry.nonzero.fill(None);
3934 entry.aggregates = reader_aggregate_sums(&reader)?;
3935 entry.distincts = (0..entry.fields.len())
3936 .map(|column| reader.distinct_values(column))
3937 .collect::<Result<Vec<_>>>()?;
3938 entry.extremes = reader_integer_extremes(&reader)?;
3939 entry.frequencies = reader_complete_numeric_frequencies(&reader)?;
3940 }
3941 let generation = slot
3942 .generation
3943 .checked_add(1)
3944 .ok_or_else(|| invalid("native file generation overflow"))?;
3945 let catalog = encode_catalog(&entries, &views)?;
3946 if catalog.len() > MAX_DIRECTORY {
3947 return Err(invalid("catalog exceeds the configured bound"));
3948 }
3949 let file = RealFilesystem::new().open(path, OpenMode::ReadWrite)?;
3950 file.write_at(size, &catalog)?;
3951 file.sync()?;
3952 let slot = Slot {
3953 offset: size,
3954 length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
3955 generation,
3956 hash: checksum(&catalog),
3957 };
3958 file.write_at(slot_offset(generation), &slot.bytes())?;
3959 file.sync()?;
3960 Ok(())
3961 }
3962
3963 pub fn certify_counts(path: impl AsRef<Path>) -> Result<()> {
3965 Self::certify_summaries(path)
3966 }
3967}
3968
3969fn append(file: &dyn rudb_io::File, at: &mut u64, bytes: &[u8]) -> Result<u64> {
3975 let offset = *at;
3976 file.write_at(offset, bytes)?;
3977 *at =
3978 at.checked_add(bytes.len() as u64).ok_or_else(|| invalid("native file length overflow"))?;
3979 Ok(offset)
3980}
3981
3982fn write_section(
3988 file: &dyn rudb_io::File,
3989 at: &mut u64,
3990 one: §ion::Attachment<'_>,
3991 generation: u64,
3992) -> Result<Section> {
3993 if !one.bytes.is_empty() && one.header_bytes as usize > one.bytes.len() {
3997 return Err(invalid("a section's header is longer than its payload"));
3998 }
3999 let mut extents = Vec::new();
4000 let mut first = 0_u64;
4001 let extent_size =
4002 if one.kind == *section::RUN_PROJECTION && one.flags == run_projection::RLE_PAGES {
4003 run_projection::RLE_PAGE_BYTES
4004 } else if one.kind == *section::SORTED_PROJECTION || one.kind == *section::RUN_PROJECTION {
4005 1 << 19
4006 } else {
4007 section::MAX_EXTENT as usize
4008 };
4009 for chunk in one.bytes.chunks(extent_size) {
4010 let offset = append(file, at, chunk)?;
4011 extents.push(section::Extent {
4012 offset,
4013 length: u32::try_from(chunk.len()).map_err(|_| invalid("extent length overflow"))?,
4014 hash: checksum(chunk),
4015 first,
4016 });
4017 first += chunk.len() as u64;
4018 }
4019 let mut table = Vec::with_capacity(extents.len() * section::EXTENT_BYTES);
4020 section::encode_extents(&extents, &mut table)?;
4021 let extent_page = if table.is_empty() { 0 } else { append(file, at, &table)? };
4025 Ok(Section {
4026 kind: one.kind,
4027 id: one.id,
4028 generation,
4029 extents: u32::try_from(extents.len()).map_err(|_| invalid("too many extents"))?,
4030 extent_page,
4031 extent_bytes: u32::try_from(table.len()).map_err(|_| invalid("extent table overflow"))?,
4032 hash: checksum(&table),
4033 flags: one.flags,
4034 header_bytes: one.header_bytes,
4035 })
4036}
4037
4038pub fn attach(
4062 path: impl AsRef<Path>,
4063 table: &str,
4064 attachments: &[section::Attachment<'_>],
4065) -> Result<Table> {
4066 let file = RealFilesystem::new().open(path.as_ref(), OpenMode::ReadWrite)?;
4067 let file = &*file;
4068 let size = file.len()?;
4069 let (slot, bytes, _) = committed_slot(file, size)?;
4070 let (mut entries, views) = decode_catalog(&bytes, size)?;
4071 let at = entries
4072 .iter()
4073 .position(|entry| entry.name == table)
4074 .ok_or_else(|| invalid(&format!("the file holds no table called {table}")))?;
4075 let mut version = [0; 4];
4076 read_at(file, 8, &mut version)?;
4077 let version = u32::from_le_bytes(version);
4078 if version != FORMAT {
4084 return Err(invalid(&format!(
4085 "the file is format {version} and a graph section needs format {FORMAT}, so it has \
4086 to be written again"
4087 )));
4088 }
4089 let mut directory = vec![0; entries[at].directory.length as usize];
4090 read_at(file, entries[at].directory.offset, &mut directory)?;
4091 if checksum(&directory) != entries[at].directory.hash {
4092 return Err(invalid(&format!("the directory of table {table} does not checksum")));
4093 }
4094 let mut held = decode_directory(&directory, size)?;
4095 let mut cursor = size;
4096 for one in attachments {
4097 let written = write_section(file, &mut cursor, one, held.generation)?;
4098 held.sections.retain(|old| !(old.kind == one.kind && old.id == one.id));
4099 held.sections.push(written);
4100 }
4101 if held.sections.len() > MAX_SECTIONS {
4102 return Err(invalid("the table would name more sections than the bound allows"));
4103 }
4104 let encoded = encode_directory(&held)?;
4105 if encoded.len() > MAX_DIRECTORY {
4106 return Err(invalid("directory exceeds the configured bound"));
4107 }
4108 let offset = append(file, &mut cursor, &encoded)?;
4109 entries[at].directory = Page {
4110 offset,
4111 length: u32::try_from(encoded.len()).map_err(|_| invalid("directory length overflow"))?,
4112 hash: checksum(&encoded),
4113 };
4114 let catalog = encode_catalog(&entries, &views)?;
4117 if catalog.len() > MAX_DIRECTORY {
4118 return Err(invalid("catalog exceeds the configured bound"));
4119 }
4120 let offset = append(file, &mut cursor, &catalog)?;
4121 file.sync()?;
4122 let generation =
4123 slot.generation.checked_add(1).ok_or_else(|| invalid("native file generation overflow"))?;
4124 let committed = Slot {
4125 offset,
4126 length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
4127 generation,
4128 hash: checksum(&catalog),
4129 };
4130 file.write_at(slot_offset(generation), &committed.bytes())?;
4131 file.sync()?;
4132 Ok(held)
4133}
4134
4135type Synopsis = Arc<Vec<(Value, u64)>>;
4138
4139#[derive(Debug, Clone)]
4141pub struct Reader {
4142 file: Arc<File>,
4143 table: Arc<Table>,
4144 dictionaries: Arc<Vec<OnceLock<Arc<Vector>>>>,
4145 loading: Arc<Vec<Mutex<()>>>,
4154 frequency_values: Arc<Vec<OnceLock<Synopsis>>>,
4157 frequency_summaries: Arc<Vec<OnceLock<Arc<FrequencySummary>>>>,
4161 opened: Arc<AtomicUsize>,
4165 sieves: Arc<Vec<Vec<SieveSlot>>>,
4169 part_ranges: Arc<Vec<Vec<RangeSlot>>>,
4172 places: Arc<Vec<Place>>,
4174 cache: Arc<Shelf>,
4175 pool: PagePool,
4177 pages: Arc<AtomicUsize>,
4180 indexes: Arc<AtomicUsize>,
4183 size: u64,
4185 directory: u64,
4187 opening: Opening,
4189}
4190
4191#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
4203pub struct Opening {
4204 pub reads: u32,
4207 pub bytes: u64,
4209}
4210
4211#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
4213pub struct Reads {
4214 pub opening: Opening,
4216 pub pages: usize,
4218 pub indexes: usize,
4220 pub dictionaries: usize,
4223}
4224
4225#[derive(Debug, Clone, Copy)]
4227struct Place {
4228 stripe: u32,
4229 part: u32,
4230 rows: u32,
4231}
4232
4233#[derive(Debug, Clone, Copy)]
4235struct PartSpan {
4236 start: usize,
4237 length: usize,
4238 hash: u64,
4239}
4240
4241#[derive(Debug, Clone)]
4247struct CachedColumn {
4248 stripe: usize,
4249 index: Arc<Vec<PartSpan>>,
4250 page: Option<Arc<HeldPage>>,
4251}
4252
4253#[derive(Debug)]
4260struct HeldPage {
4261 bytes: Vec<u8>,
4262 checked: Vec<AtomicBool>,
4263}
4264
4265impl HeldPage {
4266 fn part(&self, part: usize, span: PartSpan) -> Result<&[u8]> {
4268 let bytes = part_bytes(&self.bytes, span)?;
4269 let checked = self.checked.get(part).ok_or_else(|| invalid("part index out of range"))?;
4270 if !checked.load(Atomic::Relaxed) {
4271 verify_part(bytes, span)?;
4272 checked.store(true, Atomic::Relaxed);
4273 }
4274 Ok(bytes)
4275 }
4276}
4277
4278fn verify_part(bytes: &[u8], span: PartSpan) -> Result<()> {
4280 let got = checksum(bytes);
4281 if got != span.hash {
4282 return Err(invalid(&format!(
4283 "column page checksum differs, part at {}+{} bytes, wanted {:016x} and got {got:016x}",
4284 span.start, span.length, span.hash,
4285 )));
4286 }
4287 Ok(())
4288}
4289
4290#[derive(Debug, Default)]
4319struct Cached {
4320 pages: Vec<Option<Resident>>,
4321 loading: Vec<usize>,
4322 index: Vec<Option<Arc<Vec<PartSpan>>>>,
4323 seen: Vec<bool>,
4324 passing: VecDeque<usize>,
4325}
4326
4327#[derive(Debug, Clone)]
4329struct Resident {
4330 page: Arc<HeldPage>,
4331 used: Arc<AtomicBool>,
4332}
4333
4334#[derive(Debug)]
4336struct Shelf {
4337 columns: Vec<Mutex<Cached>>,
4338 held: Vec<AtomicUsize>,
4341 kept: AtomicUsize,
4344}
4345
4346#[derive(Debug, Clone, Default)]
4365pub struct PagePool {
4366 ring: Arc<Mutex<Ring>>,
4367 budget: Arc<AtomicUsize>,
4368}
4369
4370#[derive(Debug, Default)]
4371struct Ring {
4372 held: VecDeque<Held>,
4373 bytes: usize,
4374}
4375
4376#[derive(Debug)]
4381struct Held {
4382 shelf: Weak<Shelf>,
4383 column: usize,
4384 stripe: usize,
4385 bytes: usize,
4386 used: Arc<AtomicBool>,
4387}
4388
4389impl PagePool {
4390 #[must_use]
4392 pub fn new(budget: usize) -> Self {
4393 let pool = Self::default();
4394 pool.budget.store(budget, Atomic::Relaxed);
4395 pool
4396 }
4397
4398 #[must_use]
4404 pub fn bytes(&self) -> usize {
4405 self.ring.lock().map_or(0, |ring| ring.bytes)
4406 }
4407
4408 fn admit(&self, held: Held) {
4414 let budget = self.budget.load(Atomic::Relaxed);
4415 let mut gone = Vec::new();
4416 {
4417 let Ok(mut ring) = self.ring.lock() else { return };
4418 ring.bytes += held.bytes;
4419 ring.held.push_back(held);
4420 let mut looked = 0;
4423 let limit = ring.held.len();
4424 while ring.bytes > budget && looked < limit {
4425 looked += 1;
4426 let Some(entry) = ring.held.pop_front() else { break };
4427 let Some(shelf) = entry.shelf.upgrade() else {
4428 ring.bytes -= entry.bytes;
4429 continue;
4430 };
4431 if entry.used.swap(false, Atomic::Relaxed) {
4432 ring.held.push_back(entry);
4433 continue;
4434 }
4435 let count = &shelf.held[entry.column];
4436 if count.load(Atomic::Relaxed) <= shelf.kept.load(Atomic::Relaxed).max(1) {
4437 ring.held.push_back(entry);
4438 continue;
4439 }
4440 count.fetch_sub(1, Atomic::Relaxed);
4441 ring.bytes -= entry.bytes;
4442 gone.push((shelf, entry));
4443 }
4444 while ring.held.front().is_some_and(|entry| entry.shelf.strong_count() == 0) {
4447 if let Some(entry) = ring.held.pop_front() {
4448 ring.bytes -= entry.bytes;
4449 }
4450 }
4451 }
4452 for (shelf, entry) in gone {
4453 let Ok(mut cached) = shelf.columns[entry.column].lock() else { continue };
4454 if let Some(slot) = cached.pages.get_mut(entry.stripe) {
4455 if slot.as_ref().is_some_and(|slot| Arc::ptr_eq(&slot.used, &entry.used)) {
4456 *slot = None;
4457 }
4458 }
4459 }
4460 }
4461}
4462
4463const CACHED_STRIPES_PER_COLUMN: usize = 4;
4475
4476type SieveSlot = OnceLock<Arc<Vec<Option<Sieve>>>>;
4478
4479type RangeSlot = OnceLock<Arc<Vec<Range>>>;
4480
4481#[derive(Debug)]
4482struct NativeText {
4483 file: Arc<File>,
4484 values: usize,
4486 offsets: Vec<u8>,
4498 offset_bits: usize,
4501 value_ends: OnceLock<Option<Vec<u32>>>,
4514 value_lens: OnceLock<Option<Lengths>>,
4524 ends_asked: AtomicUsize,
4530 ranks: usize,
4532 rank_at: u64,
4536 rank_ends: Vec<u64>,
4540 rank_hashes: Vec<u64>,
4541 rank_blocks: Vec<OnceLock<Result<Vec<u8>>>>,
4542 code_bits: usize,
4545 code_ranks: OnceLock<Option<Vec<u32>>>,
4552 starts: Vec<u64>,
4559 lengths: Vec<u64>,
4560 hashes: Vec<u64>,
4561 grams: Option<NativeGrams>,
4563 blocks: Vec<OnceLock<Result<Vec<u8>>>>,
4565 char_lens: Vec<OnceLock<Box<[u32]>>>,
4574 keep_budget: usize,
4577 payload_kept: AtomicUsize,
4585 swept: Vec<AtomicBool>,
4593 visit_dropped: AtomicUsize,
4608 searched: Mutex<HashMap<Vec<u8>, (usize, bool)>>,
4625}
4626
4627#[derive(Debug)]
4628struct NativeGrams {
4629 start: u64,
4630 length: usize,
4631 width: usize,
4633 hash: u64,
4634 verdicts: Mutex<Vec<Verdict>>,
4641}
4642
4643type Verdict = (Vec<u8>, Arc<[bool]>);
4645
4646const GRAM_VERDICTS: usize = 8;
4648
4649impl NativeGrams {
4650 fn verdicts(&self, file: &File, literal: &[u8]) -> Result<Arc<[bool]>> {
4655 let mut held = self.verdicts.lock().map_err(|_| invalid("a poisoned signature verdict"))?;
4656 if let Some((_, verdict)) = held.iter().find(|(asked, _)| asked == literal) {
4657 return Ok(Arc::clone(verdict));
4658 }
4659 let wanted = literal.windows(4).map(|gram| gram_bits(gram, self.width)).collect::<Vec<_>>();
4660 let mut verdict = Vec::with_capacity(self.length / self.width);
4661 let window = GRAM_WINDOW / self.width * self.width;
4662 let hash = walk_checksummed(file, self.start, self.length, window, |bytes| {
4663 verdict.extend(bytes.chunks(self.width).map(|bits| {
4664 wanted
4665 .iter()
4666 .flatten()
4667 .all(|&bit| bits.get(bit / 8).is_some_and(|byte| byte & (1 << (bit % 8)) != 0))
4668 }));
4669 Ok(())
4670 })?;
4671 if hash != self.hash {
4672 return Err(invalid("global dictionary substring signatures checksum differs"));
4673 }
4674 let verdict: Arc<[bool]> = verdict.into();
4675 if held.len() >= GRAM_VERDICTS {
4676 held.remove(0);
4677 }
4678 held.push((literal.to_vec(), Arc::clone(&verdict)));
4679 Ok(verdict)
4680 }
4681
4682 fn footprint(&self) -> usize {
4683 self.verdicts.lock().map_or(0, |held| {
4684 held.iter().map(|(asked, verdict)| asked.capacity() + verdict.len()).sum()
4685 })
4686 }
4687}
4688
4689const TEXT_SEARCH_MEMO: usize = 64;
4694
4695const TEXT_PAYLOAD_VALUES: usize = 1024;
4711
4712const TEXT_GRAM_BYTES: usize = 8192;
4723
4724const NARROW_GRAM_BYTES: usize = 2048;
4726
4727const GRAM_WINDOW: usize = 256 << 10;
4729
4730fn gram_bits(bytes: &[u8], width: usize) -> [usize; 2] {
4733 let original = u32::from_le_bytes(bytes.try_into().expect("a four-byte gram"));
4734 let mut first = original ^ (original >> 16);
4735 first = first.wrapping_mul(0x7feb_352d);
4736 first ^= first >> 15;
4737 let mut second = original ^ (original >> 17);
4738 second = second.wrapping_mul(0x846c_a68b);
4739 second ^= second >> 16;
4740 let mask = width * 8 - 1;
4741 [(first as usize) & mask, (second as usize) & mask]
4742}
4743
4744const TEXT_KEEP_BUDGET: usize = 256 * 1024 * 1024;
4765
4766#[derive(Debug)]
4773enum Lengths {
4774 Narrow(Vec<u16>),
4776 Wide(Vec<u32>),
4778}
4779
4780impl Lengths {
4781 fn extend_at(&self, indices: &[u32], into: &mut Vec<i64>) {
4784 match self {
4785 Lengths::Narrow(lens) => into.extend(
4786 indices
4787 .iter()
4788 .map(|&index| lens.get(index as usize).map_or(0, |&len| i64::from(len))),
4789 ),
4790 Lengths::Wide(lens) => into.extend(
4791 indices
4792 .iter()
4793 .map(|&index| lens.get(index as usize).map_or(0, |&len| i64::from(len))),
4794 ),
4795 }
4796 }
4797
4798 fn footprint(&self) -> usize {
4800 match self {
4801 Lengths::Narrow(lens) => lens.capacity() * size_of::<u16>(),
4802 Lengths::Wide(lens) => lens.capacity() * size_of::<u32>(),
4803 }
4804 }
4805}
4806
4807fn lengths_of(ends: &[u32]) -> Option<Lengths> {
4816 match lengths_as::<u16>(ends)? {
4817 Some(narrow) => Some(Lengths::Narrow(narrow)),
4818 None => lengths_as::<u32>(ends)?.map(Lengths::Wide),
4819 }
4820}
4821
4822fn lengths_as<T: TryFrom<u32>>(ends: &[u32]) -> Option<Option<Vec<T>>> {
4825 let mut lens = Vec::with_capacity(ends.len());
4826 for block in ends.chunks(TEXT_PAYLOAD_VALUES) {
4827 let mut start = 0;
4828 for &end in block {
4829 let Ok(len) = T::try_from(end.checked_sub(start)?) else {
4830 return Some(None);
4831 };
4832 lens.push(len);
4833 start = end;
4834 }
4835 }
4836 Some(Some(lens))
4837}
4838
4839const TEXT_OFFSET_RUN: usize = 512;
4846
4847const DICTIONARY_HEADER: usize = 16;
4850
4851const DICTIONARY_SCATTERED: u32 = 1 << 31;
4865const DICTIONARY_GRAMS: u32 = 1 << 30;
4867const DICTIONARY_WIDE_GRAMS: u32 = 1 << 29;
4870const DICTIONARY_FLAGS: u32 = DICTIONARY_SCATTERED | DICTIONARY_GRAMS | DICTIONARY_WIDE_GRAMS;
4872
4873const TEXT_RANK_BLOCK: usize = 512;
4884
4885const RANK_BLOCK_HEADER: usize = size_of::<u64>() + 1;
4899
4900impl NativeText {
4901 fn payload_block(&self, block: usize) -> Result<Option<&[u8]>> {
4908 let Some(slot) = self.blocks.get(block) else { return Ok(None) };
4909 let bytes = slot.get_or_init(|| self.decode_block(block)).as_ref().map_err(Clone::clone)?;
4910 Ok(Some(bytes.as_slice()))
4911 }
4912
4913 fn block_chars(&self, block: usize) -> Result<&[u32]> {
4920 let slot = self
4921 .char_lens
4922 .get(block)
4923 .ok_or_else(|| invalid("a block past the global dictionary"))?;
4924 if let Some(lens) = slot.get() {
4925 return Ok(lens);
4926 }
4927 let decoded;
4928 let bytes: &[u8] = match self.blocks.get(block).and_then(OnceLock::get) {
4929 Some(Ok(kept)) => kept,
4930 _ => {
4931 decoded = self.decode_block(block)?;
4932 &decoded
4933 }
4934 };
4935 let first = block * TEXT_PAYLOAD_VALUES;
4936 let last = (first + TEXT_PAYLOAD_VALUES).min(self.values);
4937 let ends = self.ends_within(first, last)?;
4938 if ends.len() != last - first {
4939 return Err(invalid("global dictionary offsets are short"));
4940 }
4941 let mut lens = Vec::with_capacity(ends.len());
4942 let mut start = u64::from(self.start_within(first)?);
4943 for &end in &ends {
4944 let value = usize::try_from(start)
4945 .ok()
4946 .zip(usize::try_from(end).ok())
4947 .and_then(|(from, to)| bytes.get(from..to))
4948 .ok_or_else(|| invalid("global dictionary value is past its block"))?;
4949 let characters = value.iter().filter(|byte| (**byte as i8) >= -0x40).count();
4952 lens.push(u32::try_from(characters).unwrap_or(u32::MAX));
4953 start = end;
4954 }
4955 Ok(slot.get_or_init(|| lens.into_boxed_slice()))
4956 }
4957
4958 fn decode_block(&self, block: usize) -> Result<Vec<u8>> {
4963 let len = self.lengths[block];
4964 let mut stored = vec![
4965 0;
4966 usize::try_from(len).map_err(|_| invalid(
4967 "global dictionary block does not fit in memory"
4968 ))?
4969 ];
4970 read_at(&self.file, self.starts[block], &mut stored)?;
4971 if checksum(&stored) != self.hashes[block] {
4972 return Err(invalid("global dictionary payload checksum differs"));
4973 }
4974 let first = block * TEXT_PAYLOAD_VALUES;
4975 let last = (first + TEXT_PAYLOAD_VALUES).min(self.values);
4976 let want = self.end_within(last - 1)? as usize;
4977 let values = string::decode_flat(&stored)?;
4978 if values.len() != last - first {
4979 return Err(invalid("global dictionary block holds the wrong value count"));
4980 }
4981 let bytes = values.into_bytes();
4982 if bytes.len() != want {
4983 return Err(invalid("global dictionary block decodes to the wrong length"));
4984 }
4985 Ok(bytes)
4986 }
4987
4988 fn loaned_block<'a>(
4997 &'a self,
4998 block: usize,
4999 decoded: &'a mut Vec<u8>,
5000 scattered: bool,
5001 ) -> Result<&'a [u8]> {
5002 let kept = self.blocks.get(block).and_then(OnceLock::get);
5003 if let Some(Ok(kept)) = kept {
5004 return Ok(kept);
5005 }
5006 let again = kept.is_none()
5007 && self.swept.get(block).is_some_and(|swept| swept.swap(true, Atomic::Relaxed));
5008 let keep = again
5009 && (self.payload_kept.load(Atomic::Relaxed) < self.keep_budget
5010 || (scattered && self.visit_dropped.load(Atomic::Relaxed) >= self.blocks.len()));
5011 if keep {
5012 let kept = self
5013 .payload_block(block)?
5014 .ok_or_else(|| invalid("global dictionary block is past the payload"))?;
5015 self.payload_kept.fetch_add(kept.len(), Atomic::Relaxed);
5016 return Ok(kept);
5017 }
5018 *decoded = self.decode_block(block)?;
5019 if scattered && again {
5020 self.visit_dropped.fetch_add(1, Atomic::Relaxed);
5021 }
5022 Ok(decoded)
5023 }
5024
5025 fn ends_worth_unpacking(&self) -> usize {
5042 self.values.max(TEXT_PAYLOAD_VALUES)
5043 }
5044
5045 fn value_ends(&self) -> Option<&[u32]> {
5047 if let Some(built) = self.value_ends.get() {
5048 return built.as_deref();
5049 }
5050 if self.ends_asked.fetch_add(1, Atomic::Relaxed) < self.ends_worth_unpacking() {
5051 return None;
5052 }
5053 self.value_ends.get_or_init(|| self.unpack_ends()).as_deref()
5054 }
5055
5056 fn unpack_ends(&self) -> Option<Vec<u32>> {
5062 let mut ends = vec![0u32; self.values];
5063 for (run, into) in ends.chunks_mut(TEXT_OFFSET_RUN).enumerate() {
5064 let bytes = self.packed().get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)?;
5065 bitpack::unpack_tail_into(bytes, self.offset_bits, into, |bits| {
5066 u32::try_from(bits).unwrap_or(u32::MAX)
5067 })
5068 .ok()?;
5069 }
5070 if ends.contains(&u32::MAX) { None } else { Some(ends) }
5073 }
5074
5075 fn packed(&self) -> &[u8] {
5077 self.offsets.get(DICTIONARY_HEADER..).unwrap_or_default()
5078 }
5079
5080 fn end_within(&self, index: usize) -> Result<u32> {
5082 if let Some(ends) = self.value_ends() {
5083 return ends
5084 .get(index)
5085 .copied()
5086 .ok_or_else(|| invalid("global dictionary offsets are short"));
5087 }
5088 let run = index / TEXT_OFFSET_RUN;
5089 let bytes = self
5090 .packed()
5091 .get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)
5092 .ok_or_else(|| invalid("global dictionary offsets are short"))?;
5093 let end = bitpack::tail_at(bytes, self.offset_bits, index % TEXT_OFFSET_RUN)
5094 .map_err(|_| invalid("global dictionary offsets are short"))?;
5095 u32::try_from(end).map_err(|_| invalid("global dictionary offset is past the payload"))
5096 }
5097
5098 fn ends_within(&self, first: usize, last: usize) -> Result<Vec<u64>> {
5116 let mut ends = vec![0u64; last.saturating_sub(first)];
5117 let mut scratch = Vec::new();
5118 let mut at = first;
5119 while at < last {
5120 let run = at / TEXT_OFFSET_RUN;
5121 let stop = ((run + 1) * TEXT_OFFSET_RUN).min(last);
5122 let held = self.values.saturating_sub(run * TEXT_OFFSET_RUN).min(TEXT_OFFSET_RUN);
5123 let bytes = self
5124 .packed()
5125 .get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)
5126 .ok_or_else(|| invalid("global dictionary offsets are short"))?;
5127 let from = at % TEXT_OFFSET_RUN;
5128 let upto = stop - run * TEXT_OFFSET_RUN;
5129 if upto > held || bytes.len() < bitpack::tail_len(held, self.offset_bits) {
5130 return Err(invalid("global dictionary offsets are short"));
5131 }
5132 let into = &mut ends[at - first..stop - first];
5133 if from == 0 {
5134 bitpack::unpack_tail_into(bytes, self.offset_bits, into, |bits| bits)
5135 .map_err(|_| invalid("global dictionary offsets are short"))?;
5136 } else {
5137 scratch.resize(held, 0);
5138 bitpack::unpack_tail_into(bytes, self.offset_bits, &mut scratch, |bits| bits)
5139 .map_err(|_| invalid("global dictionary offsets are short"))?;
5140 into.copy_from_slice(&scratch[from..upto]);
5141 }
5142 at = stop;
5143 }
5144 Ok(ends)
5145 }
5146
5147 fn start_within(&self, index: usize) -> Result<u32> {
5150 if index % TEXT_PAYLOAD_VALUES == 0 { Ok(0) } else { self.end_within(index - 1) }
5151 }
5152
5153 fn span_within(&self, index: usize) -> Result<(u32, u32)> {
5161 if let Some(ends) = self.value_ends() {
5162 let end =
5163 *ends.get(index).ok_or_else(|| invalid("global dictionary offsets are short"))?;
5164 let start = if index % TEXT_PAYLOAD_VALUES == 0 { 0 } else { ends[index - 1] };
5167 if start > end {
5168 return Err(invalid("global dictionary value ends before it starts"));
5169 }
5170 return Ok((start, end));
5171 }
5172 let within = index % TEXT_OFFSET_RUN;
5173 let (start, end) = if within == 0 {
5174 (self.start_within(index)?, self.end_within(index)?)
5175 } else {
5176 let run = index / TEXT_OFFSET_RUN;
5177 let bytes = self
5178 .packed()
5179 .get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)
5180 .ok_or_else(|| invalid("global dictionary offsets are short"))?;
5181 let (start, end) = bitpack::tail_pair(bytes, self.offset_bits, within)
5182 .map_err(|_| invalid("global dictionary offsets are short"))?;
5183 let ends = u32::try_from(end)
5184 .map_err(|_| invalid("global dictionary offset is past the payload"))?;
5185 let starts = u32::try_from(start)
5186 .map_err(|_| invalid("global dictionary offset is past the payload"))?;
5187 (starts, ends)
5188 };
5189 if start > end {
5190 return Err(invalid("global dictionary value ends before it starts"));
5191 }
5192 Ok((start, end))
5193 }
5194
5195 fn rank_parts(&self, rank: usize) -> Result<(&[u8], usize)> {
5202 let slot = self
5203 .rank_blocks
5204 .get(rank / TEXT_RANK_BLOCK)
5205 .ok_or_else(|| invalid("global dictionary rank is past the order"))?;
5206 let block = slot
5207 .get_or_init(|| {
5208 let mut bytes = Vec::new();
5209 self.read_rank_block(rank / TEXT_RANK_BLOCK, &mut bytes)?;
5210 Ok(bytes)
5211 })
5212 .as_ref()
5213 .map_err(Clone::clone)?;
5214 Ok((block.as_slice(), rank % TEXT_RANK_BLOCK))
5215 }
5216
5217 fn read_rank_block(&self, which: usize, bytes: &mut Vec<u8>) -> Result<()> {
5220 let start = if which == 0 { 0 } else { self.rank_ends[which - 1] };
5221 let end = self.rank_ends[which];
5222 bytes.clear();
5223 bytes.resize((end - start) as usize, 0);
5224 read_at(&self.file, self.rank_at + start, bytes)?;
5225 let expected = self
5226 .rank_hashes
5227 .get(which)
5228 .ok_or_else(|| invalid("global dictionary rank block has no checksum"))?;
5229 if checksum(bytes) != *expected {
5230 return Err(invalid("global dictionary rank checksum differs"));
5231 }
5232 Ok(())
5233 }
5234
5235 fn head_at(&self, rank: usize) -> Result<u64> {
5237 let (block, within) = self.rank_parts(rank)?;
5238 let (base, width, packed) = rank_heads(block)?;
5239 let above = bitpack::tail_at(packed, width, within)
5240 .map_err(|_| invalid("global dictionary rank block is short of heads"))?;
5241 Ok(base.wrapping_add(above))
5242 }
5243
5244 fn rank_codes<'block>(&self, block: &'block [u8], count: usize) -> Result<&'block [u8]> {
5246 let (_, width, packed) = rank_heads(block)?;
5247 packed
5248 .get(bitpack::tail_len(count, width)..)
5249 .ok_or_else(|| invalid("global dictionary rank block is short of codes"))
5250 }
5251
5252 fn rank_block_len(&self, rank: usize) -> usize {
5254 let first = rank / TEXT_RANK_BLOCK * TEXT_RANK_BLOCK;
5255 TEXT_RANK_BLOCK.min(self.ranks - first)
5256 }
5257}
5258
5259fn rank_heads(block: &[u8]) -> Result<(u64, usize, &[u8])> {
5261 let header = block
5262 .get(..RANK_BLOCK_HEADER)
5263 .ok_or_else(|| invalid("global dictionary rank block is short"))?;
5264 let base = u64::from_le_bytes(header[..8].try_into().expect("eight bytes"));
5265 let width = header[8] as usize;
5266 if width > 64 {
5267 return Err(invalid("global dictionary rank block packs heads past a word"));
5268 }
5269 Ok((base, width, &block[RANK_BLOCK_HEADER..]))
5270}
5271
5272fn offset_width(ends: &[u32]) -> usize {
5279 let span = ends.iter().copied().max().unwrap_or(0);
5283 (u32::BITS - span.leading_zeros()) as usize
5284}
5285
5286fn offset_bytes(values: usize, bits: usize) -> usize {
5289 let full = values / TEXT_OFFSET_RUN;
5290 let rest = values % TEXT_OFFSET_RUN;
5291 full * TEXT_OFFSET_RUN / 8 * bits + bitpack::tail_len(rest, bits)
5292}
5293
5294fn encode_offsets(ends: &[u32], bits: usize, out: &mut Vec<u8>) -> Result<()> {
5298 let mut run = Vec::with_capacity(TEXT_OFFSET_RUN);
5299 for chunk in ends.chunks(TEXT_OFFSET_RUN) {
5300 run.clear();
5301 run.extend(chunk.iter().map(|&end| u64::from(end)));
5302 bitpack::pack_tail(&run, bits, out)
5303 .map_err(|_| invalid("global dictionary offsets do not pack"))?;
5304 }
5305 Ok(())
5306}
5307
5308fn code_width(values: usize) -> usize {
5310 match u64::try_from(values).unwrap_or(u64::MAX) {
5311 0 | 1 => 0,
5312 last => (u64::BITS - (last - 1).leading_zeros()) as usize,
5313 }
5314}
5315
5316impl TextSource for NativeText {
5317 fn len(&self) -> usize {
5318 self.values
5319 }
5320
5321 fn might_contain(&self, first: usize, literal: &[u8]) -> Result<bool> {
5322 let Some(grams) = &self.grams else { return Ok(true) };
5323 if literal.len() < 4 || first >= self.values {
5324 return Ok(true);
5325 }
5326 let verdict = grams.verdicts(&self.file, literal)?;
5327 Ok(verdict.get(first / TEXT_PAYLOAD_VALUES).copied().unwrap_or(true))
5328 }
5329
5330 fn bytes_at(&self, index: usize) -> Result<Option<&[u8]>> {
5331 if index >= self.values {
5332 return Ok(None);
5333 }
5334 let (start, end) = self.span_within(index)?;
5335 if start == end {
5336 return Ok(Some(&[]));
5337 }
5338 let block = index / TEXT_PAYLOAD_VALUES;
5341 let Some(bytes) = self.payload_block(block)? else { return Ok(None) };
5342 Ok(bytes.get(start as usize..end as usize))
5343 }
5344
5345 fn bytes_len_at(&self, index: usize) -> Result<Option<usize>> {
5346 if index >= self.values {
5347 return Ok(None);
5348 }
5349 let (start, end) = self.span_within(index)?;
5350 Ok(Some((end - start) as usize))
5351 }
5352
5353 fn bytes_lens_at(&self, indices: &[u32], into: &mut Vec<i64>) -> Result<()> {
5360 self.ends_asked.fetch_add(indices.len(), Atomic::Relaxed);
5361 into.reserve(indices.len());
5362 let Some(ends) = self.value_ends() else {
5363 for &index in indices {
5364 into.push(
5365 self.bytes_len_at(index as usize)?
5366 .map_or(0, |len| i64::try_from(len).unwrap_or(i64::MAX)),
5367 );
5368 }
5369 return Ok(());
5370 };
5371 if let Some(lens) = self.value_lens.get_or_init(|| lengths_of(ends)) {
5372 lens.extend_at(indices, into);
5373 return Ok(());
5374 }
5375 for &index in indices {
5376 let index = index as usize;
5377 let Some(&end) = ends.get(index) else {
5379 into.push(0);
5380 continue;
5381 };
5382 let start = if index % TEXT_PAYLOAD_VALUES == 0 { 0 } else { ends[index - 1] };
5383 if start > end {
5384 return Err(invalid("global dictionary value ends before it starts"));
5385 }
5386 into.push(i64::from(end - start));
5387 }
5388 Ok(())
5389 }
5390
5391 fn chars_lens_at(&self, indices: &[u32], into: &mut Vec<i64>) -> Result<()> {
5394 into.reserve(indices.len());
5395 for &index in indices {
5396 let index = index as usize;
5397 if index >= self.values {
5399 into.push(0);
5400 continue;
5401 }
5402 let lens = self.block_chars(index / TEXT_PAYLOAD_VALUES)?;
5403 let len = lens
5404 .get(index % TEXT_PAYLOAD_VALUES)
5405 .ok_or_else(|| invalid("global dictionary block holds the wrong value count"))?;
5406 into.push(i64::from(*len));
5407 }
5408 Ok(())
5409 }
5410
5411 fn sweep(
5424 &self,
5425 first: usize,
5426 limit: usize,
5427 body: &mut dyn FnMut(usize, &[u8]) -> Result<()>,
5428 ) -> Result<usize> {
5429 let limit = limit.min(self.values);
5430 if first >= limit {
5431 return Ok(first);
5432 }
5433 let block = first / TEXT_PAYLOAD_VALUES;
5434 let last = ((block + 1) * TEXT_PAYLOAD_VALUES).min(limit);
5435 let mut decoded = Vec::new();
5436 let bytes = self.loaned_block(block, &mut decoded, false)?;
5437 let ends = self.ends_within(first, last)?;
5438 if ends.len() != last - first {
5439 return Err(invalid("global dictionary offsets are short"));
5440 }
5441 let mut start = u64::from(self.start_within(first)?);
5442 for (index, &end) in (first..last).zip(&ends) {
5445 let value = usize::try_from(start)
5446 .ok()
5447 .zip(usize::try_from(end).ok())
5448 .and_then(|(from, to)| bytes.get(from..to))
5449 .ok_or_else(|| invalid("global dictionary value is past its block"))?;
5450 body(index, value)?;
5451 start = end;
5452 }
5453 Ok(last)
5454 }
5455
5456 fn visit_at(
5465 &self,
5466 indices: &[u32],
5467 body: &mut dyn FnMut(usize, &[u8]) -> Result<()>,
5468 ) -> Result<()> {
5469 let mut order = (0..indices.len()).collect::<Vec<_>>();
5470 order.sort_unstable_by_key(|&at| indices[at]);
5471 let block_of = |at: usize| {
5472 let index = indices[at] as usize;
5473 (index < self.values).then_some(index / TEXT_PAYLOAD_VALUES)
5474 };
5475 let mut decoded = Vec::new();
5476 let mut run = 0;
5477 while run < order.len() {
5478 let Some(block) = block_of(order[run]) else {
5479 for &at in &order[run..] {
5481 body(at, &[])?;
5482 }
5483 break;
5484 };
5485 let upto = run + order[run..].partition_point(|&at| block_of(at) == Some(block));
5486 let bytes = self.loaned_block(block, &mut decoded, true)?;
5487 for &at in &order[run..upto] {
5488 let (start, end) = self.span_within(indices[at] as usize)?;
5489 let value = bytes
5490 .get(start as usize..end as usize)
5491 .ok_or_else(|| invalid("global dictionary value is past its block"))?;
5492 body(at, value)?;
5493 }
5494 run = upto;
5495 }
5496 Ok(())
5497 }
5498
5499 fn visit(
5505 &self,
5506 indices: &[usize],
5507 body: &mut dyn FnMut(usize, &[u8]) -> Result<()>,
5508 ) -> Result<()> {
5509 let mut at = 0;
5510 while at < indices.len() {
5511 let block = indices[at] / TEXT_PAYLOAD_VALUES;
5512 let upto =
5513 at + indices[at..].partition_point(|&index| index / TEXT_PAYLOAD_VALUES == block);
5514 let wanted = &indices[at..upto];
5515 if wanted.iter().any(|&index| index >= self.values) {
5516 return Err(invalid("a visited value is past the global dictionary"));
5517 }
5518 let decoded;
5519 let bytes: &[u8] = match self.blocks.get(block).and_then(OnceLock::get) {
5520 Some(Ok(kept)) => kept,
5521 _ => {
5522 decoded = self.decode_block(block)?;
5523 &decoded
5524 }
5525 };
5526 for (offset, &index) in wanted.iter().enumerate() {
5527 let (start, end) = self.span_within(index)?;
5528 let value = bytes
5529 .get(start as usize..end as usize)
5530 .ok_or_else(|| invalid("global dictionary value is past its block"))?;
5531 body(at + offset, value)?;
5532 }
5533 at = upto;
5534 }
5535 Ok(())
5536 }
5537
5538 fn ranks(&self) -> Option<usize> {
5539 (self.ranks > 0).then_some(self.ranks)
5540 }
5541
5542 fn below(&self, ranks: usize, wanted: &[u8]) -> Result<(usize, bool)> {
5550 let mut memo = self.searched.lock().map_err(|_| invalid("a poisoned dictionary search"))?;
5551 if let Some(&answer) = memo.get(wanted) {
5552 return Ok(answer);
5553 }
5554 let answer = search_below(self, ranks, wanted)?;
5555 if memo.len() >= TEXT_SEARCH_MEMO {
5556 memo.clear();
5557 }
5558 memo.insert(wanted.to_vec(), answer);
5559 Ok(answer)
5560 }
5561
5562 fn compare_rank(&self, rank: usize, wanted: &[u8]) -> Result<Ordering> {
5563 let settled = self.head_at(rank)?.cmp(&head(wanted));
5567 if settled != Ordering::Equal {
5568 return Ok(settled);
5569 }
5570 let code = self.code_at_rank(rank)?;
5571 let bytes = self
5572 .bytes_at(code as usize)?
5573 .ok_or_else(|| invalid("global dictionary order names a code it does not have"))?;
5574 Ok(bytes.cmp(wanted))
5575 }
5576
5577 fn code_at_rank(&self, rank: usize) -> Result<u32> {
5578 let (block, within) = self.rank_parts(rank)?;
5579 let codes = self.rank_codes(block, self.rank_block_len(rank))?;
5580 let code = bitpack::tail_at(codes, self.code_bits, within)
5581 .map_err(|_| invalid("global dictionary rank block is short of codes"))?;
5582 let code = u32::try_from(code)
5583 .map_err(|_| invalid("global dictionary order names a code it does not have"))?;
5584 if code as usize >= self.len() {
5585 return Err(invalid("global dictionary order names a code it does not have"));
5586 }
5587 Ok(code)
5588 }
5589
5590 fn code_ranks(&self) -> Option<&[u32]> {
5591 if self.ranks == 0 || self.ranks != self.len() {
5595 return None;
5596 }
5597 self.code_ranks
5598 .get_or_init(|| {
5599 let mut ranks = vec![u32::MAX; self.ranks];
5600 let mut scratch = Vec::new();
5608 let mut codes = vec![0u64; TEXT_RANK_BLOCK];
5609 for first in (0..self.ranks).step_by(TEXT_RANK_BLOCK) {
5610 let which = first / TEXT_RANK_BLOCK;
5611 let block = match self.rank_blocks.get(which)?.get() {
5612 Some(kept) => kept.as_ref().ok()?.as_slice(),
5613 None => {
5614 self.read_rank_block(which, &mut scratch).ok()?;
5615 scratch.as_slice()
5616 }
5617 };
5618 let count = self.rank_block_len(first);
5619 let packed = self.rank_codes(block, count).ok()?;
5620 let codes = codes.get_mut(..count)?;
5621 bitpack::unpack_tail_into(packed, self.code_bits, codes, |bits| bits).ok()?;
5622 for (within, &code) in codes.iter().enumerate() {
5623 let code = usize::try_from(code).ok()?;
5624 *ranks.get_mut(code)? = u32::try_from(first + within).ok()?;
5625 }
5626 }
5627 if ranks.contains(&u32::MAX) {
5628 return None;
5629 }
5630 Some(ranks)
5631 })
5632 .as_deref()
5633 }
5634
5635 fn footprint(&self) -> usize {
5636 self.offsets.capacity()
5637 + self
5638 .value_ends
5639 .get()
5640 .and_then(Option::as_ref)
5641 .map_or(0, |ends| ends.capacity() * size_of::<u32>())
5642 + self.value_lens.get().and_then(Option::as_ref).map_or(0, Lengths::footprint)
5643 + self
5644 .code_ranks
5645 .get()
5646 .and_then(Option::as_ref)
5647 .map_or(0, |ranks| ranks.capacity() * size_of::<u32>())
5648 + self.rank_hashes.capacity() * size_of::<u64>()
5649 + self.rank_ends.capacity() * size_of::<u64>()
5650 + self.rank_blocks.capacity() * size_of::<OnceLock<Result<Vec<u8>>>>()
5651 + self
5652 .rank_blocks
5653 .iter()
5654 .filter_map(OnceLock::get)
5655 .filter_map(|result| result.as_ref().ok())
5656 .map(Vec::capacity)
5657 .sum::<usize>()
5658 + self.blocks.capacity() * size_of::<OnceLock<Result<Vec<u8>>>>()
5659 + self.char_lens.capacity() * size_of::<OnceLock<Box<[u32]>>>()
5660 + self
5661 .char_lens
5662 .iter()
5663 .filter_map(OnceLock::get)
5664 .map(|lens| lens.len() * size_of::<u32>())
5665 .sum::<usize>()
5666 + self.hashes.capacity() * size_of::<u64>()
5667 + self.starts.capacity() * size_of::<u64>()
5668 + self.lengths.capacity() * size_of::<u64>()
5669 + self.grams.as_ref().map_or(0, NativeGrams::footprint)
5670 + self
5671 .blocks
5672 .iter()
5673 .filter_map(OnceLock::get)
5674 .filter_map(|result| result.as_ref().ok())
5675 .map(Vec::capacity)
5676 .sum::<usize>()
5677 }
5678}
5679
5680fn places(table: &Table) -> Result<Vec<Place>> {
5682 let mut places = Vec::with_capacity(table.stripes.len().saturating_mul(STRIPE_PARTS));
5683 for (at, stripe) in table.stripes.iter().enumerate() {
5684 let index = u32::try_from(at).map_err(|_| invalid("too many stripes"))?;
5685 for (part, &rows) in stripe.parts.iter().enumerate() {
5686 places.push(Place {
5687 stripe: index,
5688 part: u32::try_from(part).map_err(|_| invalid("too many parts in a stripe"))?,
5689 rows,
5690 });
5691 }
5692 }
5693 Ok(places)
5694}
5695
5696fn read_index<F: Positional + ?Sized>(
5701 file: &F,
5702 stripe: &Stripe,
5703 column: usize,
5704) -> Result<Vec<PartSpan>> {
5705 let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
5706 read_index_span(file, stripe.index, *page, stripe.parts.len(), column)
5707}
5708
5709fn read_index_span<F: Positional + ?Sized>(
5710 file: &F,
5711 index: Span,
5712 page: Span,
5713 parts: usize,
5714 column: usize,
5715) -> Result<Vec<PartSpan>> {
5716 let section = index_section(parts)?;
5717 let at = column.checked_mul(section).ok_or_else(|| invalid("index page offset overflow"))?;
5718 let end = at.checked_add(section).ok_or_else(|| invalid("index page offset overflow"))?;
5719 if end > index.length as usize {
5720 return Err(invalid("index page is shorter than its columns"));
5721 }
5722 let mut bytes = vec![0; section];
5723 let offset =
5724 index.offset.checked_add(at as u64).ok_or_else(|| invalid("index page offset overflow"))?;
5725 read_at(file, offset, &mut bytes)?;
5726 let entries = section - size_of::<u64>();
5727 let stored = u64::from_le_bytes(bytes[entries..].try_into().expect("eight bytes"));
5728 if checksum(&bytes[..entries]) != stored {
5729 return Err(invalid(&format!(
5732 "index page section checksum differs, column {column} of {parts} parts at {offset}, \
5733 wanted {stored:016x} and got {:016x}",
5734 checksum(&bytes[..entries]),
5735 )));
5736 }
5737 let mut spans = Vec::with_capacity(parts);
5738 let mut start = 0_usize;
5739 for part in 0..parts {
5740 let at = part * INDEX_ENTRY;
5741 let length = u32::from_le_bytes(bytes[at..at + 4].try_into().expect("four bytes")) as usize;
5742 let hash = u64::from_le_bytes(bytes[at + 4..at + 12].try_into().expect("eight bytes"));
5743 spans.push(PartSpan { start, length, hash });
5744 start = start.checked_add(length).ok_or_else(|| invalid("column page length overflow"))?;
5745 }
5746 if start != page.length as usize {
5747 return Err(invalid("column page length differs from its index"));
5748 }
5749 Ok(spans)
5750}
5751
5752fn part_bytes(page: &[u8], span: PartSpan) -> Result<&[u8]> {
5754 let end = span.start.checked_add(span.length).ok_or_else(|| invalid("part range overflow"))?;
5755 page.get(span.start..end).ok_or_else(|| invalid("part exceeds its column page"))
5756}
5757
5758fn remember(cached: &mut Cached, held: &CachedColumn) -> Option<(usize, Arc<AtomicBool>)> {
5764 if let Some(slot) = cached.index.get_mut(held.stripe) {
5765 if slot.is_none() {
5766 *slot = Some(Arc::clone(&held.index));
5767 }
5768 }
5769 let page = held.page.clone()?;
5770 let slot = cached.pages.get_mut(held.stripe)?;
5771 if slot.is_some() {
5772 return None;
5773 }
5774 let bytes = page.bytes.len();
5775 let used = Arc::new(AtomicBool::new(true));
5778 *slot = Some(Resident { page, used: Arc::clone(&used) });
5779 Some((bytes, used))
5780}
5781
5782#[derive(Debug, Clone)]
5791pub struct Catalog {
5792 file: Arc<File>,
5793 size: u64,
5794 entries: Arc<Vec<Entry>>,
5795 views: Arc<Vec<ViewEntry>>,
5797 opening: Opening,
5798 pool: PagePool,
5800}
5801
5802#[derive(Debug, Clone, PartialEq, Eq)]
5804pub struct CertifiedSums {
5805 pub columns: Vec<(i128, u64)>,
5806 pub rows: u64,
5807}
5808
5809#[derive(Debug, Clone, Copy, PartialEq, Eq)]
5811pub enum IntegerExtremes {
5812 Null,
5813 Values { low: i128, high: i128 },
5814}
5815
5816pub type NumericFrequencies = Vec<(Option<i128>, u64)>;
5818
5819impl Catalog {
5820 pub fn open(path: impl AsRef<Path>) -> Result<Self> {
5829 Self::open_in(path, &PagePool::default())
5830 }
5831
5832 pub fn open_in(path: impl AsRef<Path>, pool: &PagePool) -> Result<Self> {
5838 let (file, size, _, bytes, opening) = slot_bytes(path)?;
5839 let (entries, views) = decode_catalog(&bytes, size)?;
5840 Ok(Self {
5841 file: Arc::new(file),
5842 size,
5843 entries: Arc::new(entries),
5844 views: Arc::new(views),
5845 opening,
5846 pool: pool.clone(),
5847 })
5848 }
5849
5850 pub fn names(&self) -> impl ExactSizeIterator<Item = &str> {
5852 self.entries.iter().map(|entry| entry.name.as_str())
5853 }
5854
5855 pub fn rows(&self) -> impl ExactSizeIterator<Item = (&str, usize)> {
5862 self.entries.iter().map(|entry| (entry.name.as_str(), entry.rows))
5863 }
5864
5865 pub fn views(&self) -> impl ExactSizeIterator<Item = &ViewEntry> {
5871 self.views.iter()
5872 }
5873
5874 #[must_use]
5876 pub fn len(&self) -> usize {
5877 self.entries.len()
5878 }
5879
5880 #[must_use]
5883 pub fn is_empty(&self) -> bool {
5884 self.entries.is_empty()
5885 }
5886
5887 pub fn table(&self, name: &str) -> Result<Reader> {
5893 let entry = self
5894 .entries
5895 .iter()
5896 .find(|entry| entry.name == name)
5897 .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
5898 let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
5902 if file_checksum(&self.file, offset, length)? != entry.directory.hash {
5903 return Err(invalid(&format!("the directory of table {name} does not checksum")));
5904 }
5905 let mut opening = self.opening;
5906 opening.reads += 1;
5907 opening.bytes += u64::from(entry.directory.length);
5908 Reader::build(
5909 Arc::clone(&self.file),
5910 self.size,
5911 read_directory(Cursor::over(&self.file, offset, length), self.size, Some(offset))?,
5912 u64::from(entry.directory.length),
5913 opening,
5914 self.pool.clone(),
5915 )
5916 }
5917
5918 pub fn integer_tally(&self, name: &str, column: usize) -> Result<Option<Vec<(i64, u64)>>> {
5926 let mut counts = BTreeMap::<i64, u64>::new();
5927 let Some(()) = self.integer_fold(name, column, |value, count| {
5928 let held = counts.entry(value).or_default();
5929 *held = held.checked_add(count).ok_or_else(|| invalid("integer count overflow"))?;
5930 Ok(())
5931 })?
5932 else {
5933 return Ok(None);
5934 };
5935 Ok(Some(counts.into_iter().collect()))
5936 }
5937
5938 pub fn integer_fold(
5945 &self,
5946 name: &str,
5947 column: usize,
5948 mut emit: impl FnMut(i64, u64) -> Result<()>,
5949 ) -> Result<Option<()>> {
5950 let entry = self
5951 .entries
5952 .iter()
5953 .find(|entry| entry.name == name)
5954 .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
5955 let field =
5956 entry.fields.get(column).ok_or_else(|| invalid("integer column index out of range"))?;
5957 if !signed_integer(&field.ty) {
5958 return Ok(None);
5959 }
5960 let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
5961 if file_checksum(&self.file, offset, length)? != entry.directory.hash {
5962 return Err(invalid(&format!("the directory of table {name} does not checksum")));
5963 }
5964 quick_integer_fold(
5965 &self.file,
5966 Cursor::over(&self.file, offset, length),
5967 entry,
5968 self.size,
5969 column,
5970 &mut emit,
5971 )?;
5972 Ok(Some(()))
5973 }
5974
5975 pub fn nonzero_count(&self, name: &str, column: usize) -> Result<Option<u64>> {
5980 let entry = self
5981 .entries
5982 .iter()
5983 .find(|entry| entry.name == name)
5984 .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
5985 let Some(field) = entry.fields.get(column) else {
5986 return Err(invalid("frequency column index out of range"));
5987 };
5988 if !matches!(
5989 field.ty,
5990 LogicalType::TinyInt
5991 | LogicalType::SmallInt
5992 | LogicalType::Integer
5993 | LogicalType::BigInt
5994 | LogicalType::UTinyInt
5995 | LogicalType::USmallInt
5996 | LogicalType::UInteger
5997 | LogicalType::UBigInt
5998 ) {
5999 return Ok(None);
6000 }
6001 let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
6002 if file_checksum(&self.file, offset, length)? != entry.directory.hash {
6003 return Err(invalid(&format!("the directory of table {name} does not checksum")));
6004 }
6005 if let Some(Some(frequencies)) = entry.frequencies.get(column) {
6006 return frequencies
6007 .iter()
6008 .filter(|(value, _)| value.is_some_and(|value| value != 0))
6009 .try_fold(0_u64, |total, (_, count)| total.checked_add(*count))
6010 .map(Some)
6011 .ok_or_else(|| invalid("numeric frequency count overflow"));
6012 }
6013 quick_nonzero(
6014 Cursor::over(&self.file, offset, length),
6015 &entry.name,
6016 &entry.fields,
6017 entry.rows,
6018 column,
6019 )
6020 }
6021
6022 pub fn aggregate_sums(&self, name: &str, columns: &[usize]) -> Result<Option<CertifiedSums>> {
6025 let entry = self
6026 .entries
6027 .iter()
6028 .find(|entry| entry.name == name)
6029 .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
6030 let mut sums = Vec::with_capacity(columns.len());
6031 for &column in columns {
6032 let Some(field) = entry.fields.get(column) else {
6033 return Err(invalid("aggregate column index out of range"));
6034 };
6035 if !signed_integer(&field.ty) {
6036 return Ok(None);
6037 }
6038 let Some(sum) = entry.aggregates[column] else {
6039 return Ok(None);
6040 };
6041 sums.push(sum);
6042 }
6043 let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
6044 if file_checksum(&self.file, offset, length)? != entry.directory.hash {
6045 return Err(invalid(&format!("the directory of table {name} does not checksum")));
6046 }
6047 Ok(Some(CertifiedSums { columns: sums, rows: entry.rows as u64 }))
6048 }
6049
6050 pub fn distinct_count(&self, name: &str, column: usize) -> Result<Option<u64>> {
6052 let entry = self
6053 .entries
6054 .iter()
6055 .find(|entry| entry.name == name)
6056 .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
6057 let Some(count) = entry.distincts.get(column).copied() else {
6058 return Err(invalid("distinct column index out of range"));
6059 };
6060 let Some(count) = count else { return Ok(None) };
6061 let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
6062 if file_checksum(&self.file, offset, length)? != entry.directory.hash {
6063 return Err(invalid(&format!("the directory of table {name} does not checksum")));
6064 }
6065 Ok(Some(count))
6066 }
6067
6068 pub fn integer_extremes(&self, name: &str, column: usize) -> Result<Option<IntegerExtremes>> {
6070 let entry = self
6071 .entries
6072 .iter()
6073 .find(|entry| entry.name == name)
6074 .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
6075 let Some(extremes) = entry.extremes.get(column).copied() else {
6076 return Err(invalid("extremes column index out of range"));
6077 };
6078 let Some(extremes) = extremes else { return Ok(None) };
6079 let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
6080 if file_checksum(&self.file, offset, length)? != entry.directory.hash {
6081 return Err(invalid(&format!("the directory of table {name} does not checksum")));
6082 }
6083 Ok(Some(match extremes {
6084 None => IntegerExtremes::Null,
6085 Some((low, high)) => IntegerExtremes::Values { low, high },
6086 }))
6087 }
6088
6089 pub fn exact_numeric_frequencies(
6091 &self,
6092 name: &str,
6093 column: usize,
6094 ) -> Result<Option<NumericFrequencies>> {
6095 let entry = self
6096 .entries
6097 .iter()
6098 .find(|entry| entry.name == name)
6099 .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
6100 let Some(frequencies) = entry.frequencies.get(column).cloned() else {
6101 return Err(invalid("numeric frequency column index out of range"));
6102 };
6103 let Some(frequencies) = frequencies else { return Ok(None) };
6104 let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
6105 if file_checksum(&self.file, offset, length)? != entry.directory.hash {
6106 return Err(invalid(&format!("the directory of table {name} does not checksum")));
6107 }
6108 Ok(Some(frequencies))
6109 }
6110
6111 pub fn table_fields(&self, name: &str) -> Option<&[Field]> {
6113 self.entries.iter().find(|entry| entry.name == name).map(|entry| entry.fields.as_slice())
6114 }
6115}
6116
6117fn slot_offset(generation: u64) -> u64 {
6122 16 + (generation - 1) % 2 * SLOT_BYTES as u64
6123}
6124
6125fn slot_bytes(path: impl AsRef<Path>) -> Result<(File, u64, Slot, Vec<u8>, Opening)> {
6130 let file = File::open(path).map_err(io)?;
6131 let size = file.metadata().map_err(io)?.len();
6132 let (slot, bytes, opening) = committed_slot(&file, size)?;
6133 Ok((file, size, slot, bytes, opening))
6134}
6135
6136fn committed_slot<F: Positional + ?Sized>(file: &F, size: u64) -> Result<(Slot, Vec<u8>, Opening)> {
6142 if size < HEADER {
6143 return Err(invalid("file is shorter than its header"));
6144 }
6145 let mut header = [0; HEADER as usize];
6146 read_at(file, 0, &mut header)?;
6147 let mut opening = Opening { reads: 1, bytes: HEADER };
6148 let version = u32::from_le_bytes([header[8], header[9], header[10], header[11]]);
6149 if &header[..8] != MAGIC {
6154 return Err(invalid("the header does not begin with a rudb native magic"));
6155 }
6156 if !READABLE.contains(&version) {
6157 return Err(invalid(&format!(
6158 "the file is format {version} and this build reads format {FORMAT}, so it has to \
6159 be written again"
6160 )));
6161 }
6162 let mut selected = None;
6163 for start in [16, 16 + SLOT_BYTES] {
6164 let slot = Slot::read(&header[start..start + SLOT_BYTES]);
6165 if slot.generation == 0 || slot.length == 0 || slot.length as usize > MAX_DIRECTORY {
6166 continue;
6167 }
6168 let Some(end) = slot.offset.checked_add(u64::from(slot.length)) else { continue };
6169 if slot.offset < HEADER || end > size {
6170 continue;
6171 }
6172 let mut bytes = vec![0; slot.length as usize];
6173 read_at(file, slot.offset, &mut bytes)?;
6174 opening.reads += 1;
6175 opening.bytes += u64::from(slot.length);
6176 if checksum(&bytes) == slot.hash
6177 && selected
6178 .as_ref()
6179 .is_none_or(|(old, _): &(Slot, Vec<u8>)| old.generation < slot.generation)
6180 {
6181 selected = Some((slot, bytes));
6182 }
6183 }
6184 let (slot, bytes) = selected.ok_or_else(|| invalid("no committed directory slot is valid"))?;
6185 Ok((slot, bytes, opening))
6186}
6187
6188impl Reader {
6189 pub fn open(path: impl AsRef<Path>) -> Result<Self> {
6196 let catalog = Catalog::open(path)?;
6197 let mut names = catalog.names();
6198 let name = names.next().ok_or_else(|| invalid("the file holds no table"))?.to_string();
6199 if names.next().is_some() {
6200 return Err(invalid(
6201 "the file holds more than one table, so it has to be opened by name",
6202 ));
6203 }
6204 catalog.table(&name)
6205 }
6206
6207 fn build(
6209 file: Arc<File>,
6210 size: u64,
6211 table: Table,
6212 directory: u64,
6213 opening: Opening,
6214 pool: PagePool,
6215 ) -> Result<Self> {
6216 let places = places(&table)?;
6217 let dictionaries = (0..table.fields.len()).map(|_| OnceLock::new()).collect();
6218 let table_fields = table.fields.len();
6219 let stripes = table.stripes.len();
6220 let columns = (0..table.fields.len())
6221 .map(|_| {
6222 Mutex::new(Cached {
6223 pages: (0..stripes).map(|_| None).collect(),
6224 index: (0..stripes).map(|_| None).collect(),
6225 seen: vec![false; stripes],
6226 ..Cached::default()
6227 })
6228 })
6229 .collect::<Vec<_>>();
6230 let cache = Shelf {
6231 columns,
6232 held: (0..table_fields).map(|_| AtomicUsize::new(0)).collect(),
6233 kept: AtomicUsize::new(CACHED_STRIPES_PER_COLUMN),
6234 };
6235 let sieves: Vec<Vec<SieveSlot>> = (0..table.fields.len())
6236 .map(|_| table.stripes.iter().map(|_| OnceLock::new()).collect())
6237 .collect();
6238 let part_ranges: Vec<Vec<RangeSlot>> = (0..table.fields.len())
6239 .map(|_| table.stripes.iter().map(|_| OnceLock::new()).collect())
6240 .collect();
6241 Ok(Self {
6242 file,
6243 table: Arc::new(table),
6244 dictionaries: Arc::new(dictionaries),
6245 loading: Arc::new((0..table_fields).map(|_| Mutex::new(())).collect()),
6246 frequency_values: Arc::new((0..table_fields).map(|_| OnceLock::new()).collect()),
6247 frequency_summaries: Arc::new((0..table_fields).map(|_| OnceLock::new()).collect()),
6248 opened: Arc::new(AtomicUsize::new(0)),
6249 sieves: Arc::new(sieves),
6250 part_ranges: Arc::new(part_ranges),
6251 places: Arc::new(places),
6252 cache: Arc::new(cache),
6253 pool,
6254 pages: Arc::new(AtomicUsize::new(0)),
6255 indexes: Arc::new(AtomicUsize::new(0)),
6256 size,
6257 directory,
6258 opening,
6259 })
6260 }
6261
6262 #[must_use]
6269 pub fn reads(&self) -> Reads {
6270 Reads {
6271 opening: self.opening,
6272 pages: self.pages.load(Atomic::Relaxed),
6273 indexes: self.indexes.load(Atomic::Relaxed),
6274 dictionaries: self.opened.load(Atomic::Relaxed),
6275 }
6276 }
6277
6278 #[must_use]
6283 pub fn layout(&self) -> Layout {
6284 let table = &self.table;
6285 let stripes = table.stripes.as_slice();
6286 let columns = table
6287 .fields
6288 .iter()
6289 .enumerate()
6290 .map(|(at, field)| ColumnLayout {
6291 name: field.name.clone(),
6292 kind: field.ty.to_string(),
6293 pages: sum(stripes.iter().map(|stripe| span_bytes(&stripe.pages, at))),
6294 memberships: sum(stripes.iter().map(|stripe| stripe.memberships.bytes(at))),
6295 sieves: sum(stripes.iter().map(|stripe| stripe.sieves.bytes(at))),
6296 part_ranges: sum(stripes.iter().map(|stripe| stripe.part_ranges.bytes(at))),
6297 dictionary: dictionary_bytes(table, at),
6298 })
6299 .collect();
6300 Layout {
6301 file: self.size,
6302 rows: table.rows,
6303 stripes: stripes.len(),
6304 parts: self.places.len(),
6305 columns,
6306 indexes: sum(stripes.iter().map(|stripe| u64::from(stripe.index.length))),
6307 directory: self.directory,
6308 header: HEADER,
6309 }
6310 }
6311
6312 pub fn stored(&self, column: usize) -> Result<Vec<StoredPart>> {
6329 let field = self
6330 .table
6331 .fields
6332 .get(column)
6333 .ok_or_else(|| invalid("stored column index out of range"))?;
6334 let mut stored = Vec::with_capacity(self.places.len());
6335 let mut row = 0;
6336 for (at, stripe) in self.table.stripes.iter().enumerate() {
6337 let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
6338 let index = read_index(&self.file, stripe, column)?;
6339 let mut bytes = vec![0; page.length as usize];
6340 read_at(&self.file, page.offset, &mut bytes)?;
6341 let ranges = self.stripe_part_ranges(at, column);
6342 for (part, &rows) in stripe.parts.iter().enumerate() {
6343 let span = *index.get(part).ok_or_else(|| invalid("part index out of range"))?;
6344 let held = part_bytes(&bytes, span)?;
6345 let range = ranges.and_then(|held| held.get(part));
6346 stored.push(StoredPart {
6347 stripe: at,
6348 part,
6349 row,
6350 rows: rows as usize,
6351 encoding: page_encoding(&field.ty, rows as usize, held),
6352 bytes: span.length as u64,
6353 page: page.offset,
6354 offset: span.start as u64,
6355 low: range
6356 .and_then(|range| range.low.clone())
6357 .and_then(|bound| bound.into_value(&field.ty)),
6358 high: range
6359 .and_then(|range| range.high.clone())
6360 .and_then(|bound| bound.into_value(&field.ty)),
6361 nulls: range.map(|range| range.nulls),
6362 });
6363 row += rows as usize;
6364 }
6365 }
6366 Ok(stored)
6367 }
6368
6369 #[must_use]
6371 pub fn parts(&self) -> usize {
6372 self.places.len()
6373 }
6374
6375 #[must_use]
6382 pub fn stripe_parts(&self) -> Vec<std::ops::Range<usize>> {
6383 let mut runs = Vec::with_capacity(self.table.stripes.len());
6384 let mut start = 0;
6385 for stripe in &self.table.stripes {
6386 let end = start + stripe.parts.len();
6387 runs.push(start..end);
6388 start = end;
6389 }
6390 runs
6391 }
6392
6393 #[must_use]
6398 pub fn stripe_rows(&self, stripe: usize) -> usize {
6399 self.table.stripes.get(stripe).map_or(0, |held| held.rows)
6400 }
6401
6402 pub fn keep_stripes(&self, stripes: usize) {
6409 self.cache.kept.fetch_max(stripes, Atomic::Relaxed);
6410 }
6411
6412 #[must_use]
6414 pub fn part_rows(&self, at: usize) -> usize {
6415 self.places.get(at).map_or(0, |place| place.rows as usize)
6416 }
6417
6418 #[must_use]
6420 pub fn table(&self) -> &Table {
6421 &self.table
6422 }
6423
6424 pub fn top_frequencies(&self, column: usize, top: usize) -> Result<Option<Vec<(Value, u64)>>> {
6433 let field = self
6434 .table
6435 .fields
6436 .get(column)
6437 .ok_or_else(|| invalid("frequency column index out of range"))?;
6438 let Some(summary) = self.frequency_summary(column)? else {
6439 return Ok(None);
6440 };
6441 if top == 0 || summary.entries.len() < top {
6442 return Ok(None);
6443 }
6444 let boundary = summary.entries[top - 1].count;
6445 if boundary <= summary.omitted_max {
6446 return Ok(None);
6447 }
6448 self.decode_frequencies(column, &field.ty, &summary.entries).map(Some)
6449 }
6450
6451 pub fn top_pair_frequencies(
6459 &self,
6460 first: usize,
6461 second: usize,
6462 _top: usize,
6463 ) -> Result<Option<PairFrequencyCounts>> {
6464 if first >= self.table.fields.len() || second >= self.table.fields.len() {
6465 return Err(invalid("pair frequency column index out of range"));
6466 }
6467 Ok(None)
6468 }
6469
6470 pub fn exact_frequencies(&self, column: usize) -> Result<Option<Vec<(Value, u64)>>> {
6490 let Some(prefix) = self.frequency_prefix(column)? else {
6491 return Ok(None);
6492 };
6493 Ok((prefix.omitted_max == 0).then_some(prefix.entries))
6494 }
6495
6496 pub fn frequency_prefix(&self, column: usize) -> Result<Option<FrequencyPrefix>> {
6519 let field = self
6520 .table
6521 .fields
6522 .get(column)
6523 .ok_or_else(|| invalid("frequency column index out of range"))?;
6524 let Some(summary) = self.frequency_summary(column)? else {
6525 return Ok(None);
6526 };
6527 let entries = self.decode_frequencies(column, &field.ty, &summary.entries)?;
6528 Ok(Some(FrequencyPrefix { entries, omitted_max: summary.omitted_max }))
6529 }
6530
6531 fn frequency_summary(&self, column: usize) -> Result<Option<Cow<'_, FrequencySummary>>> {
6533 Ok(match self.table.frequencies.get(column) {
6534 None | Some(None) => None,
6535 Some(Some(Frequencies::Held(summary))) => Some(Cow::Borrowed(summary)),
6536 Some(Some(Frequencies::Stored { span, values, entries })) => {
6537 let slot = self
6538 .frequency_summaries
6539 .get(column)
6540 .ok_or_else(|| invalid("frequency column index out of range"))?;
6541 if let Some(summary) = slot.get() {
6542 return Ok(Some(Cow::Borrowed(summary.as_ref())));
6543 }
6544 let field = self
6545 .table
6546 .fields
6547 .get(column)
6548 .ok_or_else(|| invalid("frequency column index out of range"))?;
6549 let mut bytes = vec![0; span.length as usize];
6550 read_at(&self.file, span.offset, &mut bytes)?;
6551 let mut cur = Cursor::new(&bytes);
6552 let summary = decode_summary(&mut cur, field, self.table.rows, *values)?;
6553 let summary = summary.ok_or_else(|| invalid("a stored synopsis is missing"))?;
6554 if !cur.done() || summary.entries.len() != *entries {
6555 return Err(invalid("a stored synopsis differs from its directory span"));
6556 }
6557 let _ = slot.set(Arc::new(summary));
6558 Some(Cow::Borrowed(slot.get().expect("the decoded summary was stored").as_ref()))
6559 }
6560 })
6561 }
6562
6563 fn decode_frequencies(
6571 &self,
6572 column: usize,
6573 ty: &LogicalType,
6574 entries: &[FrequencyEntry],
6575 ) -> Result<Vec<(Value, u64)>> {
6576 if let Some(values) = self.frequency_values.get(column).and_then(OnceLock::get) {
6577 return Ok(values.as_ref().clone());
6578 }
6579 let values = self.decode_frequencies_once(column, ty, entries)?;
6580 if let Some(slot) = self.frequency_values.get(column) {
6581 let _ = slot.set(Arc::new(values.clone()));
6582 }
6583 Ok(values)
6584 }
6585
6586 fn decode_frequencies_once(
6587 &self,
6588 column: usize,
6589 ty: &LogicalType,
6590 entries: &[FrequencyEntry],
6591 ) -> Result<Vec<(Value, u64)>> {
6592 let stored_texts = self.table.frequency_texts.get(column).filter(|texts| !texts.is_empty());
6593 if stored_texts.is_some_and(|texts| texts.len() != entries.len()) {
6594 return Err(invalid("frequency text count differs from its synopsis"));
6595 }
6596 let dictionary =
6597 if coded_type(ty) && stored_texts.is_none() { self.dictionary(column)? } else { None };
6598 let mut codes = entries
6599 .iter()
6600 .filter_map(|entry| match entry.value {
6601 FrequencyValue::Code(code) => Some(code as usize),
6602 _ => None,
6603 })
6604 .collect::<Vec<_>>();
6605 codes.sort_unstable();
6606 codes.dedup();
6607 let texts = match &dictionary {
6608 Some(dictionary) if !codes.is_empty() => dictionary.try_values_visited(&codes)?,
6609 _ => Vec::new(),
6610 };
6611 let mut out = Vec::with_capacity(entries.len());
6612 for (entry_at, entry) in entries.iter().enumerate() {
6613 let value = match entry.value {
6614 FrequencyValue::Null => {
6615 if stored_texts.and_then(|texts| texts[entry_at].as_ref()).is_some() {
6616 return Err(invalid("a null frequency entry has text"));
6617 }
6618 Value::Null
6619 }
6620 FrequencyValue::Integer(value) => match *ty {
6621 LogicalType::TinyInt => Value::TinyInt(
6622 i8::try_from(value)
6623 .map_err(|_| invalid("frequency TINYINT is out of range"))?,
6624 ),
6625 LogicalType::UTinyInt => Value::UTinyInt(
6626 u8::try_from(value)
6627 .map_err(|_| invalid("frequency UTINYINT is out of range"))?,
6628 ),
6629 LogicalType::USmallInt => Value::USmallInt(
6630 u16::try_from(value)
6631 .map_err(|_| invalid("frequency USMALLINT is out of range"))?,
6632 ),
6633 LogicalType::UInteger => Value::UInteger(
6634 u32::try_from(value)
6635 .map_err(|_| invalid("frequency UINTEGER is out of range"))?,
6636 ),
6637 LogicalType::UBigInt => Value::UBigInt(
6638 u64::try_from(value)
6639 .map_err(|_| invalid("frequency UBIGINT is out of range"))?,
6640 ),
6641 LogicalType::SmallInt => Value::SmallInt(
6642 i16::try_from(value)
6643 .map_err(|_| invalid("frequency SMALLINT is out of range"))?,
6644 ),
6645 LogicalType::Integer => Value::Integer(
6646 i32::try_from(value)
6647 .map_err(|_| invalid("frequency INTEGER is out of range"))?,
6648 ),
6649 LogicalType::BigInt => Value::BigInt(
6650 i64::try_from(value)
6651 .map_err(|_| invalid("frequency BIGINT is out of range"))?,
6652 ),
6653 LogicalType::Date => Value::Date(
6654 i32::try_from(value)
6655 .map_err(|_| invalid("frequency DATE is out of range"))?,
6656 ),
6657 LogicalType::Timestamp => Value::Timestamp(
6658 i64::try_from(value)
6659 .map_err(|_| invalid("frequency TIMESTAMP is out of range"))?,
6660 ),
6661 _ => return Err(invalid("integer frequency belongs to another type")),
6662 },
6663 FrequencyValue::Code(code) => {
6664 if let Some(text) = stored_texts.and_then(|texts| texts[entry_at].as_ref()) {
6665 if *ty == LogicalType::Blob {
6666 Value::Blob(text.clone())
6667 } else {
6668 Value::Varchar(
6669 String::from_utf8(text.clone())
6670 .map_err(|_| invalid("frequency text is not UTF-8"))?,
6671 )
6672 }
6673 } else {
6674 if dictionary.is_none() {
6675 return Err(invalid("frequency code has no dictionary or stored text"));
6676 }
6677 let at = codes
6678 .binary_search(&(code as usize))
6679 .map_err(|_| invalid("frequency code was not among the codes read"))?;
6680 texts[at].clone()
6681 }
6682 }
6683 };
6684 out.push((value, entry.count));
6685 }
6686 Ok(out)
6687 }
6688
6689 pub fn frequency_occurrences(&self, column: usize) -> Result<Option<FrequencyOccurrences>> {
6699 let field = self
6700 .table
6701 .fields
6702 .get(column)
6703 .ok_or_else(|| invalid("frequency column index out of range"))?;
6704 let Some(summary) = self.frequency_summary(column)? else {
6705 return Ok(None);
6706 };
6707 if summary.ordinals.is_empty() {
6708 return Ok(None);
6709 }
6710 let (anchors, anchor_indices) = if summary.ordinal_entries.len() == summary.ordinals.len() {
6711 let entries = self.decode_frequencies(column, &field.ty, &summary.entries)?;
6712 (entries.into_iter().map(|(value, _)| value).collect(), summary.ordinal_entries.clone())
6713 } else {
6714 (Vec::new(), Vec::new())
6715 };
6716 Ok(Some(FrequencyOccurrences {
6717 omitted_max: summary.omitted_max,
6718 ordinals: summary.ordinals.clone(),
6719 anchors,
6720 anchor_indices,
6721 }))
6722 }
6723
6724 pub fn distinct_values(&self, column: usize) -> Result<Option<u64>> {
6750 self.table
6751 .distincts
6752 .get(column)
6753 .copied()
6754 .ok_or_else(|| invalid("distinct column index out of range"))
6755 }
6756
6757 pub fn null_count(&self, column: usize) -> Result<u64> {
6768 if column >= self.table.fields.len() {
6769 return Err(invalid("null count column index out of range"));
6770 }
6771 let mut nulls = 0_u64;
6772 for stripe in &self.table.stripes {
6773 let range = stripe
6774 .zone
6775 .column(column)
6776 .ok_or_else(|| invalid("stripe zone is narrower than the schema"))?;
6777 nulls = nulls
6778 .checked_add(range.nulls as u64)
6779 .ok_or_else(|| invalid("null count overflow"))?;
6780 }
6781 Ok(nulls)
6782 }
6783
6784 pub fn text_extremes(&self, column: usize) -> Result<Option<(Value, Value)>> {
6799 if self.null_count(column)? > 0 || self.demoted(column) {
6800 return Ok(None);
6801 }
6802 let Some(dictionary) = self.dictionary(column)? else { return Ok(None) };
6803 let Some(ranks) = dictionary.ranks() else { return Ok(None) };
6804 if ranks == 0 {
6805 return Ok(None);
6806 }
6807 let low = text_at_rank(&dictionary, 0)?;
6808 let high = text_at_rank(&dictionary, ranks - 1)?;
6809 Ok(Some((low, high)))
6810 }
6811
6812 pub fn exact_extremes(&self, column: usize) -> Result<Option<(Bound, Bound)>> {
6835 if column >= self.table.fields.len() {
6836 return Err(invalid("extremes column index out of range"));
6837 }
6838 let mut low: Option<Bound> = None;
6839 let mut high: Option<Bound> = None;
6840 for stripe in &self.table.stripes {
6841 let range = stripe
6842 .zone
6843 .column(column)
6844 .ok_or_else(|| invalid("stripe zone is narrower than the schema"))?;
6845 if !range.exact {
6846 return Ok(None);
6847 }
6848 let (Some(small), Some(large)) = (range.low.as_ref(), range.high.as_ref()) else {
6853 if stripe.rows > range.nulls {
6854 return Ok(None);
6855 }
6856 continue;
6857 };
6858 low = Some(low.map_or_else(|| small.clone(), |held| held.smaller(small.clone())));
6859 high = Some(high.map_or_else(|| large.clone(), |held| held.larger(large.clone())));
6860 }
6861 Ok(low.zip(high))
6862 }
6863
6864 pub fn exact_sum(&self, column: usize) -> Result<Option<(i128, u64)>> {
6877 if column >= self.table.fields.len() {
6878 return Err(invalid("sum column index out of range"));
6879 }
6880 let mut total = 0_i128;
6881 let mut rows = 0_u64;
6882 for stripe in &self.table.stripes {
6883 let range = stripe
6884 .zone
6885 .column(column)
6886 .ok_or_else(|| invalid("stripe zone is narrower than the schema"))?;
6887 let Some(part) = range.sum else { return Ok(None) };
6888 let Some(sum) = total.checked_add(part) else { return Ok(None) };
6889 total = sum;
6890 rows = rows.saturating_add(stripe.rows as u64 - range.nulls as u64);
6891 }
6892 Ok(Some((total, rows)))
6893 }
6894
6895 pub fn host_groups(
6897 &self,
6898 column: usize,
6899 _minimum_count: u64,
6900 ) -> Result<Option<Vec<host::HostEntry>>> {
6901 if column >= self.table.fields.len() {
6902 return Err(invalid("host group column index out of range"));
6903 }
6904 Ok(None)
6905 }
6906
6907 #[must_use]
6911 pub fn demoted(&self, column: usize) -> bool {
6912 self.table.demoted.get(column).copied().unwrap_or(false)
6913 }
6914
6915 fn dictionary(&self, column: usize) -> Result<Option<Arc<Vector>>> {
6924 let Some(page) = self.table.dictionaries[column] else { return Ok(None) };
6925 if let Some(dictionary) = self.dictionaries[column].get() {
6926 return Ok(Some(Arc::clone(dictionary)));
6927 }
6928 let _queued = self.loading[column].lock().map_err(|_| invalid("a poisoned dictionary"))?;
6929 if let Some(dictionary) = self.dictionaries[column].get() {
6930 return Ok(Some(Arc::clone(dictionary)));
6931 }
6932 self.opened.fetch_add(1, Atomic::Relaxed);
6933 let dictionary = Arc::new(open_global_dictionary(
6934 Arc::clone(&self.file),
6935 page,
6936 &self.table.fields[column].ty,
6937 TEXT_KEEP_BUDGET,
6938 )?);
6939 let _ = self.dictionaries[column].set(Arc::clone(&dictionary));
6940 Ok(Some(dictionary))
6941 }
6942
6943 pub fn extents(&self, of: &Section) -> Result<Vec<section::Extent>> {
6950 if of.extent_bytes == 0 {
6951 return Ok(Vec::new());
6952 }
6953 let mut bytes = vec![0; of.extent_bytes as usize];
6954 read_at(&self.file, of.extent_page, &mut bytes)?;
6955 if checksum(&bytes) != of.hash {
6956 return Err(invalid("a section's extent table does not checksum"));
6957 }
6958 let extents = section::decode_extents(&bytes)?;
6959 if extents.len() != of.extents as usize {
6960 return Err(invalid("a section's extent table is not the length the entry says"));
6961 }
6962 Ok(extents)
6963 }
6964
6965 pub fn extent(&self, of: §ion::Extent) -> Result<Vec<u8>> {
6975 let mut bytes = Vec::new();
6976 self.extent_into(of, &mut bytes)?;
6977 Ok(bytes)
6978 }
6979
6980 fn extent_into(&self, of: §ion::Extent, bytes: &mut Vec<u8>) -> Result<()> {
6982 let end = of
6983 .offset
6984 .checked_add(u64::from(of.length))
6985 .ok_or_else(|| invalid("an extent overflows the file"))?;
6986 if of.offset < HEADER || end > self.size {
6987 return Err(invalid("an extent is outside the file"));
6988 }
6989 bytes.resize(of.length as usize, 0);
6990 read_at(&self.file, of.offset, bytes)?;
6991 if checksum(bytes) != of.hash {
6992 return Err(invalid("an extent does not checksum"));
6993 }
6994 Ok(())
6995 }
6996
6997 pub fn payload_head(&self, of: &Section, len: usize) -> Result<Vec<u8>> {
7010 let extents = self.extents(of)?;
7011 let Some(first) = extents.first() else { return Ok(Vec::new()) };
7012 let end = first
7013 .offset
7014 .checked_add(u64::from(first.length))
7015 .ok_or_else(|| invalid("an extent overflows the file"))?;
7016 if first.offset < HEADER || end > self.size {
7017 return Err(invalid("an extent is outside the file"));
7018 }
7019 let mut bytes = vec![0; len.min(first.length as usize)];
7020 read_at(&self.file, first.offset, &mut bytes)?;
7021 Ok(bytes)
7022 }
7023
7024 pub fn payload(&self, of: &Section) -> Result<Vec<u8>> {
7033 let extents = self.extents(of)?;
7034 let mut bytes =
7035 Vec::with_capacity(sum(extents.iter().map(|one| u64::from(one.length))) as usize);
7036 for one in &extents {
7037 if one.first != bytes.len() as u64 {
7038 return Err(invalid("a section's extents do not join up"));
7039 }
7040 bytes.extend_from_slice(&self.extent(one)?);
7041 }
7042 if !bytes.is_empty() && of.header_bytes as usize > bytes.len() {
7045 return Err(invalid("a section's header is longer than its payload"));
7046 }
7047 Ok(bytes)
7048 }
7049
7050 pub fn read(&self, part: usize, columns: &[usize]) -> Result<Chunk> {
7059 self.read_impl(part, columns, true, None)
7060 }
7061
7062 pub fn read_sparse(&self, part: usize, columns: &[usize]) -> Result<Chunk> {
7072 self.read_impl(part, columns, false, None)
7073 }
7074
7075 pub fn integer_tally(&self, part: usize, column: usize) -> Result<Option<Vec<(i64, u64)>>> {
7083 let place = *self.places.get(part).ok_or_else(|| invalid("part index out of range"))?;
7084 let field =
7085 self.table.fields.get(column).ok_or_else(|| invalid("column index out of range"))?;
7086 if !matches!(
7087 field.ty,
7088 LogicalType::TinyInt
7089 | LogicalType::SmallInt
7090 | LogicalType::Integer
7091 | LogicalType::BigInt
7092 ) {
7093 return Ok(None);
7094 }
7095 let (rows, counts) = match self.with_part(place, column, |bytes| {
7096 if bytes.first() != Some(&5) || bytes.get(1) != Some(&0) {
7097 return Ok(None);
7098 }
7099 integer::tally(&bytes[2..]).map(Some)
7100 })? {
7101 Some(tallied) => tallied,
7102 None => return Ok(None),
7103 };
7104 if rows != place.rows as usize {
7105 return Err(invalid("encoded integer part holds the wrong number of rows"));
7106 }
7107 for &(value, _) in &counts {
7108 let fits = match field.ty {
7109 LogicalType::TinyInt => i8::try_from(value).is_ok(),
7110 LogicalType::SmallInt => i16::try_from(value).is_ok(),
7111 LogicalType::Integer => i32::try_from(value).is_ok(),
7112 LogicalType::BigInt => true,
7113 _ => false,
7114 };
7115 if !fits {
7116 return Err(invalid("encoded integer value is outside its column type"));
7117 }
7118 }
7119 Ok(Some(counts))
7120 }
7121
7122 pub fn rows_holding(
7134 &self,
7135 part: usize,
7136 column: usize,
7137 sequence: &Sequence,
7138 negated: bool,
7139 ) -> Result<Option<Vec<u32>>> {
7140 let place = *self.places.get(part).ok_or_else(|| invalid("part index out of range"))?;
7141 let field =
7142 self.table.fields.get(column).ok_or_else(|| invalid("column index out of range"))?;
7143 if field.ty != LogicalType::Varchar {
7144 return Ok(None);
7145 }
7146 let rows = place.rows as usize;
7147 self.with_part(place, column, |bytes| {
7148 if bytes.first() != Some(&6) {
7149 return Ok(None);
7150 }
7151 let mut cur = Cursor::new(bytes);
7152 cur.u8()?;
7153 let mask = match cur.u8()? {
7154 0 => None,
7155 1 => return Ok(Some(Vec::new())),
7156 2 => {
7157 let from = cur.at;
7158 cur.take(rows.div_ceil(8))?;
7159 Some(&bytes[from..cur.at])
7160 }
7161 _ => return Err(invalid("page validity tag differs")),
7162 };
7163 let Some(held) = string::holds_in(&bytes[cur.at..], sequence)? else {
7164 return Ok(None);
7165 };
7166 if held.len() != rows {
7167 return Err(invalid("compressed text page holds the wrong number of rows"));
7168 }
7169 let valid = |row: usize| mask.is_none_or(|mask| mask[row / 8] >> (row % 8) & 1 == 1);
7170 Ok(Some(
7171 (0..rows)
7172 .filter(|&row| held[row] != negated && valid(row))
7173 .map(|row| row as u32)
7174 .collect(),
7175 ))
7176 })
7177 }
7178
7179 fn with_part<T>(
7182 &self,
7183 place: Place,
7184 column: usize,
7185 read: impl FnOnce(&[u8]) -> Result<T>,
7186 ) -> Result<T> {
7187 let stripe_index = place.stripe as usize;
7188 let stripe = self
7189 .table
7190 .stripes
7191 .get(stripe_index)
7192 .ok_or_else(|| invalid("stripe index out of range"))?;
7193 let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
7194 let held = self.held(stripe_index, stripe, column, true)?;
7195 let span = *held
7196 .index
7197 .get(place.part as usize)
7198 .ok_or_else(|| invalid("part index out of range"))?;
7199 match &held.page {
7200 Some(page) => read(page.part(place.part as usize, span)?),
7201 None => {
7202 let offset = page
7203 .offset
7204 .checked_add(span.start as u64)
7205 .ok_or_else(|| invalid("part range overflow"))?;
7206 let mut bytes = vec![0; span.length];
7207 read_at(&self.file, offset, &mut bytes)?;
7208 verify_part(&bytes, span)?;
7209 read(&bytes)
7210 }
7211 }
7212 }
7213
7214 pub fn read_rows(
7227 &self,
7228 part: usize,
7229 columns: &[usize],
7230 positions: &[u32],
7231 whole: bool,
7232 ) -> Result<Chunk> {
7233 self.read_impl(part, columns, whole, Some(positions))
7234 }
7235
7236 pub fn skips_codes(&self, part: usize, column: usize, candidates: &[u32]) -> Result<bool> {
7243 if self.demoted(column) {
7246 return Ok(false);
7247 }
7248 if candidates.is_empty() {
7249 return Ok(true);
7250 }
7251 if candidates.windows(2).any(|pair| pair[0] >= pair[1]) {
7252 return Err(Error::internal("native code candidates are not sorted and unique"));
7253 }
7254 let stripe = self.stripe_of(part)?;
7255 let Some(page) = stripe.memberships.get(column) else {
7256 return Ok(false);
7257 };
7258 let mut bytes = vec![0; page.length as usize];
7259 read_at(&self.file, page.offset, &mut bytes)?;
7260 if checksum(&bytes) != page.hash {
7261 return Err(invalid("membership page checksum differs"));
7262 }
7263 let codes = decode_membership(&bytes)?;
7264 let mut left = 0;
7265 let mut right = 0;
7266 while left < codes.len() && right < candidates.len() {
7267 match codes[left].cmp(&candidates[right]) {
7268 Ordering::Less => left += 1,
7269 Ordering::Greater => right += 1,
7270 Ordering::Equal => return Ok(false),
7271 }
7272 }
7273 Ok(true)
7274 }
7275
7276 fn stripe_of(&self, part: usize) -> Result<&Stripe> {
7277 let place = self.places.get(part).ok_or_else(|| invalid("part index out of range"))?;
7278 self.table
7279 .stripes
7280 .get(place.stripe as usize)
7281 .ok_or_else(|| invalid("stripe index out of range"))
7282 }
7283
7284 fn held(&self, at: usize, stripe: &Stripe, column: usize, whole: bool) -> Result<CachedColumn> {
7301 let cache =
7302 self.cache.columns.get(column).ok_or_else(|| invalid("column index out of range"))?;
7303 let mut cached = cache.lock().map_err(|_| invalid("column page cache is poisoned"))?;
7304 let known = cached.index.get(at).and_then(Clone::clone);
7305 let page = cached.pages.get(at).and_then(Option::as_ref).map(|slot| {
7306 slot.used.store(true, Atomic::Relaxed);
7307 Arc::clone(&slot.page)
7308 });
7309 if let Some(index) = known.clone() {
7310 if !whole || page.is_some() {
7311 return Ok(CachedColumn { stripe: at, index, page });
7312 }
7313 }
7314 if cached.loading.contains(&at) {
7315 drop(cached);
7316 if let Some(index) = known {
7320 return Ok(CachedColumn { stripe: at, index, page: None });
7321 }
7322 let held = self.page_of(stripe, column, at, false, None)?;
7323 let mut cached = cache.lock().map_err(|_| invalid("column page cache is poisoned"))?;
7324 remember(&mut cached, &held);
7325 return Ok(held);
7326 }
7327 cached.loading.push(at);
7328 drop(cached);
7329
7330 let read = self.page_of(stripe, column, at, whole, known);
7331
7332 let mut cached = cache.lock().map_err(|_| invalid("column page cache is poisoned"))?;
7336 if let Some(position) = cached.loading.iter().position(|loading| *loading == at) {
7337 cached.loading.remove(position);
7338 }
7339 let held = read?;
7340 let taken = remember(&mut cached, &held);
7341 let first = taken.is_some()
7342 && cached.seen.get_mut(at).is_some_and(|seen| !std::mem::replace(seen, true));
7343 if first {
7344 let floor = self.cache.kept.load(Atomic::Relaxed).max(1);
7345 cached.passing.push_back(at);
7346 while cached.passing.len() > floor {
7347 let Some(old) = cached.passing.pop_front() else { break };
7348 if let Some(slot) = cached.pages.get_mut(old) {
7349 *slot = None;
7350 }
7351 }
7352 return Ok(held);
7353 }
7354 drop(cached);
7355 if let Some((bytes, used)) = taken {
7356 self.cache.held[column].fetch_add(1, Atomic::Relaxed);
7357 self.pool.admit(Held {
7358 shelf: Arc::downgrade(&self.cache),
7359 column,
7360 stripe: at,
7361 bytes,
7362 used,
7363 });
7364 }
7365 Ok(held)
7366 }
7367
7368 fn page_of(
7374 &self,
7375 stripe: &Stripe,
7376 column: usize,
7377 at: usize,
7378 whole: bool,
7379 known: Option<Arc<Vec<PartSpan>>>,
7380 ) -> Result<CachedColumn> {
7381 let index = match known {
7382 Some(index) => index,
7383 None => {
7384 self.indexes.fetch_add(1, Atomic::Relaxed);
7385 Arc::new(read_index(&self.file, stripe, column)?)
7386 }
7387 };
7388 let page = if whole {
7389 self.pages.fetch_add(1, Atomic::Relaxed);
7390 let span = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
7391 let mut bytes = vec![0; span.length as usize];
7392 read_at(&self.file, span.offset, &mut bytes)?;
7393 let checked = index.iter().map(|_| AtomicBool::new(false)).collect();
7394 Some(Arc::new(HeldPage { bytes, checked }))
7395 } else {
7396 None
7397 };
7398 Ok(CachedColumn { stripe: at, index, page })
7399 }
7400
7401 fn read_impl(
7402 &self,
7403 at: usize,
7404 columns: &[usize],
7405 whole: bool,
7406 positions: Option<&[u32]>,
7407 ) -> Result<Chunk> {
7408 let place = *self.places.get(at).ok_or_else(|| invalid("part index out of range"))?;
7409 let index = place.stripe as usize;
7410 let stripe =
7411 self.table.stripes.get(index).ok_or_else(|| invalid("stripe index out of range"))?;
7412 let rows = place.rows as usize;
7413 let mut picked = Vec::with_capacity(columns.len());
7414 for &column in columns {
7415 let field = self
7416 .table
7417 .fields
7418 .get(column)
7419 .ok_or_else(|| invalid("column index out of range"))?;
7420 let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
7421 let held = self.held(index, stripe, column, whole)?;
7422 let span = *held
7423 .index
7424 .get(place.part as usize)
7425 .ok_or_else(|| invalid("part index out of range"))?;
7426 let owned;
7427 let bytes = match &held.page {
7428 Some(held) => held.part(place.part as usize, span),
7429 None => {
7430 let offset = page
7431 .offset
7432 .checked_add(span.start as u64)
7433 .ok_or_else(|| invalid("part range overflow"))?;
7434 let mut bytes = vec![0; span.length];
7435 read_at(&self.file, offset, &mut bytes)?;
7436 owned = bytes;
7437 verify_part(&owned, span).map(|()| owned.as_slice())
7438 }
7439 }
7440 .map_err(|error| {
7441 invalid(&format!(
7442 "{}, column {column} part {} of the page at {}",
7443 error.message(),
7444 place.part,
7445 page.offset,
7446 ))
7447 })?;
7448 let dictionary = self.dictionary(column)?;
7449 let mut vector = match positions {
7455 None => decode(&field.ty, rows, bytes, dictionary)?,
7456 Some(positions) => decode_at(&field.ty, rows, bytes, dictionary, positions)?,
7457 };
7458 if self.demoted(column) && vector.stable_dictionary_parts().is_some() {
7462 vector = vector.flatten()?;
7463 }
7464 picked.push(vector.into_pages());
7465 }
7466 Chunk::with_rows(picked, positions.map_or(rows, <[u32]>::len))
7467 }
7468
7469 #[must_use]
7485 pub fn skips(&self, part: usize, probes: &[Probe]) -> bool {
7486 let Some(place) = self.places.get(part).copied() else { return false };
7487 let Some(stripe) = self.table.stripes.get(place.stripe as usize) else { return false };
7488 if stripe.zone.skips(probes) {
7489 return true;
7490 }
7491 probes.iter().any(|probe| self.outside(place, probe) || self.sifted(place, probe))
7492 }
7493
7494 #[must_use]
7501 pub fn ruled_by(&self, part: usize, column: usize, rule: impl Fn(&Range) -> bool) -> bool {
7502 let Some(place) = self.places.get(part).copied() else { return false };
7503 let Some(stripe) = self.table.stripes.get(place.stripe as usize) else { return false };
7504 if stripe.zone.column(column).is_some_and(&rule) {
7505 return true;
7506 }
7507 self.stripe_part_ranges(place.stripe as usize, column)
7508 .and_then(|ranges| ranges.get(place.part as usize))
7509 .is_some_and(rule)
7510 }
7511
7512 #[must_use]
7515 pub fn part_range(&self, part: usize, column: usize) -> Option<Range> {
7516 let place = self.places.get(part).copied()?;
7517 let own = self
7518 .stripe_part_ranges(place.stripe as usize, column)
7519 .and_then(|ranges| ranges.get(place.part as usize));
7520 own.or_else(|| self.table.stripes.get(place.stripe as usize)?.zone.column(column)).cloned()
7521 }
7522
7523 #[must_use]
7525 pub fn stripe_ruled_by(
7526 &self,
7527 stripe: usize,
7528 column: usize,
7529 rule: impl Fn(&Range) -> bool,
7530 ) -> bool {
7531 self.table.stripes.get(stripe).and_then(|held| held.zone.column(column)).is_some_and(rule)
7532 }
7533
7534 fn outside(&self, place: Place, probe: &Probe) -> bool {
7540 match self.stripe_part_ranges(place.stripe as usize, probe.column) {
7541 Some(ranges) => ranges
7542 .get(place.part as usize)
7543 .is_some_and(|range| range.excludes(probe.op, &probe.value)),
7544 None => false,
7545 }
7546 }
7547
7548 fn stripe_part_ranges(&self, stripe: usize, column: usize) -> Option<&[Range]> {
7554 let slot = self.part_ranges.get(column)?.get(stripe)?;
7555 if let Some(held) = slot.get() {
7556 return Some(held);
7557 }
7558 let page = self.table.stripes.get(stripe)?.part_ranges.get(column)?;
7559 let mut bytes = vec![0; page.length as usize];
7560 read_at(&self.file, page.offset, &mut bytes).ok()?;
7561 if checksum(&bytes) != page.hash {
7562 return None;
7563 }
7564 let ranges = Arc::new(decode_part_ranges(&bytes).ok()?);
7565 let _ = slot.set(ranges);
7566 slot.get().map(|held| held.as_slice())
7567 }
7568
7569 #[must_use]
7586 pub fn certain(&self, part: usize, probes: &[Probe]) -> bool {
7587 let Some(place) = self.places.get(part).copied() else { return false };
7588 let Some(stripe) = self.table.stripes.get(place.stripe as usize) else { return false };
7589 if stripe.zone.certain(probes) {
7590 return true;
7591 }
7592 probes
7593 .iter()
7594 .all(|probe| stripe.zone.certain(slice::from_ref(probe)) || self.inside(place, probe))
7595 }
7596
7597 fn inside(&self, place: Place, probe: &Probe) -> bool {
7603 match self.stripe_part_ranges(place.stripe as usize, probe.column) {
7604 Some(ranges) => ranges
7605 .get(place.part as usize)
7606 .is_some_and(|range| range.certain(probe.op, &probe.value)),
7607 None => false,
7608 }
7609 }
7610
7611 #[must_use]
7622 pub fn stripe_skips(&self, stripe: usize, probes: &[Probe]) -> bool {
7623 self.table.stripes.get(stripe).is_some_and(|held| held.zone.skips(probes))
7624 }
7625
7626 fn sifted(&self, place: Place, probe: &Probe) -> bool {
7632 if probe.op != Op::Equal {
7633 return false;
7634 }
7635 match self.stripe_sieves(place.stripe as usize, probe.column) {
7636 Some(sieves) => sieves
7637 .get(place.part as usize)
7638 .and_then(Option::as_ref)
7639 .is_some_and(|sieve| sieve.excludes(&probe.value)),
7640 None => false,
7641 }
7642 }
7643
7644 fn stripe_sieves(&self, stripe: usize, column: usize) -> Option<&[Option<Sieve>]> {
7651 let slot = self.sieves.get(column)?.get(stripe)?;
7652 if let Some(held) = slot.get() {
7653 return Some(held);
7654 }
7655 let page = self.table.stripes.get(stripe)?.sieves.get(column)?;
7656 let mut bytes = vec![0; page.length as usize];
7657 read_at(&self.file, page.offset, &mut bytes).ok()?;
7658 if checksum(&bytes) != page.hash {
7659 return None;
7660 }
7661 let sieves = Arc::new(decode_sieves(&bytes).ok()?);
7662 let _ = slot.set(sieves);
7663 slot.get().map(|held| held.as_slice())
7664 }
7665}
7666
7667fn text_at_rank(dictionary: &Vector, rank: usize) -> Result<Value> {
7669 let code = dictionary.code_at_rank(rank)? as usize;
7670 if dictionary.logical_type() == &LogicalType::Blob {
7671 let bytes = dictionary
7672 .try_bytes_at(code)?
7673 .ok_or_else(|| invalid("global dictionary order names a code it does not have"))?;
7674 return Ok(Value::Blob(bytes.to_vec()));
7675 }
7676 let text = dictionary
7677 .try_text_at(code)?
7678 .ok_or_else(|| invalid("global dictionary order names a code it does not have"))?;
7679 Ok(Value::Varchar(text.into()))
7680}
7681
7682fn read_at<F: Positional + ?Sized>(file: &F, offset: u64, bytes: &mut [u8]) -> Result<()> {
7692 file.fill_at(offset, bytes)
7693}
7694
7695trait Positional {
7703 fn fill_at(&self, offset: u64, bytes: &mut [u8]) -> Result<()>;
7708}
7709
7710impl<T: Positional + ?Sized> Positional for &T {
7711 fn fill_at(&self, offset: u64, bytes: &mut [u8]) -> Result<()> {
7712 (**self).fill_at(offset, bytes)
7713 }
7714}
7715
7716impl<T: Positional + ?Sized> Positional for Arc<T> {
7717 fn fill_at(&self, offset: u64, bytes: &mut [u8]) -> Result<()> {
7718 (**self).fill_at(offset, bytes)
7719 }
7720}
7721
7722impl<T: Positional + ?Sized> Positional for Box<T> {
7723 fn fill_at(&self, offset: u64, bytes: &mut [u8]) -> Result<()> {
7724 (**self).fill_at(offset, bytes)
7725 }
7726}
7727
7728impl Positional for dyn rudb_io::File + '_ {
7729 fn fill_at(&self, mut offset: u64, mut bytes: &mut [u8]) -> Result<()> {
7730 while !bytes.is_empty() {
7731 let read = self.read_at(offset, bytes)?;
7732 if read == 0 {
7733 return Err(invalid("column page ends before its declared length"));
7734 }
7735 offset += read as u64;
7736 bytes = &mut bytes[read..];
7737 }
7738 Ok(())
7739 }
7740}
7741
7742impl Positional for File {
7743 #[cfg(unix)]
7744 fn fill_at(&self, mut offset: u64, mut bytes: &mut [u8]) -> Result<()> {
7745 use std::os::unix::fs::FileExt;
7746 while !bytes.is_empty() {
7747 let read = self.read_at(bytes, offset).map_err(io)?;
7748 if read == 0 {
7749 return Err(invalid("column page ends before its declared length"));
7750 }
7751 offset += read as u64;
7752 bytes = &mut bytes[read..];
7753 }
7754 Ok(())
7755 }
7756
7757 #[cfg(windows)]
7763 fn fill_at(&self, mut offset: u64, mut bytes: &mut [u8]) -> Result<()> {
7764 use std::os::windows::fs::FileExt;
7765 while !bytes.is_empty() {
7766 let read = self.seek_read(bytes, offset).map_err(io)?;
7767 if read == 0 {
7768 return Err(invalid("column page ends before its declared length"));
7769 }
7770 offset += read as u64;
7771 bytes = &mut bytes[read..];
7772 }
7773 Ok(())
7774 }
7775
7776 #[cfg(not(any(unix, windows)))]
7781 fn fill_at(&self, offset: u64, bytes: &mut [u8]) -> Result<()> {
7782 use std::io::{Read, Seek, SeekFrom};
7783 let mut file = self.try_clone().map_err(io)?;
7784 file.seek(SeekFrom::Start(offset)).map_err(io)?;
7785 file.read_exact(bytes).map_err(io)
7786 }
7787}
7788
7789#[cfg(test)]
7794fn write_at(file: &File, offset: u64, bytes: &[u8]) -> Result<()> {
7795 use std::io::{Seek, SeekFrom, Write};
7796 let mut file = file;
7797 file.seek(SeekFrom::Start(offset)).map_err(io)?;
7798 file.write_all(bytes).map_err(io)
7799}
7800
7801fn type_tag(ty: &LogicalType) -> Result<u8> {
7808 match ty {
7809 LogicalType::SmallInt => Ok(1),
7810 LogicalType::Integer => Ok(2),
7811 LogicalType::BigInt => Ok(3),
7812 LogicalType::Varchar => Ok(4),
7813 LogicalType::Date => Ok(5),
7814 LogicalType::Timestamp => Ok(6),
7815 LogicalType::Boolean => Ok(7),
7816 LogicalType::TinyInt => Ok(8),
7817 LogicalType::UTinyInt => Ok(9),
7818 LogicalType::USmallInt => Ok(10),
7819 LogicalType::UInteger => Ok(11),
7820 LogicalType::UBigInt => Ok(12),
7821 LogicalType::Decimal { .. } => Ok(13),
7822 LogicalType::Float => Ok(14),
7823 LogicalType::Double => Ok(15),
7824 LogicalType::HugeInt => Ok(16),
7825 LogicalType::UHugeInt => Ok(17),
7826 LogicalType::Time => Ok(18),
7827 LogicalType::TimeTz => Ok(19),
7828 LogicalType::TimestampTz => Ok(20),
7829 LogicalType::Interval => Ok(21),
7830 LogicalType::Uuid => Ok(22),
7831 LogicalType::Blob => Ok(23),
7832 LogicalType::Bit => Ok(24),
7833 LogicalType::TimestampS => Ok(25),
7834 LogicalType::TimestampMs => Ok(26),
7835 LogicalType::TimestampNs => Ok(27),
7836 _ => Err(Error::not_implemented(format!("native storage for {ty}"))),
7837 }
7838}
7839
7840fn put_type(out: &mut Vec<u8>, ty: &LogicalType) -> Result<()> {
7846 out.push(type_tag(ty)?);
7847 if let LogicalType::Decimal { width, scale } = ty {
7848 out.push(*width);
7849 out.push(*scale);
7850 }
7851 Ok(())
7852}
7853
7854fn read_type(cur: &mut Cursor<'_>) -> Result<LogicalType> {
7856 let tag = cur.u8()?;
7857 if tag == 13 {
7858 let width = cur.u8()?;
7859 let scale = cur.u8()?;
7860 return LogicalType::decimal(width, scale)
7861 .map_err(|_| invalid("decimal column width and scale are not a decimal"));
7862 }
7863 tag_type(tag)
7864}
7865
7866fn tag_type(tag: u8) -> Result<LogicalType> {
7867 match tag {
7868 1 => Ok(LogicalType::SmallInt),
7869 2 => Ok(LogicalType::Integer),
7870 3 => Ok(LogicalType::BigInt),
7871 4 => Ok(LogicalType::Varchar),
7872 5 => Ok(LogicalType::Date),
7873 6 => Ok(LogicalType::Timestamp),
7874 7 => Ok(LogicalType::Boolean),
7875 8 => Ok(LogicalType::TinyInt),
7876 9 => Ok(LogicalType::UTinyInt),
7877 10 => Ok(LogicalType::USmallInt),
7878 11 => Ok(LogicalType::UInteger),
7879 12 => Ok(LogicalType::UBigInt),
7880 14 => Ok(LogicalType::Float),
7881 15 => Ok(LogicalType::Double),
7882 16 => Ok(LogicalType::HugeInt),
7883 17 => Ok(LogicalType::UHugeInt),
7884 18 => Ok(LogicalType::Time),
7885 19 => Ok(LogicalType::TimeTz),
7886 20 => Ok(LogicalType::TimestampTz),
7887 21 => Ok(LogicalType::Interval),
7888 22 => Ok(LogicalType::Uuid),
7889 23 => Ok(LogicalType::Blob),
7890 24 => Ok(LogicalType::Bit),
7891 25 => Ok(LogicalType::TimestampS),
7892 26 => Ok(LogicalType::TimestampMs),
7893 27 => Ok(LogicalType::TimestampNs),
7894 _ => Err(invalid("column type tag is unknown")),
7895 }
7896}
7897
7898fn put_u16(out: &mut Vec<u8>, value: u16) {
7899 out.extend_from_slice(&value.to_le_bytes());
7900}
7901fn put_u32(out: &mut Vec<u8>, value: u32) {
7902 out.extend_from_slice(&value.to_le_bytes());
7903}
7904fn put_u64(out: &mut Vec<u8>, value: u64) {
7905 out.extend_from_slice(&value.to_le_bytes());
7906}
7907fn put_var_u64(out: &mut Vec<u8>, mut value: u64) {
7908 while value >= 0x80 {
7909 out.push((value as u8 & 0x7f) | 0x80);
7910 value >>= 7;
7911 }
7912 out.push(value as u8);
7913}
7914
7915fn frequency_order(left: FrequencyValue, right: FrequencyValue) -> Ordering {
7916 match (left, right) {
7917 (FrequencyValue::Null, FrequencyValue::Null) => Ordering::Equal,
7918 (FrequencyValue::Null, _) => Ordering::Less,
7919 (_, FrequencyValue::Null) => Ordering::Greater,
7920 (FrequencyValue::Integer(left), FrequencyValue::Integer(right)) => left.cmp(&right),
7921 (FrequencyValue::Code(left), FrequencyValue::Code(right)) => left.cmp(&right),
7922 (FrequencyValue::Integer(_), FrequencyValue::Code(_)) => Ordering::Less,
7923 (FrequencyValue::Code(_), FrequencyValue::Integer(_)) => Ordering::Greater,
7924 }
7925}
7926
7927fn keep_most_frequent(entries: &mut Vec<FrequencyEntry>) -> u64 {
7940 let order = |left: &FrequencyEntry, right: &FrequencyEntry| {
7941 right.count.cmp(&left.count).then_with(|| frequency_order(left.value, right.value))
7942 };
7943 let omitted_max = if entries.len() > FREQUENCY_ENTRIES {
7944 let (_, next, _) = entries.select_nth_unstable_by(FREQUENCY_ENTRIES, order);
7945 let omitted_max = next.count;
7946 entries.truncate(FREQUENCY_ENTRIES);
7947 omitted_max
7948 } else {
7949 0
7950 };
7951 entries.sort_unstable_by(order);
7952 omitted_max
7953}
7954
7955fn code_frequency(
7956 dictionary: &GlobalDictionary,
7957 flat: &[u8],
7958 bases: &[u64],
7959) -> Result<(FrequencySummary, Vec<Option<Vec<u8>>>)> {
7960 let mut entries = dictionary
7961 .counts
7962 .iter()
7963 .enumerate()
7964 .filter(|(_, count)| **count != 0)
7965 .map(|(code, &count)| FrequencyEntry { value: FrequencyValue::Code(code as u32), count })
7966 .collect::<Vec<_>>();
7967 if dictionary.nulls != 0 {
7968 entries.push(FrequencyEntry { value: FrequencyValue::Null, count: dictionary.nulls });
7969 }
7970 let omitted_max = keep_most_frequent(&mut entries);
7971 let mut spans = Vec::with_capacity(entries.len());
7972 let mut text_bytes = 0_usize;
7973 for entry in &entries {
7974 let span = match entry.value {
7975 FrequencyValue::Code(code) => {
7976 let span = GlobalDictionary::value_span(&dictionary.ends, bases, code as usize);
7977 let bytes = flat
7978 .get(span.0..span.1)
7979 .ok_or_else(|| invalid("a frequency code is outside its dictionary"))?;
7980 text_bytes = text_bytes.saturating_add(bytes.len());
7981 Some(span)
7982 }
7983 FrequencyValue::Null | FrequencyValue::Integer(_) => None,
7984 };
7985 spans.push(span);
7986 }
7987 let texts = if text_bytes > FREQUENCY_TEXT_BUDGET {
7988 Vec::new()
7989 } else {
7990 spans.into_iter().map(|span| span.map(|(from, to)| flat[from..to].to_vec())).collect()
7991 };
7992 Ok((
7993 FrequencySummary {
7994 entries,
7995 omitted_max,
7996 ordinals: Vec::new(),
7997 ordinal_entries: Vec::new(),
7998 },
7999 texts,
8000 ))
8001}
8002
8003fn encode_directory(table: &Table) -> Result<Vec<u8>> {
8004 let mut out = DIRECTORY.to_vec();
8005 let name = table.name.as_bytes();
8006 put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("table name too long"))?);
8007 out.extend_from_slice(name);
8008 put_u16(&mut out, u16::try_from(table.fields.len()).map_err(|_| invalid("too many columns"))?);
8009 for field in &table.fields {
8010 let name = field.name.as_bytes();
8011 put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("column name too long"))?);
8012 out.extend_from_slice(name);
8013 put_type(&mut out, &field.ty)?;
8014 out.push(u8::from(field.not_null));
8015 }
8016 for (field, dictionary) in table.fields.iter().zip(&table.dictionaries) {
8017 match dictionary {
8018 None => out.push(0),
8019 Some(page) => {
8020 out.push(dictionary_tag(&field.ty));
8021 put_u64(&mut out, page.offset);
8022 put_u32(&mut out, page.length);
8023 put_u64(&mut out, page.hash);
8024 }
8025 }
8026 }
8027 for distinct in &table.distincts {
8028 match distinct {
8029 None => out.push(0),
8030 Some(count) => {
8031 out.push(1);
8032 put_u64(&mut out, *count);
8033 }
8034 }
8035 }
8036 put_u64(&mut out, u64::try_from(table.rows).map_err(|_| invalid("row count overflow"))?);
8037 put_u32(&mut out, u32::try_from(table.stripes.len()).map_err(|_| invalid("too many stripes"))?);
8038 for stripe in &table.stripes {
8039 put_u32(
8040 &mut out,
8041 u32::try_from(stripe.parts.len()).map_err(|_| invalid("too many parts in a stripe"))?,
8042 );
8043 for &rows in &stripe.parts {
8044 put_u32(&mut out, rows);
8045 }
8046 put_u64(&mut out, stripe.index.offset);
8047 put_u32(&mut out, stripe.index.length);
8048 for page in &stripe.pages {
8049 put_u64(&mut out, page.offset);
8050 put_u32(&mut out, page.length);
8051 }
8052 for (column, ((field, dictionary), membership)) in
8057 table.fields.iter().zip(&table.dictionaries).zip(stripe.memberships.slots()).enumerate()
8058 {
8059 if !coded_type(&field.ty) || dictionary.is_none() {
8060 continue;
8061 }
8062 let page = match membership {
8063 Some(page) => page,
8064 None if table.demoted.get(column).copied().unwrap_or(false) => {
8065 Page { offset: HEADER, length: 0, hash: 0 }
8066 }
8067 None => return Err(invalid("string page has no code membership index")),
8068 };
8069 put_u64(&mut out, page.offset);
8070 put_u32(&mut out, page.length);
8071 put_u64(&mut out, page.hash);
8072 }
8073 for sieve in stripe.sieves.slots() {
8074 match sieve {
8075 None => out.push(0),
8076 Some(page) => {
8077 out.push(1);
8078 put_u64(&mut out, page.offset);
8079 put_u32(&mut out, page.length);
8080 put_u64(&mut out, page.hash);
8081 }
8082 }
8083 }
8084 for held in stripe.part_ranges.slots() {
8085 match held {
8086 None => out.push(0),
8087 Some(page) => {
8088 out.push(1);
8089 put_u64(&mut out, page.offset);
8090 put_u32(&mut out, page.length);
8091 put_u64(&mut out, page.hash);
8092 }
8093 }
8094 }
8095 for range in stripe.zone.columns() {
8096 put_bound(&mut out, range.low.as_ref())?;
8097 put_bound(&mut out, range.high.as_ref())?;
8098 put_u32(
8099 &mut out,
8100 u32::try_from(range.nulls).map_err(|_| invalid("null count overflow"))?,
8101 );
8102 out.push(u8::from(range.exact));
8103 match range.sum {
8104 None => out.push(0),
8105 Some(total) => {
8106 out.push(1);
8107 out.extend_from_slice(&total.to_le_bytes());
8108 }
8109 }
8110 }
8111 }
8112 out.extend_from_slice(FREQUENCIES_SPANS);
8113 put_u16(
8114 &mut out,
8115 u16::try_from(table.frequencies.len())
8116 .map_err(|_| invalid("too many frequency columns"))?,
8117 );
8118 for summary in &table.frequencies {
8119 let summary = match summary {
8120 None => {
8121 put_u32(&mut out, 0);
8122 put_u32(&mut out, 0);
8123 continue;
8124 }
8125 Some(Frequencies::Held(summary)) => summary,
8126 Some(Frequencies::Stored { .. }) => {
8128 return Err(invalid("a synopsis left in the file cannot be written back"));
8129 }
8130 };
8131 let length_at = out.len();
8132 put_u32(&mut out, 0);
8133 put_u32(
8134 &mut out,
8135 u32::try_from(summary.entries.len())
8136 .map_err(|_| invalid("too many frequency entries"))?,
8137 );
8138 let start = out.len();
8139 out.push(1);
8140 put_u64(&mut out, summary.omitted_max);
8141 put_u32(
8142 &mut out,
8143 u32::try_from(summary.entries.len())
8144 .map_err(|_| invalid("too many frequency entries"))?,
8145 );
8146 for entry in &summary.entries {
8147 match entry.value {
8148 FrequencyValue::Null => out.push(0),
8149 FrequencyValue::Integer(value) => {
8150 out.push(1);
8151 out.extend_from_slice(&value.to_le_bytes());
8152 }
8153 FrequencyValue::Code(value) => {
8154 out.push(2);
8155 put_u32(&mut out, value);
8156 }
8157 }
8158 put_u64(&mut out, entry.count);
8159 }
8160 put_u32(
8161 &mut out,
8162 u32::try_from(summary.ordinals.len())
8163 .map_err(|_| invalid("too many frequency ordinals"))?,
8164 );
8165 let mut previous = 0_u64;
8166 for (at, &ordinal) in summary.ordinals.iter().enumerate() {
8167 let delta = if at == 0 {
8168 ordinal
8169 } else {
8170 ordinal
8171 .checked_sub(previous)
8172 .ok_or_else(|| invalid("frequency ordinals are not ordered"))?
8173 };
8174 if at != 0 && delta == 0 {
8175 return Err(invalid("frequency ordinals are not unique"));
8176 }
8177 put_var_u64(&mut out, delta);
8178 previous = ordinal;
8179 }
8180 if summary.ordinal_entries.len() != summary.ordinals.len() {
8181 return Err(invalid("frequency ordinal values have a different length"));
8182 }
8183 for &entry in &summary.ordinal_entries {
8184 if entry as usize >= summary.entries.len() {
8185 return Err(invalid("frequency ordinal value is outside its entries"));
8186 }
8187 put_u16(&mut out, entry);
8188 }
8189 let length = u32::try_from(out.len() - start)
8190 .map_err(|_| invalid("a frequency synopsis is too long"))?;
8191 out[length_at..length_at + 4].copy_from_slice(&length.to_le_bytes());
8192 }
8193 if !table.pair_frequencies.is_empty() {
8194 out.extend_from_slice(PAIR_FREQUENCIES);
8195 put_u16(
8196 &mut out,
8197 u16::try_from(table.pair_frequencies.len())
8198 .map_err(|_| invalid("too many pair frequency summaries"))?,
8199 );
8200 for summary in &table.pair_frequencies {
8201 put_u16(&mut out, summary.first);
8202 put_u16(&mut out, summary.second);
8203 put_u64(&mut out, summary.omitted_max);
8204 put_u16(
8205 &mut out,
8206 u16::try_from(summary.entries.len())
8207 .map_err(|_| invalid("too many pair frequency entries"))?,
8208 );
8209 for entry in &summary.entries {
8210 put_u16(&mut out, entry.first_entry);
8211 match entry.second {
8212 None => out.push(0),
8213 Some(code) => {
8214 out.push(1);
8215 put_u32(&mut out, code);
8216 }
8217 }
8218 put_u64(&mut out, entry.count);
8219 }
8220 }
8221 }
8222 let text_columns = table.frequency_texts.iter().filter(|texts| !texts.is_empty()).count();
8223 if text_columns != 0 {
8224 out.extend_from_slice(FREQUENCY_TEXTS);
8225 put_u16(
8226 &mut out,
8227 u16::try_from(text_columns)
8228 .map_err(|_| invalid("too many string frequency columns"))?,
8229 );
8230 for (column, texts) in table.frequency_texts.iter().enumerate() {
8231 if texts.is_empty() {
8232 continue;
8233 }
8234 put_u16(
8235 &mut out,
8236 u16::try_from(column).map_err(|_| invalid("frequency text column overflows"))?,
8237 );
8238 put_u16(
8239 &mut out,
8240 u16::try_from(texts.len())
8241 .map_err(|_| invalid("too many frequency text entries"))?,
8242 );
8243 for text in texts {
8244 match text {
8245 None => out.push(0),
8246 Some(text) => {
8247 out.push(1);
8248 put_u32(
8249 &mut out,
8250 u32::try_from(text.len())
8251 .map_err(|_| invalid("frequency text is too long"))?,
8252 );
8253 out.extend_from_slice(text);
8254 }
8255 }
8256 }
8257 }
8258 }
8259 if let Some(summary) = &table.host_groups {
8260 out.extend_from_slice(HOST_GROUPS);
8261 put_u16(
8262 &mut out,
8263 u16::try_from(summary.column).map_err(|_| invalid("host column overflows"))?,
8264 );
8265 put_u64(&mut out, summary.omitted_max);
8266 put_u16(
8267 &mut out,
8268 u16::try_from(summary.entries.len()).map_err(|_| invalid("too many host groups"))?,
8269 );
8270 for entry in &summary.entries {
8271 put_u32(
8272 &mut out,
8273 u32::try_from(entry.host.len()).map_err(|_| invalid("host name is too long"))?,
8274 );
8275 out.extend_from_slice(entry.host.as_bytes());
8276 put_u64(&mut out, entry.count);
8277 out.extend_from_slice(&entry.bytes_sum.to_le_bytes());
8278 put_u32(
8279 &mut out,
8280 u32::try_from(entry.minimum.len())
8281 .map_err(|_| invalid("host minimum is too long"))?,
8282 );
8283 out.extend_from_slice(entry.minimum.as_bytes());
8284 }
8285 }
8286 if let Some(clustering) = &table.clustering {
8289 out.extend_from_slice(CLUSTERING);
8290 out.push(clustering.width().tag());
8291 put_u16(
8292 &mut out,
8293 u16::try_from(clustering.columns().len())
8294 .map_err(|_| invalid("too many clustering columns"))?,
8295 );
8296 for &column in clustering.columns() {
8297 put_u16(
8298 &mut out,
8299 u16::try_from(column).map_err(|_| invalid("clustering column index overflow"))?,
8300 );
8301 }
8302 }
8303 let demoted = (0..table.fields.len())
8304 .filter(|&column| table.demoted.get(column).copied().unwrap_or(false))
8305 .collect::<Vec<_>>();
8306 if !demoted.is_empty() {
8307 out.extend_from_slice(DEMOTED);
8308 put_u16(
8309 &mut out,
8310 u16::try_from(demoted.len()).map_err(|_| invalid("too many demoted columns"))?,
8311 );
8312 for column in demoted {
8313 put_u16(
8314 &mut out,
8315 u16::try_from(column).map_err(|_| invalid("demoted column index overflow"))?,
8316 );
8317 }
8318 }
8319 if !table.constraints.is_empty() {
8320 out.extend_from_slice(KEYS);
8321 put_count(&mut out, table.constraints.keys.len())?;
8322 for (columns, primary) in &table.constraints.keys {
8323 out.push(u8::from(*primary));
8324 put_columns(&mut out, columns)?;
8325 }
8326 put_count(&mut out, table.constraints.foreign.len())?;
8327 for foreign in &table.constraints.foreign {
8328 put_columns(&mut out, &foreign.columns)?;
8329 put_columns(&mut out, &foreign.referenced)?;
8330 put_u32(
8331 &mut out,
8332 u32::try_from(foreign.table.len())
8333 .map_err(|_| invalid("table name is too long"))?,
8334 );
8335 out.extend_from_slice(foreign.table.as_bytes());
8336 }
8337 }
8338 out.extend_from_slice(SECTIONS);
8344 put_u64(&mut out, table.generation);
8345 put_u16(
8346 &mut out,
8347 u16::try_from(table.sections.len()).map_err(|_| invalid("too many sections"))?,
8348 );
8349 for held in &table.sections {
8350 held.encode(&mut out)?;
8351 }
8352 if table.dictionary_payloads.iter().any(|&bytes| bytes != 0) {
8353 out.extend_from_slice(DICTIONARY_PAYLOADS);
8354 put_u16(
8355 &mut out,
8356 u16::try_from(table.fields.len()).map_err(|_| invalid("too many columns"))?,
8357 );
8358 for at in 0..table.fields.len() {
8359 put_u64(&mut out, table.dictionary_payloads.get(at).copied().unwrap_or(0));
8360 }
8361 }
8362 Ok(out)
8363}
8364
8365fn signed_integer(ty: &LogicalType) -> bool {
8374 matches!(
8375 ty,
8376 LogicalType::TinyInt | LogicalType::SmallInt | LogicalType::Integer | LogicalType::BigInt
8377 )
8378}
8379
8380fn integer_or_date(ty: &LogicalType) -> bool {
8381 matches!(
8382 ty,
8383 LogicalType::TinyInt
8384 | LogicalType::SmallInt
8385 | LogicalType::Integer
8386 | LogicalType::BigInt
8387 | LogicalType::UTinyInt
8388 | LogicalType::USmallInt
8389 | LogicalType::UInteger
8390 | LogicalType::UBigInt
8391 | LogicalType::Date
8392 )
8393}
8394
8395fn table_integer_extremes(table: &Table) -> Vec<StoredIntegerExtremes> {
8396 table
8397 .fields
8398 .iter()
8399 .enumerate()
8400 .map(|(column, field)| {
8401 if !integer_or_date(&field.ty) {
8402 return None;
8403 }
8404 let mut low: Option<i128> = None;
8405 let mut high: Option<i128> = None;
8406 for stripe in &table.stripes {
8407 let range = stripe.zone.column(column)?;
8408 if !range.exact {
8409 return None;
8410 }
8411 match (range.low.as_ref(), range.high.as_ref()) {
8412 (Some(Bound::Int(small)), Some(Bound::Int(large))) => {
8413 low = Some(low.map_or(*small, |held| held.min(*small)));
8414 high = Some(high.map_or(*large, |held| held.max(*large)));
8415 }
8416 (None, None) if stripe.rows == range.nulls => {}
8417 _ => return None,
8418 }
8419 }
8420 Some(low.zip(high))
8421 })
8422 .collect()
8423}
8424
8425fn reader_integer_extremes(reader: &Reader) -> Result<Vec<StoredIntegerExtremes>> {
8426 reader
8427 .table
8428 .fields
8429 .iter()
8430 .enumerate()
8431 .map(|(column, field)| {
8432 if !integer_or_date(&field.ty) {
8433 return Ok(None);
8434 }
8435 match reader.exact_extremes(column)? {
8436 Some((Bound::Int(low), Bound::Int(high))) => Ok(Some(Some((low, high)))),
8437 None if reader.null_count(column)? == reader.table.rows as u64 => Ok(Some(None)),
8438 _ => Ok(None),
8439 }
8440 })
8441 .collect()
8442}
8443
8444fn table_complete_numeric_frequencies(table: &Table) -> Vec<StoredNumericFrequencies> {
8445 table
8446 .fields
8447 .iter()
8448 .enumerate()
8449 .map(|(column, field)| {
8450 if !integer_or_date(&field.ty) {
8451 return None;
8452 }
8453 let Some(Frequencies::Held(summary)) = table.frequencies.get(column)?.as_ref() else {
8454 return None;
8455 };
8456 if summary.omitted_max != 0 || summary.entries.len() > MAX_CATALOG_FREQUENCIES {
8457 return None;
8458 }
8459 let entries = summary
8460 .entries
8461 .iter()
8462 .map(|entry| {
8463 let value = match entry.value {
8464 FrequencyValue::Null => None,
8465 FrequencyValue::Integer(value) => Some(value),
8466 FrequencyValue::Code(_) => return None,
8467 };
8468 Some((value, entry.count))
8469 })
8470 .collect::<Option<Vec<_>>>()?;
8471 let rows = entries.iter().try_fold(0_u64, |sum, (_, count)| sum.checked_add(*count))?;
8472 (rows == table.rows as u64).then_some(entries)
8473 })
8474 .collect()
8475}
8476
8477fn integer_value(bits: u64, signed: bool) -> FrequencyValue {
8484 if signed {
8485 FrequencyValue::Integer(i128::from(bits as i64))
8486 } else {
8487 FrequencyValue::Integer(i128::from(bits))
8488 }
8489}
8490
8491fn frequency_bits(value: &Value) -> Option<u64> {
8492 Some(match value {
8493 Value::TinyInt(value) => i64::from(*value) as u64,
8494 Value::SmallInt(value) => i64::from(*value) as u64,
8495 Value::Integer(value) | Value::Date(value) => i64::from(*value) as u64,
8496 Value::BigInt(value) | Value::Timestamp(value) => *value as u64,
8497 Value::UTinyInt(value) => u64::from(*value),
8498 Value::USmallInt(value) => u64::from(*value),
8499 Value::UInteger(value) => u64::from(*value),
8500 Value::UBigInt(value) => *value,
8501 _ => return None,
8502 })
8503}
8504
8505fn numeric_frequency_value(value: &Value) -> Option<Option<i128>> {
8506 Some(match value {
8507 Value::Null => None,
8508 Value::TinyInt(value) => Some(i128::from(*value)),
8509 Value::SmallInt(value) => Some(i128::from(*value)),
8510 Value::Integer(value) | Value::Date(value) => Some(i128::from(*value)),
8511 Value::BigInt(value) => Some(i128::from(*value)),
8512 Value::UTinyInt(value) => Some(i128::from(*value)),
8513 Value::USmallInt(value) => Some(i128::from(*value)),
8514 Value::UInteger(value) => Some(i128::from(*value)),
8515 Value::UBigInt(value) => Some(i128::from(*value)),
8516 _ => return None,
8517 })
8518}
8519
8520fn reader_complete_numeric_frequencies(reader: &Reader) -> Result<Vec<StoredNumericFrequencies>> {
8521 reader
8522 .table
8523 .fields
8524 .iter()
8525 .enumerate()
8526 .map(|(column, field)| {
8527 if !integer_or_date(&field.ty) {
8528 return Ok(None);
8529 }
8530 let Some(summary) = reader.frequency_summary(column)? else { return Ok(None) };
8531 if summary.omitted_max != 0 || summary.entries.len() > MAX_CATALOG_FREQUENCIES {
8532 return Ok(None);
8533 }
8534 let entries = reader.decode_frequencies(column, &field.ty, &summary.entries)?;
8535 let Some(entries) = entries
8536 .iter()
8537 .map(|(value, count)| Some((numeric_frequency_value(value)?, *count)))
8538 .collect::<Option<Vec<_>>>()
8539 else {
8540 return Ok(None);
8541 };
8542 let rows = entries.iter().try_fold(0_u64, |sum, (_, count)| sum.checked_add(*count));
8543 Ok((rows == Some(reader.table.rows as u64)).then_some(entries))
8544 })
8545 .collect()
8546}
8547
8548fn table_exact_sum(table: &Table, column: usize) -> Option<(i128, u64)> {
8549 table.stripes.iter().try_fold((0_i128, 0_u64), |(sum, count), stripe| {
8550 let range = stripe.zone.column(column)?;
8551 let sum = sum.checked_add(range.sum?)?;
8552 let nonnull = (stripe.rows as u64).checked_sub(range.nulls as u64)?;
8553 Some((sum, count.checked_add(nonnull)?))
8554 })
8555}
8556
8557fn table_aggregate_sums(table: &Table) -> Vec<Option<(i128, u64)>> {
8558 table
8559 .fields
8560 .iter()
8561 .enumerate()
8562 .map(|(column, field)| {
8563 signed_integer(&field.ty).then(|| table_exact_sum(table, column)).flatten()
8564 })
8565 .collect()
8566}
8567
8568fn reader_aggregate_sums(reader: &Reader) -> Result<Vec<Option<(i128, u64)>>> {
8569 reader
8570 .table
8571 .fields
8572 .iter()
8573 .enumerate()
8574 .map(
8575 |(column, field)| {
8576 if signed_integer(&field.ty) { reader.exact_sum(column) } else { Ok(None) }
8577 },
8578 )
8579 .collect()
8580}
8581
8582fn encode_catalog(entries: &[Entry], views: &[ViewEntry]) -> Result<Vec<u8>> {
8583 let mut out = CATALOG.to_vec();
8584 put_u32(&mut out, u32::try_from(entries.len()).map_err(|_| invalid("too many tables"))?);
8585 for entry in entries {
8586 let name = entry.name.as_bytes();
8587 put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("table name too long"))?);
8588 out.extend_from_slice(name);
8589 put_u64(&mut out, u64::try_from(entry.rows).map_err(|_| invalid("row count overflow"))?);
8590 put_u16(
8591 &mut out,
8592 u16::try_from(entry.fields.len()).map_err(|_| invalid("too many columns"))?,
8593 );
8594 for field in &entry.fields {
8595 let name = field.name.as_bytes();
8596 put_u16(
8597 &mut out,
8598 u16::try_from(name.len()).map_err(|_| invalid("column name too long"))?,
8599 );
8600 out.extend_from_slice(name);
8601 put_type(&mut out, &field.ty)?;
8602 out.push(u8::from(field.not_null));
8603 }
8604 put_u64(&mut out, entry.directory.offset);
8605 put_u32(&mut out, entry.directory.length);
8606 put_u64(&mut out, entry.directory.hash);
8607 }
8608 put_u32(&mut out, u32::try_from(views.len()).map_err(|_| invalid("too many views"))?);
8609 for view in views {
8610 let name = view.name.as_bytes();
8611 put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("view name too long"))?);
8612 out.extend_from_slice(name);
8613 put_long_text(&mut out, &view.sql, "view body")?;
8614 put_long_text(&mut out, &view.statement, "view statement")?;
8615 put_u16(
8616 &mut out,
8617 u16::try_from(view.aliases.len()).map_err(|_| invalid("too many aliases"))?,
8618 );
8619 for alias in &view.aliases {
8620 let alias = alias.as_bytes();
8621 put_u16(
8622 &mut out,
8623 u16::try_from(alias.len()).map_err(|_| invalid("alias name too long"))?,
8624 );
8625 out.extend_from_slice(alias);
8626 }
8627 put_u16(
8628 &mut out,
8629 u16::try_from(view.columns.len()).map_err(|_| invalid("too many columns"))?,
8630 );
8631 for field in &view.columns {
8632 let name = field.name.as_bytes();
8633 put_u16(
8634 &mut out,
8635 u16::try_from(name.len()).map_err(|_| invalid("column name too long"))?,
8636 );
8637 out.extend_from_slice(name);
8638 put_type(&mut out, &field.ty)?;
8639 out.push(u8::from(field.not_null));
8640 }
8641 }
8642 out.extend_from_slice(NONZERO_COUNTS);
8643 for entry in entries {
8644 if entry.nonzero.len() != entry.fields.len() {
8645 return Err(invalid("nonzero count width differs from schema"));
8646 }
8647 for count in &entry.nonzero {
8648 match count {
8649 None => out.push(0),
8650 Some(count) => {
8651 out.push(1);
8652 put_u64(&mut out, *count);
8653 }
8654 }
8655 }
8656 }
8657 out.extend_from_slice(AGGREGATE_SUMS);
8658 for entry in entries {
8659 if entry.aggregates.len() != entry.fields.len() {
8660 return Err(invalid("aggregate sum width differs from schema"));
8661 }
8662 for summary in &entry.aggregates {
8663 match summary {
8664 None => out.push(0),
8665 Some((sum, count)) => {
8666 out.push(1);
8667 out.extend_from_slice(&sum.to_le_bytes());
8668 put_u64(&mut out, *count);
8669 }
8670 }
8671 }
8672 }
8673 out.extend_from_slice(DISTINCT_COUNTS);
8674 for entry in entries {
8675 if entry.distincts.len() != entry.fields.len() {
8676 return Err(invalid("distinct count width differs from schema"));
8677 }
8678 for count in &entry.distincts {
8679 match count {
8680 None => out.push(0),
8681 Some(count) => {
8682 if *count > entry.rows as u64 {
8683 return Err(invalid("distinct count exceeds table rows"));
8684 }
8685 out.push(1);
8686 put_u64(&mut out, *count);
8687 }
8688 }
8689 }
8690 }
8691 out.extend_from_slice(INTEGER_EXTREMES);
8692 for entry in entries {
8693 if entry.extremes.len() != entry.fields.len() {
8694 return Err(invalid("integer extremes width differs from schema"));
8695 }
8696 for (field, extremes) in entry.fields.iter().zip(&entry.extremes) {
8697 match extremes {
8698 None => out.push(0),
8699 Some(None) if integer_or_date(&field.ty) => out.push(1),
8700 Some(Some((low, high))) if integer_or_date(&field.ty) && low <= high => {
8701 out.push(2);
8702 out.extend_from_slice(&low.to_le_bytes());
8703 out.extend_from_slice(&high.to_le_bytes());
8704 }
8705 _ => return Err(invalid("integer extremes type or range differs")),
8706 }
8707 }
8708 }
8709 out.extend_from_slice(COMPLETE_FREQUENCIES);
8710 for entry in entries {
8711 if entry.frequencies.len() != entry.fields.len() {
8712 return Err(invalid("numeric frequency width differs from schema"));
8713 }
8714 for (field, frequencies) in entry.fields.iter().zip(&entry.frequencies) {
8715 match frequencies {
8716 None => out.push(0),
8717 Some(entries)
8718 if integer_or_date(&field.ty) && entries.len() <= MAX_CATALOG_FREQUENCIES =>
8719 {
8720 let mut total = 0_u64;
8721 for (at, (value, count)) in entries.iter().enumerate() {
8722 if entries[..at].iter().any(|(held, _)| held == value) {
8723 return Err(invalid("numeric frequency value repeats"));
8724 }
8725 total = total
8726 .checked_add(*count)
8727 .ok_or_else(|| invalid("numeric frequency count overflows"))?;
8728 }
8729 if total != entry.rows as u64 {
8730 return Err(invalid("numeric frequencies do not cover table rows"));
8731 }
8732 out.push(1);
8733 out.push(entries.len() as u8);
8734 for (value, count) in entries {
8735 match value {
8736 None => out.push(0),
8737 Some(value) => {
8738 out.push(1);
8739 out.extend_from_slice(&value.to_le_bytes());
8740 }
8741 }
8742 put_u64(&mut out, *count);
8743 }
8744 }
8745 _ => return Err(invalid("numeric frequency type or width differs")),
8746 }
8747 }
8748 }
8749 Ok(out)
8750}
8751
8752fn put_long_text(out: &mut Vec<u8>, text: &str, what: &str) -> Result<()> {
8754 let bytes = text.as_bytes();
8755 put_u32(out, u32::try_from(bytes.len()).map_err(|_| invalid(&format!("{what} too long")))?);
8756 out.extend_from_slice(bytes);
8757 Ok(())
8758}
8759
8760fn decode_catalog(bytes: &[u8], size: u64) -> Result<(Vec<Entry>, Vec<ViewEntry>)> {
8763 let mut cur = Cursor::new(bytes);
8764 if cur.take(8)? != CATALOG {
8765 return Err(invalid("catalog magic differs"));
8766 }
8767 let count = cur.u32()? as usize;
8768 let mut entries: Vec<Entry> = Vec::with_capacity(count.min(1024));
8769 for _ in 0..count {
8770 let name = cur.text()?;
8771 let rows = usize::try_from(cur.u64()?).map_err(|_| invalid("row count does not fit"))?;
8772 let width = cur.u16()? as usize;
8773 let mut fields = Vec::with_capacity(width);
8774 for _ in 0..width {
8775 let name = cur.text()?;
8776 let ty = read_type(&mut cur)?;
8777 let not_null = match cur.u8()? {
8778 0 => false,
8779 1 => true,
8780 _ => return Err(invalid("nullability flag differs")),
8781 };
8782 fields.push(Field { name, ty, not_null });
8783 }
8784 let directory = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
8785 let end = directory
8786 .offset
8787 .checked_add(u64::from(directory.length))
8788 .ok_or_else(|| invalid("table directory offset overflow"))?;
8789 if directory.offset < HEADER
8790 || end > size
8791 || directory.length as usize > MAX_DIRECTORY
8792 || directory.length == 0
8793 {
8794 return Err(invalid("table directory range is outside the file"));
8795 }
8796 if entries.iter().any(|held| held.name == name) {
8797 return Err(invalid("two tables in the catalog have the same name"));
8798 }
8799 let nonzero = vec![None; fields.len()];
8800 let aggregates = vec![None; fields.len()];
8801 let distincts = vec![None; fields.len()];
8802 let extremes = vec![None; fields.len()];
8803 let frequencies = vec![None; fields.len()];
8804 entries.push(Entry {
8805 name,
8806 fields,
8807 rows,
8808 directory,
8809 nonzero,
8810 aggregates,
8811 distincts,
8812 extremes,
8813 frequencies,
8814 });
8815 }
8816 let count = if cur.done() { 0 } else { cur.u32()? as usize };
8821 let mut views: Vec<ViewEntry> = Vec::with_capacity(count.min(1024));
8822 for _ in 0..count {
8823 let name = cur.text()?;
8824 let sql = cur.long_text()?;
8825 let statement = cur.long_text()?;
8826 let width = cur.u16()? as usize;
8827 let mut aliases = Vec::with_capacity(width);
8828 for _ in 0..width {
8829 aliases.push(cur.text()?);
8830 }
8831 let width = cur.u16()? as usize;
8832 let mut columns = Vec::with_capacity(width);
8833 for _ in 0..width {
8834 let name = cur.text()?;
8835 let ty = read_type(&mut cur)?;
8836 let not_null = match cur.u8()? {
8837 0 => false,
8838 1 => true,
8839 _ => return Err(invalid("nullability flag differs")),
8840 };
8841 columns.push(Field { name, ty, not_null });
8842 }
8843 if views.iter().any(|held| held.name == name) {
8847 return Err(invalid("two views in the catalog have the same name"));
8848 }
8849 if entries.iter().any(|held| held.name == name) {
8850 return Err(invalid("a table and a view in the catalog have the same name"));
8851 }
8852 views.push(ViewEntry { name, sql, statement, aliases, columns });
8853 }
8854 if !cur.done() {
8855 if cur.take(8)? != NONZERO_COUNTS {
8856 return Err(invalid("catalog extension magic differs"));
8857 }
8858 for entry in &mut entries {
8859 for (field, count) in entry.fields.iter().zip(&mut entry.nonzero) {
8860 *count = match cur.u8()? {
8861 0 => None,
8862 1 if matches!(
8863 field.ty,
8864 LogicalType::TinyInt
8865 | LogicalType::SmallInt
8866 | LogicalType::Integer
8867 | LogicalType::BigInt
8868 | LogicalType::UTinyInt
8869 | LogicalType::USmallInt
8870 | LogicalType::UInteger
8871 | LogicalType::UBigInt
8872 ) =>
8873 {
8874 let value = cur.u64()?;
8875 if value > entry.rows as u64 {
8876 return Err(invalid("nonzero count exceeds rows"));
8877 }
8878 Some(value)
8879 }
8880 _ => return Err(invalid("nonzero count tag or column type differs")),
8881 };
8882 }
8883 }
8884 }
8885 if !cur.done() {
8886 if cur.take(8)? != AGGREGATE_SUMS {
8887 return Err(invalid("aggregate catalog extension magic differs"));
8888 }
8889 for entry in &mut entries {
8890 for (field, summary) in entry.fields.iter().zip(&mut entry.aggregates) {
8891 *summary = match cur.u8()? {
8892 0 => None,
8893 1 if signed_integer(&field.ty) => {
8894 let sum = i128::from_le_bytes(
8895 cur.take(16)?
8896 .try_into()
8897 .map_err(|_| invalid("aggregate sum is truncated"))?,
8898 );
8899 let count = cur.u64()?;
8900 if count > entry.rows as u64 {
8901 return Err(invalid("aggregate count exceeds table rows"));
8902 }
8903 Some((sum, count))
8904 }
8905 _ => return Err(invalid("aggregate sum tag or column type differs")),
8906 };
8907 }
8908 }
8909 }
8910 if !cur.done() {
8911 if cur.take(8)? != DISTINCT_COUNTS {
8912 return Err(invalid("distinct catalog extension magic differs"));
8913 }
8914 for entry in &mut entries {
8915 for count in &mut entry.distincts {
8916 *count = match cur.u8()? {
8917 0 => None,
8918 1 => {
8919 let value = cur.u64()?;
8920 if value > entry.rows as u64 {
8921 return Err(invalid("distinct count exceeds table rows"));
8922 }
8923 Some(value)
8924 }
8925 _ => return Err(invalid("distinct count tag differs")),
8926 };
8927 }
8928 }
8929 }
8930 if !cur.done() {
8931 if cur.take(8)? != INTEGER_EXTREMES {
8932 return Err(invalid("integer extremes catalog extension magic differs"));
8933 }
8934 for entry in &mut entries {
8935 for (field, extremes) in entry.fields.iter().zip(&mut entry.extremes) {
8936 *extremes = match cur.u8()? {
8937 0 => None,
8938 1 if integer_or_date(&field.ty) => Some(None),
8939 2 if integer_or_date(&field.ty) => {
8940 let low = i128::from_le_bytes(
8941 cur.take(16)?
8942 .try_into()
8943 .map_err(|_| invalid("minimum is truncated"))?,
8944 );
8945 let high = i128::from_le_bytes(
8946 cur.take(16)?
8947 .try_into()
8948 .map_err(|_| invalid("maximum is truncated"))?,
8949 );
8950 if low > high {
8951 return Err(invalid("integer extremes are reversed"));
8952 }
8953 Some(Some((low, high)))
8954 }
8955 _ => return Err(invalid("integer extremes tag or type differs")),
8956 };
8957 }
8958 }
8959 }
8960 if !cur.done() {
8961 if cur.take(8)? != COMPLETE_FREQUENCIES {
8962 return Err(invalid("numeric frequency catalog extension magic differs"));
8963 }
8964 for entry in &mut entries {
8965 for (field, frequencies) in entry.fields.iter().zip(&mut entry.frequencies) {
8966 *frequencies = match cur.u8()? {
8967 0 => None,
8968 1 if integer_or_date(&field.ty) => {
8969 let len = cur.u8()? as usize;
8970 if len > MAX_CATALOG_FREQUENCIES {
8971 return Err(invalid("too many catalog numeric frequencies"));
8972 }
8973 let mut values = Vec::with_capacity(len);
8974 let mut total = 0_u64;
8975 for _ in 0..len {
8976 let value = match cur.u8()? {
8977 0 => None,
8978 1 => Some(i128::from_le_bytes(cur.take(16)?.try_into().map_err(
8979 |_| invalid("numeric frequency value is truncated"),
8980 )?)),
8981 _ => return Err(invalid("numeric frequency value tag differs")),
8982 };
8983 if values.iter().any(|(held, _)| *held == value) {
8984 return Err(invalid("numeric frequency value repeats"));
8985 }
8986 let count = cur.u64()?;
8987 total = total
8988 .checked_add(count)
8989 .ok_or_else(|| invalid("numeric frequency count overflows"))?;
8990 values.push((value, count));
8991 }
8992 if total != entry.rows as u64 {
8993 return Err(invalid("numeric frequencies do not cover table rows"));
8994 }
8995 Some(values)
8996 }
8997 _ => return Err(invalid("numeric frequency tag or type differs")),
8998 };
8999 }
9000 }
9001 }
9002 if !cur.done() {
9003 return Err(invalid("catalog has trailing bytes"));
9004 }
9005 Ok((entries, views))
9006}
9007
9008struct Cursor<'a> {
9016 bytes: &'a [u8],
9017 at: usize,
9018 window: Option<Window<'a>>,
9019}
9020
9021struct Window<'a> {
9023 file: &'a File,
9024 offset: u64,
9025 length: usize,
9026 start: usize,
9028 held: Vec<u8>,
9029 size: usize,
9031}
9032
9033const DIRECTORY_WINDOW: usize = 64 << 10;
9035
9036impl<'a> Cursor<'a> {
9037 fn new(bytes: &'a [u8]) -> Self {
9038 Self { bytes, at: 0, window: None }
9039 }
9040
9041 fn over(file: &'a File, offset: u64, length: usize) -> Self {
9043 let window =
9044 Window { file, offset, length, start: 0, held: Vec::new(), size: DIRECTORY_WINDOW };
9045 Self { bytes: &[], at: 0, window: Some(window) }
9046 }
9047
9048 fn len(&self) -> usize {
9050 self.window.as_ref().map_or(self.bytes.len(), |window| window.length)
9051 }
9052
9053 fn ensure(&mut self, len: usize) -> Result<()> {
9055 let end = self.at.checked_add(len).ok_or_else(|| invalid("directory offset overflow"))?;
9056 if end > self.len() {
9057 return Err(invalid("directory is truncated"));
9058 }
9059 let Some(window) = &mut self.window else { return Ok(()) };
9060 if self.at < window.start || end > window.start + window.held.len() {
9061 let want = len.max(window.size).min(window.length - self.at);
9062 window.start = self.at;
9063 window.held.resize(want, 0);
9064 read_at(window.file, window.offset + self.at as u64, &mut window.held)?;
9065 }
9066 Ok(())
9067 }
9068
9069 fn held(&self, at: usize, len: usize) -> &[u8] {
9071 match &self.window {
9072 Some(window) => &window.held[at - window.start..at - window.start + len],
9073 None => &self.bytes[at..at + len],
9074 }
9075 }
9076
9077 #[inline]
9079 fn peek(&mut self, len: usize) -> Result<&[u8]> {
9080 if self.window.is_none() {
9081 let bytes = self.bytes;
9082 return Ok(&bytes[self.at..self.end(len)?]);
9083 }
9084 self.ensure(len)?;
9085 Ok(self.held(self.at, len))
9086 }
9087
9088 #[inline]
9094 fn take(&mut self, len: usize) -> Result<&[u8]> {
9095 if self.window.is_none() {
9096 let bytes = self.bytes;
9097 let (at, end) = (self.at, self.end(len)?);
9098 self.at = end;
9099 return Ok(&bytes[at..end]);
9100 }
9101 self.take_windowed(len)
9102 }
9103
9104 fn skip(&mut self, len: usize) -> Result<()> {
9106 let end = self.at.checked_add(len).ok_or_else(|| invalid("directory offset overflow"))?;
9107 if end > self.len() {
9108 return Err(invalid("directory is truncated"));
9109 }
9110 self.at = end;
9111 Ok(())
9112 }
9113
9114 fn skip_bound(&mut self) -> Result<()> {
9115 match self.u8()? {
9116 0 => Ok(()),
9117 1 => self.skip(16),
9118 2 => self.skip(8),
9119 3 => {
9120 let length = self.u32()? as usize;
9121 self.skip(length)
9122 }
9123 4 => self.skip(17),
9124 _ => Err(invalid("a stored bound has an unknown tag")),
9125 }
9126 }
9127
9128 #[inline]
9130 fn end(&self, len: usize) -> Result<usize> {
9131 let end = self.at.checked_add(len).ok_or_else(|| invalid("directory offset overflow"))?;
9132 if end > self.bytes.len() {
9133 return Err(invalid("directory is truncated"));
9134 }
9135 Ok(end)
9136 }
9137
9138 #[inline(never)]
9140 fn take_windowed(&mut self, len: usize) -> Result<&[u8]> {
9141 self.ensure(len)?;
9142 self.at += len;
9143 Ok(self.held(self.at - len, len))
9144 }
9145 #[inline]
9146 fn u8(&mut self) -> Result<u8> {
9147 Ok(self.take(1)?[0])
9148 }
9149 #[inline]
9150 fn u16(&mut self) -> Result<u16> {
9151 Ok(u16::from_le_bytes(self.take(2)?.try_into().expect("two bytes")))
9152 }
9153 #[inline]
9154 fn u32(&mut self) -> Result<u32> {
9155 Ok(u32::from_le_bytes(self.take(4)?.try_into().expect("four bytes")))
9156 }
9157 #[inline]
9158 fn u64(&mut self) -> Result<u64> {
9159 Ok(u64::from_le_bytes(self.take(8)?.try_into().expect("eight bytes")))
9160 }
9161 fn var_u64(&mut self) -> Result<u64> {
9162 let mut value = 0_u64;
9163 for shift in (0..=63).step_by(7) {
9164 let byte = self.u8()?;
9165 let part = u64::from(byte & 0x7f);
9166 if shift == 63 && part > 1 {
9167 return Err(invalid("frequency ordinal varint overflows"));
9168 }
9169 value |= part << shift;
9170 if byte & 0x80 == 0 {
9171 return Ok(value);
9172 }
9173 }
9174 Err(invalid("frequency ordinal varint is too long"))
9175 }
9176 fn bound(&mut self) -> Result<Option<Bound>> {
9185 let rest = self.len().saturating_sub(self.at);
9186 let mut want = 32;
9187 loop {
9188 let offered = self.peek(want.min(rest))?;
9189 let mut used = 0;
9190 match bounds::get(offered, &mut used) {
9191 Ok(bound) => {
9192 self.at += used;
9193 return Ok(bound);
9194 }
9195 Err(_) if want < rest => want *= 2,
9196 Err(error) => return Err(error),
9197 }
9198 }
9199 }
9200 fn text(&mut self) -> Result<String> {
9201 let len = self.u16()? as usize;
9202 String::from_utf8(self.take(len)?.to_vec()).map_err(|_| invalid("name is not UTF-8"))
9203 }
9204 fn done(&self) -> bool {
9207 self.at >= self.len()
9208 }
9209 fn long_text(&mut self) -> Result<String> {
9216 let len = self.u32()? as usize;
9217 String::from_utf8(self.take(len)?.to_vec()).map_err(|_| invalid("text is not UTF-8"))
9218 }
9219}
9220
9221fn decode_summary(
9223 cur: &mut Cursor<'_>,
9224 field: &Field,
9225 rows: usize,
9226 values: bool,
9227) -> Result<Option<FrequencySummary>> {
9228 Ok(match cur.u8()? {
9229 0 => None,
9230 1 => {
9231 let omitted_max = cur.u64()?;
9232 let count = cur.u32()? as usize;
9233 if count > FREQUENCY_ENTRIES {
9234 return Err(invalid("frequency entry count exceeds its bound"));
9235 }
9236 let mut entries = Vec::with_capacity(count);
9237 for _ in 0..count {
9239 let value = match cur.u8()? {
9240 0 => FrequencyValue::Null,
9241 1 => FrequencyValue::Integer(i128::from_le_bytes(
9242 cur.take(16)?.try_into().expect("sixteen bytes"),
9243 )),
9244 2 => FrequencyValue::Code(cur.u32()?),
9245 _ => return Err(invalid("frequency value tag differs")),
9246 };
9247 let valid = matches!(
9248 (&field.ty, value),
9249 (_, FrequencyValue::Null)
9250 | (LogicalType::Varchar | LogicalType::Blob, FrequencyValue::Code(_))
9251 | (
9252 LogicalType::TinyInt
9253 | LogicalType::SmallInt
9254 | LogicalType::Integer
9255 | LogicalType::BigInt
9256 | LogicalType::UTinyInt
9257 | LogicalType::USmallInt
9258 | LogicalType::UInteger
9259 | LogicalType::UBigInt
9260 | LogicalType::Date
9261 | LogicalType::Timestamp,
9262 FrequencyValue::Integer(_),
9263 )
9264 );
9265 if !valid {
9266 return Err(invalid("frequency value does not match its column"));
9267 }
9268 let count = cur.u64()?;
9269 if count == 0 || count > rows as u64 {
9270 return Err(invalid("frequency count is outside the table"));
9271 }
9272 entries.push(FrequencyEntry { value, count });
9273 }
9274 if entries.windows(2).any(|pair| pair[0].count < pair[1].count) {
9275 return Err(invalid("frequency entries are not descending"));
9276 }
9277 let ordinals = {
9278 let ordinal_count = cur.u32()? as usize;
9279 if ordinal_count > FREQUENCY_ORDINALS || ordinal_count > rows {
9280 return Err(invalid("frequency ordinal count exceeds its bound"));
9281 }
9282 let mut ordinals = Vec::with_capacity(ordinal_count);
9283 let mut previous = 0_u64;
9284 for at in 0..ordinal_count {
9285 let delta = cur.var_u64()?;
9286 if at != 0 && delta == 0 {
9287 return Err(invalid("frequency ordinals are not increasing"));
9288 }
9289 let ordinal = if at == 0 {
9290 delta
9291 } else {
9292 previous
9293 .checked_add(delta)
9294 .ok_or_else(|| invalid("frequency ordinal overflows"))?
9295 };
9296 if ordinal >= rows as u64 {
9297 return Err(invalid("frequency ordinal is outside the table"));
9298 }
9299 ordinals.push(ordinal);
9300 previous = ordinal;
9301 }
9302 ordinals
9303 };
9304 let ordinal_entries = if values {
9305 let mut ordinal_entries = Vec::with_capacity(ordinals.len());
9306 for _ in 0..ordinals.len() {
9307 let entry = cur.u16()?;
9308 if entry as usize >= entries.len() {
9309 return Err(invalid("frequency ordinal value is outside its entries"));
9310 }
9311 ordinal_entries.push(entry);
9312 }
9313 ordinal_entries
9314 } else {
9315 Vec::new()
9316 };
9317 Some(FrequencySummary { entries, omitted_max, ordinals, ordinal_entries })
9318 }
9319 _ => return Err(invalid("frequency summary tag differs")),
9320 })
9321}
9322
9323fn summary_span(cur: &mut Cursor<'_>) -> Result<Option<(usize, usize)>> {
9325 let length = cur.u32()? as usize;
9326 let entries = cur.u32()? as usize;
9327 if entries > FREQUENCY_ENTRIES {
9328 return Err(invalid("frequency entry count exceeds its bound"));
9329 }
9330 if length == 0 {
9331 if entries != 0 {
9332 return Err(invalid("missing frequency synopsis has entries"));
9333 }
9334 return Ok(None);
9335 }
9336 if length > MAX_DIRECTORY || length > cur.len().saturating_sub(cur.at) {
9337 return Err(invalid("frequency synopsis span is outside the directory"));
9338 }
9339 Ok(Some((length, entries)))
9340}
9341
9342fn skip_summary(cur: &mut Cursor<'_>, values: bool, rows: usize) -> Result<()> {
9345 match cur.u8()? {
9346 0 => Ok(()),
9347 1 => {
9348 cur.skip(8)?;
9349 let entries = cur.u32()? as usize;
9350 if entries > FREQUENCY_ENTRIES {
9351 return Err(invalid("frequency entry count exceeds its bound"));
9352 }
9353 for _ in 0..entries {
9354 match cur.u8()? {
9355 0 => {}
9356 1 => cur.skip(16)?,
9357 2 => cur.skip(4)?,
9358 _ => return Err(invalid("frequency value tag differs")),
9359 }
9360 cur.skip(8)?;
9361 }
9362 let ordinals = cur.u32()? as usize;
9363 if ordinals > FREQUENCY_ORDINALS || ordinals > rows {
9364 return Err(invalid("frequency ordinal count exceeds its bound"));
9365 }
9366 for _ in 0..ordinals {
9367 cur.var_u64()?;
9368 }
9369 if values {
9370 cur.skip(ordinals * 2)?;
9371 }
9372 Ok(())
9373 }
9374 _ => Err(invalid("frequency summary tag differs")),
9375 }
9376}
9377
9378fn quick_nonzero(
9382 mut cur: Cursor<'_>,
9383 name: &str,
9384 fields: &[Field],
9385 rows: usize,
9386 wanted: usize,
9387) -> Result<Option<u64>> {
9388 if cur.take(8)? != DIRECTORY || cur.text()? != name {
9389 return Err(invalid("table directory differs from the catalog"));
9390 }
9391 let width = cur.u16()? as usize;
9392 if width != fields.len() {
9393 return Err(invalid("table directory width differs from the catalog"));
9394 }
9395 for field in fields {
9396 let stored =
9397 Field { name: cur.text()?, ty: read_type(&mut cur)?, not_null: cur.u8()? != 0 };
9398 if &stored != field {
9399 return Err(invalid("table directory schema differs from the catalog"));
9400 }
9401 }
9402 let mut dictionaries = Vec::with_capacity(width);
9403 for field in fields {
9404 let held = match cur.u8()? {
9405 0 => false,
9406 tag if coded_type(&field.ty) && tag == dictionary_tag(&field.ty) => {
9407 cur.skip(20)?;
9408 true
9409 }
9410 _ => return Err(invalid("dictionary page tag differs")),
9411 };
9412 dictionaries.push(held);
9413 }
9414 for _ in 0..width {
9415 match cur.u8()? {
9416 0 => {}
9417 1 => cur.skip(8)?,
9418 _ => return Err(invalid("distinct count tag differs")),
9419 }
9420 }
9421 if cur.u64()? != rows as u64 {
9422 return Err(invalid("table row count differs from the catalog"));
9423 }
9424 let stripes = cur.u32()? as usize;
9425 let mut total = 0_usize;
9426 let mut nulls = 0_u64;
9427 for _ in 0..stripes {
9428 let parts = cur.u32()? as usize;
9429 if parts == 0 || parts > STRIPE_PARTS {
9430 return Err(invalid("stripe part count is outside its bound"));
9431 }
9432 let mut stripe_rows = 0_usize;
9433 for _ in 0..parts {
9434 stripe_rows = stripe_rows
9435 .checked_add(cur.u32()? as usize)
9436 .ok_or_else(|| invalid("stripe row count overflow"))?;
9437 }
9438 total =
9439 total.checked_add(stripe_rows).ok_or_else(|| invalid("stripe row count overflow"))?;
9440 cur.skip(12 + width * 12)?;
9441 for (field, held) in fields.iter().zip(&dictionaries) {
9442 if coded_type(&field.ty) && *held {
9443 cur.skip(20)?;
9444 }
9445 }
9446 for _ in 0..width * 2 {
9447 match cur.u8()? {
9448 0 => {}
9449 1 => cur.skip(20)?,
9450 _ => return Err(invalid("stripe page tag differs")),
9451 }
9452 }
9453 for column in 0..width {
9454 cur.skip_bound()?;
9455 cur.skip_bound()?;
9456 let count = cur.u32()? as u64;
9457 if count > stripe_rows as u64 {
9458 return Err(invalid("null count exceeds stripe rows"));
9459 }
9460 if column == wanted {
9461 nulls = nulls.checked_add(count).ok_or_else(|| invalid("null count overflow"))?;
9462 }
9463 cur.skip(1)?;
9464 match cur.u8()? {
9465 0 => {}
9466 1 => cur.skip(16)?,
9467 _ => return Err(invalid("a stripe sum has an unknown tag")),
9468 }
9469 }
9470 }
9471 if total != rows {
9472 return Err(invalid("table row count differs from stripes"));
9473 }
9474 if cur.done() {
9475 return Ok(None);
9476 }
9477 let magic = cur.take(8)?;
9478 let spanned = magic == FREQUENCIES_SPANS;
9479 let values = magic == FREQUENCIES || spanned;
9480 if !values && magic != FREQUENCIES_V2 {
9481 return Err(invalid("directory extension magic differs"));
9482 }
9483 if cur.u16()? as usize != width {
9484 return Err(invalid("frequency column count differs"));
9485 }
9486 for _ in 0..wanted {
9487 if spanned {
9488 if let Some((length, _)) = summary_span(&mut cur)? {
9489 cur.skip(length)?;
9490 }
9491 } else {
9492 skip_summary(&mut cur, values, rows)?;
9493 }
9494 }
9495 let summary = if spanned {
9496 let Some((length, entries)) = summary_span(&mut cur)? else {
9497 return Ok(None);
9498 };
9499 let start = cur.at;
9500 let summary = decode_summary(&mut cur, &fields[wanted], rows, values)?
9501 .ok_or_else(|| invalid("a stored synopsis is missing"))?;
9502 if cur.at - start != length || summary.entries.len() != entries {
9503 return Err(invalid("a stored synopsis differs from its directory span"));
9504 }
9505 summary
9506 } else {
9507 let Some(summary) = decode_summary(&mut cur, &fields[wanted], rows, values)? else {
9508 return Ok(None);
9509 };
9510 summary
9511 };
9512 let zero = summary
9513 .entries
9514 .iter()
9515 .find(|entry| entry.value == FrequencyValue::Integer(0))
9516 .map(|entry| entry.count)
9517 .or_else(|| (summary.omitted_max == 0).then_some(0));
9518 Ok(zero.and_then(|zero| (rows as u64).checked_sub(nulls)?.checked_sub(zero)))
9519}
9520
9521fn quick_integer_fold(
9524 file: &File,
9525 mut cur: Cursor<'_>,
9526 entry: &Entry,
9527 size: u64,
9528 wanted: usize,
9529 emit: &mut impl FnMut(i64, u64) -> Result<()>,
9530) -> Result<()> {
9531 let name = &entry.name;
9532 let fields = &entry.fields;
9533 let rows = entry.rows;
9534 if cur.take(8)? != DIRECTORY || cur.text()? != name.as_str() {
9535 return Err(invalid("table directory differs from the catalog"));
9536 }
9537 let width = cur.u16()? as usize;
9538 if width != fields.len() {
9539 return Err(invalid("table directory width differs from the catalog"));
9540 }
9541 for field in fields {
9542 let stored =
9543 Field { name: cur.text()?, ty: read_type(&mut cur)?, not_null: cur.u8()? != 0 };
9544 if &stored != field {
9545 return Err(invalid("table directory schema differs from the catalog"));
9546 }
9547 }
9548 let mut dictionaries = Vec::with_capacity(width);
9549 for field in fields {
9550 dictionaries.push(match cur.u8()? {
9551 0 => false,
9552 tag if coded_type(&field.ty) && tag == dictionary_tag(&field.ty) => {
9553 cur.skip(20)?;
9554 true
9555 }
9556 _ => return Err(invalid("dictionary page tag differs")),
9557 });
9558 }
9559 for _ in 0..width {
9560 match cur.u8()? {
9561 0 => {}
9562 1 => cur.skip(8)?,
9563 _ => return Err(invalid("distinct count tag differs")),
9564 }
9565 }
9566 if cur.u64()? != rows as u64 {
9567 return Err(invalid("table row count differs from the catalog"));
9568 }
9569 let stripes = cur.u32()? as usize;
9570 let mut total = 0_usize;
9571 let mut bytes = Vec::new();
9572 for _ in 0..stripes {
9573 let parts = cur.u32()? as usize;
9574 if parts == 0 || parts > STRIPE_PARTS {
9575 return Err(invalid("stripe part count is outside its bound"));
9576 }
9577 let mut part_rows = Vec::with_capacity(parts);
9578 for _ in 0..parts {
9579 let count = cur.u32()? as usize;
9580 if count == 0 {
9581 return Err(invalid("empty part"));
9582 }
9583 total = total.checked_add(count).ok_or_else(|| invalid("stripe row count overflow"))?;
9584 part_rows.push(count);
9585 }
9586 let index = Span { offset: cur.u64()?, length: cur.u32()? };
9587 let section = index_section(parts)?;
9588 let index_length =
9589 section.checked_mul(width).ok_or_else(|| invalid("index page length overflow"))?;
9590 if index.offset < HEADER
9591 || index.offset.checked_add(u64::from(index.length)).is_none_or(|end| end > size)
9592 || index.length as usize != index_length
9593 {
9594 return Err(invalid("index page range is outside the file"));
9595 }
9596 cur.skip(wanted * 12)?;
9597 let page = Span { offset: cur.u64()?, length: cur.u32()? };
9598 if page.offset < HEADER
9599 || page.offset.checked_add(u64::from(page.length)).is_none_or(|end| end > size)
9600 || page.length as usize > MAX_PAGE
9601 {
9602 return Err(invalid("column page range is outside the file"));
9603 }
9604 cur.skip((width - wanted - 1) * 12)?;
9605 for (field, held) in fields.iter().zip(&dictionaries) {
9606 if coded_type(&field.ty) && *held {
9607 cur.skip(20)?;
9608 }
9609 }
9610 for _ in 0..width * 2 {
9611 match cur.u8()? {
9612 0 => {}
9613 1 => cur.skip(20)?,
9614 _ => return Err(invalid("stripe page tag differs")),
9615 }
9616 }
9617 for _ in 0..width {
9618 cur.skip_bound()?;
9619 cur.skip_bound()?;
9620 cur.skip(5)?;
9621 match cur.u8()? {
9622 0 => {}
9623 1 => cur.skip(16)?,
9624 _ => return Err(invalid("a stripe sum has an unknown tag")),
9625 }
9626 }
9627 let spans = read_index_span(file, index, page, parts, wanted)?;
9628 for (span, expected_rows) in spans.into_iter().zip(part_rows) {
9629 bytes.resize(span.length, 0);
9630 let at = page
9631 .offset
9632 .checked_add(span.start as u64)
9633 .ok_or_else(|| invalid("part range overflow"))?;
9634 read_at(file, at, &mut bytes)?;
9635 if checksum(&bytes) != span.hash {
9636 return Err(invalid("integer part checksum differs"));
9637 }
9638 if bytes.first() == Some(&5) && bytes.get(1) == Some(&0) {
9639 let decoded_rows = integer::fold(&bytes[2..], |value, count| {
9640 check_integer_tally_value(value, &fields[wanted].ty)?;
9641 emit(value, count)
9642 })?;
9643 if decoded_rows != expected_rows {
9644 return Err(invalid("encoded integer part holds the wrong number of rows"));
9645 }
9646 } else {
9647 let column = decode(&fields[wanted].ty, expected_rows, &bytes, None)?;
9648 if let Some(packed) = column.packed_parts() {
9649 let validity = column.validity();
9650 let all_valid = column.none_null();
9651 let base = packed.base();
9652 let mut codes = [0_u64; 64];
9653 for from in (0..expected_rows).step_by(codes.len()) {
9654 let count = (expected_rows - from).min(codes.len());
9655 packed.unpack(from, &mut codes[..count]);
9656 for (offset, &code) in codes[..count].iter().enumerate() {
9657 if all_valid || validity.is_valid(from + offset) {
9658 emit((base + i128::from(code)) as i64, 1)?;
9660 }
9661 }
9662 }
9663 continue;
9664 }
9665 let column = column.into_flat()?;
9666 let validity = column.validity();
9667 macro_rules! count_decoded {
9668 ($values:expr) => {
9669 for (row, &value) in $values.as_slice().iter().enumerate() {
9670 if validity.is_valid(row) {
9671 emit(i64::from(value), 1)?;
9672 }
9673 }
9674 };
9675 }
9676 match column.data() {
9677 Some(Data::Int8(values)) => count_decoded!(values),
9678 Some(Data::Int16(values)) => count_decoded!(values),
9679 Some(Data::Int32(values)) => count_decoded!(values),
9680 Some(Data::Int64(values)) => count_decoded!(values),
9681 _ => return Err(invalid("decoded integer part has the wrong type")),
9682 }
9683 }
9684 }
9685 }
9686 if total != rows {
9687 return Err(invalid("table row count differs from stripes"));
9688 }
9689 Ok(())
9690}
9691
9692fn check_integer_tally_value(value: i64, ty: &LogicalType) -> Result<()> {
9693 let fits = match ty {
9694 LogicalType::TinyInt => i8::try_from(value).is_ok(),
9695 LogicalType::SmallInt => i16::try_from(value).is_ok(),
9696 LogicalType::Integer => i32::try_from(value).is_ok(),
9697 LogicalType::BigInt => true,
9698 _ => false,
9699 };
9700 if fits { Ok(()) } else { Err(invalid("encoded integer value is outside its column type")) }
9701}
9702
9703fn decode_directory(bytes: &[u8], size: u64) -> Result<Table> {
9704 read_directory(Cursor::new(bytes), size, None)
9705}
9706
9707fn read_directory(mut cur: Cursor<'_>, size: u64, stored_at: Option<u64>) -> Result<Table> {
9712 if cur.take(8)? != DIRECTORY {
9713 return Err(invalid("directory magic differs"));
9714 }
9715 let name = cur.text()?;
9716 let width = cur.u16()? as usize;
9717 let mut fields = Vec::with_capacity(width);
9718 for _ in 0..width {
9719 let name = cur.text()?;
9720 let ty = read_type(&mut cur)?;
9721 let not_null = match cur.u8()? {
9722 0 => false,
9723 1 => true,
9724 _ => return Err(invalid("nullability flag differs")),
9725 };
9726 fields.push(Field { name, ty, not_null });
9727 }
9728 let mut dictionaries = Vec::with_capacity(width);
9729 for field in &fields {
9730 dictionaries.push(match cur.u8()? {
9731 0 => None,
9732 tag if tag == dictionary_tag(&field.ty) => {
9733 let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
9734 let end = page
9735 .offset
9736 .checked_add(u64::from(page.length))
9737 .ok_or_else(|| invalid("dictionary page offset overflow"))?;
9738 if page.offset < HEADER || end > size {
9743 return Err(invalid("dictionary page range is outside the file"));
9744 }
9745 Some(page)
9746 }
9747 _ => return Err(invalid("dictionary page tag differs")),
9748 });
9749 }
9750 let mut distincts = Vec::with_capacity(width);
9751 for _ in 0..width {
9752 distincts.push(match cur.u8()? {
9753 0 => None,
9754 1 => Some(cur.u64()?),
9755 _ => return Err(invalid("distinct count tag differs")),
9756 });
9757 }
9758 let rows = usize::try_from(cur.u64()?).map_err(|_| invalid("row count does not fit"))?;
9759 let count = cur.u32()? as usize;
9760 let mut stripes = Vec::with_capacity(count);
9761 let mut total = 0_usize;
9762 for _ in 0..count {
9763 let count = cur.u32()? as usize;
9764 if count == 0 || count > STRIPE_PARTS {
9765 return Err(invalid("stripe part count is outside its bound"));
9766 }
9767 let mut parts = Vec::with_capacity(count);
9768 let mut stripe_rows = 0_usize;
9769 for _ in 0..count {
9770 let rows = cur.u32()?;
9771 if rows == 0 {
9772 return Err(invalid("empty part"));
9773 }
9774 parts.push(rows);
9775 stripe_rows = stripe_rows
9776 .checked_add(rows as usize)
9777 .ok_or_else(|| invalid("stripe row count overflow"))?;
9778 }
9779 total =
9780 total.checked_add(stripe_rows).ok_or_else(|| invalid("stripe row count overflow"))?;
9781 let index = Span { offset: cur.u64()?, length: cur.u32()? };
9782 let section = index_section(count)?;
9783 let wanted = section
9784 .checked_mul(width)
9785 .and_then(|bytes| u32::try_from(bytes).ok())
9786 .ok_or_else(|| invalid("index page length overflow"))?;
9787 let end = index
9788 .offset
9789 .checked_add(u64::from(index.length))
9790 .ok_or_else(|| invalid("index page offset overflow"))?;
9791 if index.offset < HEADER || end > size || index.length != wanted {
9792 return Err(invalid("index page range is outside the file"));
9793 }
9794 let mut pages = Vec::with_capacity(width);
9795 for _ in 0..width {
9796 let offset = cur.u64()?;
9797 let length = cur.u32()?;
9798 let end = offset
9799 .checked_add(u64::from(length))
9800 .ok_or_else(|| invalid("page offset overflow"))?;
9801 if offset < HEADER || end > size || length as usize > MAX_PAGE {
9802 return Err(invalid("page range is outside the file"));
9803 }
9804 pages.push(Span { offset, length });
9805 }
9806 let mut memberships = vec![None; width];
9807 for (column, field) in fields.iter().enumerate() {
9808 if !coded_type(&field.ty) || dictionaries[column].is_none() {
9809 continue;
9810 }
9811 let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
9812 let end = page
9813 .offset
9814 .checked_add(u64::from(page.length))
9815 .ok_or_else(|| invalid("membership page offset overflow"))?;
9816 if page.offset < HEADER || end > size || page.length as usize > MAX_PAGE {
9817 return Err(invalid("membership page range is outside the file"));
9818 }
9819 if page.length != 0 {
9822 memberships[column] = Some(page);
9823 }
9824 }
9825 let mut sieves = vec![None; width];
9826 for sieve in sieves.iter_mut().take(width) {
9827 match cur.u8()? {
9828 0 => continue,
9829 1 => {}
9830 _ => return Err(invalid("a sieve page has an unknown tag")),
9831 }
9832 let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
9833 let end = page
9834 .offset
9835 .checked_add(u64::from(page.length))
9836 .ok_or_else(|| invalid("sieve page offset overflow"))?;
9837 if page.offset < HEADER || end > size || page.length as usize > MAX_PAGE {
9838 return Err(invalid("sieve page range is outside the file"));
9839 }
9840 *sieve = Some(page);
9841 }
9842 let mut part_ranges = vec![None; width];
9843 for held in part_ranges.iter_mut().take(width) {
9844 match cur.u8()? {
9845 0 => continue,
9846 1 => {}
9847 _ => return Err(invalid("a part range page has an unknown tag")),
9848 }
9849 let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
9850 let end = page
9851 .offset
9852 .checked_add(u64::from(page.length))
9853 .ok_or_else(|| invalid("part range page offset overflow"))?;
9854 if page.offset < HEADER || end > size || page.length as usize > MAX_PAGE {
9855 return Err(invalid("part range page range is outside the file"));
9856 }
9857 *held = Some(page);
9858 }
9859 let mut ranges = Vec::with_capacity(width);
9860 for column in 0..width {
9861 let low = cur.bound()?;
9862 let high = cur.bound()?;
9863 let nulls = cur.u32()? as usize;
9864 if nulls > stripe_rows {
9865 return Err(invalid("null count exceeds stripe rows"));
9866 }
9867 let exact = cur.u8()? != 0;
9868 let sum = match cur.u8()? {
9869 0 => None,
9870 1 => Some(i128::from_le_bytes(
9871 cur.take(16)?.try_into().map_err(|_| invalid("a stripe sum is truncated"))?,
9872 )),
9873 _ => return Err(invalid("a stripe sum has an unknown tag")),
9874 };
9875 let ty = &fields.get(column).ok_or_else(|| invalid("a stripe range has no column"))?.ty;
9881 let low = low.map(|bound| scaled_as(bound, ty));
9882 let high = high.map(|bound| scaled_as(bound, ty));
9883 ranges.push(Range { low, high, nulls, exact, sum });
9884 }
9885 stripes.push(Stripe {
9886 rows: stripe_rows,
9887 parts,
9888 index,
9889 pages,
9890 memberships: Pages::from_slots(memberships)?,
9891 sieves: Pages::from_slots(sieves)?,
9892 part_ranges: Pages::from_slots(part_ranges)?,
9893 zone: Zone::from_ranges(ranges),
9894 });
9895 }
9896 if total != rows {
9897 return Err(invalid("table row count differs from stripes"));
9898 }
9899 let mut entry_counts = vec![0; width];
9902 let frequencies = if cur.done() {
9903 vec![None; width]
9904 } else {
9905 let frequency_magic = cur.take(8)?;
9906 let spanned = frequency_magic == FREQUENCIES_SPANS;
9907 let frequency_values = frequency_magic == FREQUENCIES || spanned;
9908 if !frequency_values && frequency_magic != FREQUENCIES_V2 {
9909 return Err(invalid("directory extension magic differs"));
9910 }
9911 if cur.u16()? as usize != width {
9912 return Err(invalid("frequency column count differs"));
9913 }
9914 let mut frequencies = Vec::with_capacity(width);
9915 for (field, entry_count) in fields.iter().zip(&mut entry_counts) {
9916 if spanned {
9917 let Some((length, entries)) = summary_span(&mut cur)? else {
9918 frequencies.push(None);
9919 continue;
9920 };
9921 *entry_count = entries;
9922 let start = cur.at;
9923 if let Some(offset) = stored_at {
9924 cur.skip(length)?;
9925 frequencies.push(Some(Frequencies::Stored {
9926 span: Span {
9927 offset: offset
9928 .checked_add(start as u64)
9929 .ok_or_else(|| invalid("frequency synopsis offset overflow"))?,
9930 length: u32::try_from(length)
9931 .map_err(|_| invalid("a frequency synopsis is too long"))?,
9932 },
9933 values: true,
9934 entries,
9935 }));
9936 } else {
9937 let summary = decode_summary(&mut cur, field, rows, true)?
9938 .ok_or_else(|| invalid("a stored synopsis is missing"))?;
9939 if cur.at - start != length || summary.entries.len() != entries {
9940 return Err(invalid("a stored synopsis differs from its directory span"));
9941 }
9942 frequencies.push(Some(Frequencies::Held(summary)));
9943 }
9944 continue;
9945 }
9946 let start = cur.at;
9947 let summary = decode_summary(&mut cur, field, rows, frequency_values)?;
9948 *entry_count = summary.as_ref().map_or(0, |summary| summary.entries.len());
9949 frequencies.push(match (summary, stored_at) {
9950 (None, _) => None,
9951 (Some(summary), None) => Some(Frequencies::Held(summary)),
9952 (Some(summary), Some(offset)) => Some(Frequencies::Stored {
9953 span: Span {
9954 offset: offset + start as u64,
9955 length: u32::try_from(cur.at - start)
9956 .map_err(|_| invalid("a frequency synopsis is too long"))?,
9957 },
9958 values: frequency_values,
9959 entries: summary.entries.len(),
9960 }),
9961 });
9962 }
9963 frequencies
9964 };
9965 let mut clustering = None;
9975 let mut sections = Vec::new();
9976 let mut pair_frequencies = Vec::new();
9977 let mut seen_pair_frequencies = false;
9978 let mut frequency_texts = vec![Vec::new(); width];
9979 let mut seen_frequency_texts = false;
9980 let mut host_groups = None;
9981 let mut demoted = Vec::new();
9982 let mut seen_sections = false;
9983 let mut dictionary_payloads = Vec::new();
9984 let mut seen_payloads = false;
9985 let mut constraints = Constraints::default();
9986 let mut generation = 0;
9989 while !cur.done() {
9990 let mut tag = [0u8; 8];
9991 tag.copy_from_slice(cur.take(8)?);
9992 if &tag == PAIR_FREQUENCIES {
9993 if seen_pair_frequencies {
9994 return Err(invalid("directory names two pair frequency blocks"));
9995 }
9996 seen_pair_frequencies = true;
9997 let count = cur.u16()? as usize;
9998 if count > MAX_PAIR_FREQUENCIES {
9999 return Err(invalid("pair frequency count exceeds its bound"));
10000 }
10001 pair_frequencies = Vec::with_capacity(count);
10002 for _ in 0..count {
10003 let first = cur.u16()?;
10004 let second = cur.u16()?;
10005 let first_at = first as usize;
10006 let second_at = second as usize;
10007 if frequencies.get(first_at).and_then(Option::as_ref).is_none() {
10008 return Err(invalid("pair frequency first column has no synopsis"));
10009 }
10010 let first_entries = entry_counts[first_at];
10011 if !matches!(fields.get(second_at), Some(field) if field.ty == LogicalType::Varchar)
10012 || dictionaries.get(second_at).copied().flatten().is_none()
10013 {
10014 return Err(invalid("pair frequency second column has no stable dictionary"));
10015 }
10016 if pair_frequencies
10017 .iter()
10018 .any(|held: &PairFrequencySummary| held.first == first && held.second == second)
10019 {
10020 return Err(invalid("directory repeats a pair frequency summary"));
10021 }
10022 let omitted_max = cur.u64()?;
10023 if omitted_max > rows as u64 {
10024 return Err(invalid("pair frequency omitted count exceeds the table"));
10025 }
10026 let entries_count = cur.u16()? as usize;
10027 if entries_count > FREQUENCY_ENTRIES {
10028 return Err(invalid("pair frequency entry count exceeds its bound"));
10029 }
10030 let mut entries = Vec::with_capacity(entries_count);
10031 for _ in 0..entries_count {
10032 let first_entry = cur.u16()?;
10033 if first_entry as usize >= first_entries {
10034 return Err(invalid("pair frequency anchor is outside its synopsis"));
10035 }
10036 let second = match cur.u8()? {
10037 0 => None,
10038 1 => Some(cur.u32()?),
10039 _ => return Err(invalid("pair frequency string tag differs")),
10040 };
10041 let count = cur.u64()?;
10042 if count == 0 || count > rows as u64 {
10043 return Err(invalid("pair frequency count is outside the table"));
10044 }
10045 entries.push(PairFrequencyEntry { first_entry, second, count });
10046 }
10047 if entries.windows(2).any(|pair| pair[0].count < pair[1].count) {
10048 return Err(invalid("pair frequency entries are not descending"));
10049 }
10050 pair_frequencies.push(PairFrequencySummary { first, second, entries, omitted_max });
10051 }
10052 } else if &tag == FREQUENCY_TEXTS {
10053 if seen_frequency_texts {
10054 return Err(invalid("directory names two frequency text blocks"));
10055 }
10056 seen_frequency_texts = true;
10057 let columns = cur.u16()? as usize;
10058 if columns > width {
10059 return Err(invalid("frequency text column count exceeds the schema"));
10060 }
10061 for _ in 0..columns {
10062 let column = cur.u16()? as usize;
10063 if !frequency_texts.get(column).is_some_and(Vec::is_empty) {
10064 return Err(invalid("frequency text column is repeated or out of range"));
10065 }
10066 if !matches!(fields.get(column), Some(field) if coded_type(&field.ty))
10067 || dictionaries.get(column).copied().flatten().is_none()
10068 || frequencies.get(column).and_then(Option::as_ref).is_none()
10069 {
10070 return Err(invalid("frequency texts belong to a non-string synopsis"));
10071 }
10072 let count = cur.u16()? as usize;
10073 if count == 0 || count != entry_counts[column] {
10074 return Err(invalid("frequency text count differs from its synopsis"));
10075 }
10076 let mut texts = Vec::with_capacity(count);
10077 for _ in 0..count {
10078 texts.push(match cur.u8()? {
10079 0 => None,
10080 1 => {
10081 let length = cur.u32()? as usize;
10082 let bytes = cur.take(length)?.to_vec();
10083 if fields[column].ty == LogicalType::Varchar {
10084 std::str::from_utf8(&bytes)
10085 .map_err(|_| invalid("frequency text is not UTF-8"))?;
10086 }
10087 Some(bytes)
10088 }
10089 _ => return Err(invalid("frequency text tag differs")),
10090 });
10091 }
10092 frequency_texts[column] = texts;
10093 }
10094 } else if &tag == HOST_GROUPS {
10095 if host_groups.is_some() {
10096 return Err(invalid("directory names two host group blocks"));
10097 }
10098 let column = cur.u16()? as usize;
10099 if !matches!(fields.get(column), Some(field) if field.ty == LogicalType::Varchar)
10100 || dictionaries.get(column).copied().flatten().is_none()
10101 {
10102 return Err(invalid("host groups belong to a non-string dictionary"));
10103 }
10104 let omitted_max = cur.u64()?;
10105 if omitted_max > rows as u64 {
10106 return Err(invalid("host group bound exceeds the table"));
10107 }
10108 let count = cur.u16()? as usize;
10109 if count > host::CAPACITY {
10110 return Err(invalid("host group count exceeds its bound"));
10111 }
10112 let mut entries = Vec::with_capacity(count);
10113 let mut bytes = 0_usize;
10114 for _ in 0..count {
10115 let host_len = cur.u32()? as usize;
10116 bytes =
10117 bytes.checked_add(host_len).ok_or_else(|| invalid("host bytes overflow"))?;
10118 if bytes > host::BYTE_BUDGET {
10119 return Err(invalid("host groups exceed their byte budget"));
10120 }
10121 let host = std::str::from_utf8(cur.take(host_len)?)
10122 .map_err(|_| invalid("host is not UTF-8"))?
10123 .to_owned();
10124 let count = cur.u64()?;
10125 if count == 0 || count > rows as u64 {
10126 return Err(invalid("host group count exceeds the table"));
10127 }
10128 let bytes_sum = i128::from_le_bytes(
10129 cur.take(16)?
10130 .try_into()
10131 .map_err(|_| invalid("host length sum is truncated"))?,
10132 );
10133 if bytes_sum < 0 {
10134 return Err(invalid("host length sum is negative"));
10135 }
10136 let minimum_len = cur.u32()? as usize;
10137 bytes =
10138 bytes.checked_add(minimum_len).ok_or_else(|| invalid("host bytes overflow"))?;
10139 if bytes > host::BYTE_BUDGET {
10140 return Err(invalid("host groups exceed their byte budget"));
10141 }
10142 let minimum = std::str::from_utf8(cur.take(minimum_len)?)
10143 .map_err(|_| invalid("host minimum is not UTF-8"))?
10144 .to_owned();
10145 entries.push(host::HostEntry { host, count, bytes_sum, minimum });
10146 }
10147 if entries.windows(2).any(|pair| pair[0].count < pair[1].count)
10148 || entries.iter().any(|entry| entry.host.is_empty() || entry.minimum.is_empty())
10149 {
10150 return Err(invalid("host groups are not in certified order"));
10151 }
10152 host_groups = Some(host::HostSummary { column, omitted_max, entries });
10153 } else if &tag == CLUSTERING {
10154 if clustering.is_some() {
10155 return Err(invalid("directory names two clustering declarations"));
10156 }
10157 let bucket = Width::from_tag(cur.u8()?)
10158 .ok_or_else(|| invalid("clustering width tag differs"))?;
10159 let count = cur.u16()? as usize;
10160 let mut columns = Vec::with_capacity(count.min(fields.len()));
10161 for _ in 0..count {
10162 columns.push(u32::from(cur.u16()?));
10163 }
10164 clustering = Some(Clustering::new(columns, bucket, &fields).map_err(|_| {
10167 invalid("stored clustering declaration does not match the table it is on")
10168 })?);
10169 } else if &tag == DEMOTED {
10170 if !demoted.is_empty() {
10171 return Err(invalid("directory names two demoted column blocks"));
10172 }
10173 let count = cur.u16()? as usize;
10174 if count == 0 || count > width {
10175 return Err(invalid("demoted column count is outside the schema"));
10176 }
10177 demoted = vec![false; width];
10178 for _ in 0..count {
10179 let column = cur.u16()? as usize;
10180 if dictionaries.get(column).copied().flatten().is_none() || demoted[column] {
10181 return Err(invalid("a demoted column is repeated or has no dictionary"));
10182 }
10183 demoted[column] = true;
10184 }
10185 } else if &tag == SECTIONS {
10186 if seen_sections {
10187 return Err(invalid("directory names two section tables"));
10188 }
10189 seen_sections = true;
10190 generation = cur.u64()?;
10191 let count = cur.u16()? as usize;
10192 if count > MAX_SECTIONS {
10193 return Err(invalid("section count exceeds its bound"));
10194 }
10195 sections = Vec::with_capacity(count);
10196 for _ in 0..count {
10199 sections.push(Section::decode(cur.take(section::ENTRY_BYTES)?)?);
10200 }
10201 for held in §ions {
10202 let Some(end) = held.extent_page.checked_add(u64::from(held.extent_bytes)) else {
10203 return Err(invalid("a section's extent table overflows the file"));
10204 };
10205 if held.extent_bytes != 0 && (held.extent_page < HEADER || end > size) {
10209 return Err(invalid("a section's extent table is outside the file"));
10210 }
10211 if held.extents == 0 && held.extent_bytes != 0 {
10212 return Err(invalid("a section with no extents names an extent table"));
10213 }
10214 }
10215 } else if &tag == DICTIONARY_PAYLOADS {
10216 if seen_payloads {
10217 return Err(invalid("directory names two dictionary payload blocks"));
10218 }
10219 seen_payloads = true;
10220 let count = cur.u16()? as usize;
10221 if count != fields.len() {
10222 return Err(invalid("dictionary payload block does not match the table's columns"));
10223 }
10224 dictionary_payloads = Vec::with_capacity(count);
10225 for _ in 0..count {
10226 let bytes = cur.u64()?;
10227 if bytes > size {
10228 return Err(invalid("a dictionary payload is larger than the file"));
10229 }
10230 dictionary_payloads.push(bytes);
10231 }
10232 } else if &tag == KEYS {
10233 if !constraints.is_empty() {
10234 return Err(invalid("directory names two key blocks"));
10235 }
10236 let fits = |columns: &[u16]| {
10237 !columns.is_empty() && columns.iter().all(|&column| usize::from(column) < width)
10238 };
10239 let count = cur.u16()? as usize;
10240 for _ in 0..count {
10241 let primary = cur.u8()? != 0;
10242 let columns = columns_of(&mut cur)?;
10243 if !fits(&columns) {
10244 return Err(invalid("a stored key names a column the table does not have"));
10245 }
10246 constraints.keys.push((columns, primary));
10247 }
10248 let count = cur.u16()? as usize;
10249 for _ in 0..count {
10250 let columns = columns_of(&mut cur)?;
10251 let referenced = columns_of(&mut cur)?;
10252 let len = cur.u32()? as usize;
10253 let table = std::str::from_utf8(cur.take(len)?)
10254 .map_err(|_| invalid("a foreign key's table name is not UTF-8"))?
10255 .to_owned();
10256 if !fits(&columns) || referenced.len() != columns.len() || table.is_empty() {
10257 return Err(invalid("a stored foreign key does not match its table"));
10258 }
10259 constraints.foreign.push(StoredForeign { columns, table, referenced });
10260 }
10261 if constraints.is_empty() {
10262 return Err(invalid("a key block holds no key"));
10263 }
10264 } else {
10265 return Err(invalid("directory extension magic differs"));
10266 }
10267 }
10268 if !cur.done() {
10269 return Err(invalid("directory has trailing bytes"));
10270 }
10271 for stripe in &stripes {
10272 for (column, field) in fields.iter().enumerate() {
10273 if coded_type(&field.ty)
10274 && dictionaries[column].is_some()
10275 && stripe.memberships.get(column).is_none()
10276 && !demoted.get(column).copied().unwrap_or(false)
10277 {
10278 return Err(invalid("string page has no code membership index"));
10279 }
10280 }
10281 }
10282 Ok(Table {
10283 name,
10284 fields,
10285 stripes,
10286 rows,
10287 dictionaries,
10288 dictionary_payloads,
10289 demoted,
10290 distincts,
10291 frequencies,
10292 pair_frequencies,
10293 frequency_texts,
10294 host_groups,
10295 clustering,
10296 generation,
10297 sections,
10298 constraints,
10299 })
10300}
10301
10302fn put_count(out: &mut Vec<u8>, count: usize) -> Result<()> {
10304 put_u16(out, u16::try_from(count).map_err(|_| invalid("too many constraints"))?);
10305 Ok(())
10306}
10307
10308fn put_columns(out: &mut Vec<u8>, columns: &[u16]) -> Result<()> {
10310 put_count(out, columns.len())?;
10311 for &column in columns {
10312 put_u16(out, column);
10313 }
10314 Ok(())
10315}
10316
10317fn columns_of(cur: &mut Cursor<'_>) -> Result<Vec<u16>> {
10319 let count = cur.u16()? as usize;
10320 (0..count).map(|_| cur.u16()).collect()
10321}
10322
10323fn put_bound(out: &mut Vec<u8>, bound: Option<&Bound>) -> Result<()> {
10325 bounds::put(out, bound)
10326}
10327
10328#[derive(Debug)]
10345struct Codes;
10346
10347impl chooser::Chooser for Codes {
10348 fn name(&self) -> &'static str {
10349 "codes"
10350 }
10351
10352 fn narrow_strings(
10353 &self,
10354 _values: &[&[u8]],
10355 offered: &[string::Kind],
10356 _depth: u8,
10357 ) -> Vec<string::Kind> {
10358 offered.to_vec()
10361 }
10362
10363 fn narrow_integers(
10364 &self,
10365 _values: &[i64],
10366 offered: &[integer::Kind],
10367 depth: u8,
10368 ) -> Vec<integer::Kind> {
10369 narrowed_to(Codes::keep(depth), offered)
10372 }
10373
10374 fn considers_integer(&self, kind: integer::Kind, depth: u8) -> bool {
10375 Codes::keep(depth).contains(&kind)
10376 }
10377}
10378
10379impl Codes {
10380 fn keep(depth: u8) -> &'static [integer::Kind] {
10381 if depth == 0 {
10382 &[integer::Kind::Constant, integer::Kind::Packed, integer::Kind::Rle]
10383 } else {
10384 &[integer::Kind::Constant, integer::Kind::Packed]
10385 }
10386 }
10387}
10388
10389fn narrowed_to(keep: &[integer::Kind], offered: &[integer::Kind]) -> Vec<integer::Kind> {
10397 let narrowed: Vec<integer::Kind> =
10398 offered.iter().copied().filter(|kind| keep.contains(kind)).collect();
10399 if narrowed.is_empty() { offered.to_vec() } else { narrowed }
10400}
10401
10402#[derive(Debug)]
10414struct Fixed;
10415
10416impl chooser::Chooser for Fixed {
10417 fn name(&self) -> &'static str {
10418 "fixed"
10419 }
10420
10421 fn narrow_strings(
10422 &self,
10423 _values: &[&[u8]],
10424 offered: &[string::Kind],
10425 _depth: u8,
10426 ) -> Vec<string::Kind> {
10427 offered.to_vec()
10428 }
10429
10430 fn narrow_integers(
10431 &self,
10432 _values: &[i64],
10433 offered: &[integer::Kind],
10434 depth: u8,
10435 ) -> Vec<integer::Kind> {
10436 narrowed_to(Fixed::keep(depth), offered)
10437 }
10438
10439 fn considers_integer(&self, kind: integer::Kind, depth: u8) -> bool {
10440 Fixed::keep(depth).contains(&kind)
10441 }
10442}
10443
10444impl Fixed {
10445 fn keep(depth: u8) -> &'static [integer::Kind] {
10446 if depth == 0 {
10447 &[
10448 integer::Kind::Constant,
10449 integer::Kind::Packed,
10450 integer::Kind::Delta,
10451 integer::Kind::Rle,
10452 integer::Kind::Sparse,
10453 integer::Kind::Strided,
10454 ]
10455 } else {
10456 &[integer::Kind::Constant, integer::Kind::Packed, integer::Kind::Delta]
10457 }
10458 }
10459}
10460
10461fn widened(data: &Data) -> Option<Vec<i64>> {
10468 match data {
10469 Data::Int8(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
10470 Data::UInt8(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
10471 Data::Int16(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
10472 Data::UInt16(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
10473 Data::Int32(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
10474 Data::UInt32(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
10475 Data::Int64(values) => Some(values.to_vec()),
10476 _ => None,
10477 }
10478}
10479
10480fn cascade(ty: &LogicalType, bytes: &[u8], rows: usize) -> Result<Data> {
10486 fn wanted<T: integer::Lane>(bytes: &[u8], rows: usize) -> Result<Vec<T>> {
10487 let values = integer::decode_as::<T>(bytes)
10488 .map_err(|error| invalid(&format!("page value is not of its type: {error}")))?;
10489 if values.len() != rows {
10490 return Err(invalid("cascade page holds the wrong number of rows"));
10491 }
10492 Ok(values)
10493 }
10494 Ok(match ty {
10495 LogicalType::TinyInt => Data::Int8(wanted::<i8>(bytes, rows)?.into()),
10496 LogicalType::UTinyInt => Data::UInt8(wanted::<u8>(bytes, rows)?.into()),
10497 LogicalType::SmallInt => Data::Int16(wanted::<i16>(bytes, rows)?.into()),
10498 LogicalType::USmallInt => Data::UInt16(wanted::<u16>(bytes, rows)?.into()),
10499 LogicalType::Integer | LogicalType::Date => Data::Int32(wanted::<i32>(bytes, rows)?.into()),
10500 LogicalType::UInteger => Data::UInt32(wanted::<u32>(bytes, rows)?.into()),
10501 LogicalType::BigInt
10502 | LogicalType::Timestamp
10503 | LogicalType::Time
10504 | LogicalType::TimeTz
10505 | LogicalType::TimestampTz
10506 | LogicalType::TimestampS
10507 | LogicalType::TimestampMs
10508 | LogicalType::TimestampNs => Data::Int64(wanted::<i64>(bytes, rows)?.into()),
10509 LogicalType::Decimal { .. } => match ty.physical() {
10512 PhysicalType::Int16 => Data::Int16(wanted::<i16>(bytes, rows)?.into()),
10513 PhysicalType::Int32 => Data::Int32(wanted::<i32>(bytes, rows)?.into()),
10514 PhysicalType::Int64 => Data::Int64(wanted::<i64>(bytes, rows)?.into()),
10515 _ => return Err(invalid("cascade codec belongs to a decimal that is not an integer")),
10516 },
10517 _ => return Err(invalid("cascade codec belongs to a page that is not integers")),
10518 })
10519}
10520
10521fn plain_width(ty: &LogicalType) -> Option<usize> {
10524 Some(match ty {
10525 LogicalType::TinyInt | LogicalType::UTinyInt => 1,
10526 LogicalType::SmallInt | LogicalType::USmallInt => 2,
10527 LogicalType::Integer | LogicalType::UInteger | LogicalType::Date => 4,
10528 LogicalType::BigInt
10529 | LogicalType::Timestamp
10530 | LogicalType::Time
10531 | LogicalType::TimeTz
10532 | LogicalType::TimestampTz
10533 | LogicalType::TimestampS
10534 | LogicalType::TimestampMs
10535 | LogicalType::TimestampNs => 8,
10536 LogicalType::Decimal { .. } => match ty.physical() {
10537 PhysicalType::Int16 => 2,
10538 PhysicalType::Int32 => 4,
10539 PhysicalType::Int64 => 8,
10540 _ => return None,
10543 },
10544 _ => return None,
10545 })
10546}
10547
10548fn cascaded(
10554 flat: &Vector,
10555 ty: &LogicalType,
10556 packed: Option<&Packed<'_>>,
10557 settling: &mut Settling,
10558) -> Result<Option<Vec<u8>>> {
10559 let (Some(width), Some(data)) = (plain_width(ty), flat.data()) else { return Ok(None) };
10560 let Some(values) = widened(data) else { return Ok(None) };
10561 let plain = values.len().saturating_mul(width);
10562 let best = match packed {
10563 Some(packed) => plain.min(21 + size_of_val(packed.words())),
10565 None => plain,
10566 };
10567 let out = settling.encode(&values)?;
10568 Ok((out.len() < best).then_some(out))
10569}
10570
10571const SEARCH_EVERY: usize = 16;
10578
10579#[derive(Debug, Default)]
10585struct Settling {
10586 shape: Option<Shape>,
10589 since: usize,
10591 symbols: Option<Symbols>,
10593}
10594
10595#[derive(Debug)]
10598struct Symbols {
10599 shape: chooser::Settled,
10600 len: usize,
10603 payload: usize,
10604 since: usize,
10605}
10606
10607impl Settling {
10608 fn text(&mut self, values: &[&[u8]], payload: usize) -> Result<Option<Vec<u8>>> {
10616 if let Some(symbols) = self.symbols.as_mut().filter(|symbols| symbols.since < SEARCH_EVERY)
10617 {
10618 let out = string::encode_fsst(values, &symbols.shape)?;
10619 let held = match &out {
10621 None => symbols.len == 0,
10622 Some(out) => {
10623 (out.len() as u128) * (symbols.payload as u128) * 4
10624 <= (symbols.len as u128) * (payload as u128) * 5
10625 }
10626 };
10627 if held {
10628 symbols.since += 1;
10629 return Ok(out);
10630 }
10631 }
10632 let shape = string::fsst_shape(values);
10633 let out = string::encode_fsst(values, &shape)?;
10634 let len = out.as_ref().map_or(0, Vec::len);
10635 self.symbols = Some(Symbols { shape, len, payload: payload.max(1), since: 0 });
10636 Ok(out)
10637 }
10638
10639 fn encode(&mut self, values: &[i64]) -> Result<Vec<u8>> {
10646 if let Some(shape) = self.shape.as_ref().filter(|_| self.since < SEARCH_EVERY) {
10647 let replay = chooser::Replay::new(&shape.kinds, &Fixed).expecting(&shape.offered);
10648 let out = integer::encode_with(values, &replay)?;
10649 if !replay.held() {
10650 self.settle(&out, values.len(), replay.first_offered())?;
10651 return Ok(out);
10652 }
10653 let grown = (out.len() as u128) * (shape.rows as u128) * 4;
10654 if grown <= (shape.len as u128) * (values.len() as u128) * 5 {
10655 self.since += 1;
10656 return Ok(out);
10657 }
10658 }
10659 let search = chooser::Replay::new(&[], &Fixed);
10661 let out = integer::encode_with(values, &search)?;
10662 self.settle(&out, values.len(), search.first_offered())?;
10663 Ok(out)
10664 }
10665
10666 fn settle(&mut self, out: &[u8], rows: usize, offered: Vec<integer::Kind>) -> Result<()> {
10667 let kinds = integer::shape(out)?;
10668 self.shape = Some(Shape { kinds, offered, len: out.len().max(1), rows: rows.max(1) });
10669 self.since = 0;
10670 Ok(())
10671 }
10672}
10673
10674#[derive(Debug)]
10676struct Shape {
10677 kinds: Vec<integer::Kind>,
10678 offered: Vec<integer::Kind>,
10679 len: usize,
10680 rows: usize,
10681}
10682
10683fn text_compressed(flat: &Vector, settling: &mut Settling) -> Result<Option<Vec<u8>>> {
10724 let mut values: Vec<&[u8]> = Vec::with_capacity(flat.len());
10725 let mut payload = 0_usize;
10726 for row in 0..flat.len() {
10727 let text = flat.bytes_at(row).unwrap_or(b"");
10730 payload = payload.saturating_add(text.len());
10731 values.push(text);
10732 }
10733 let plain = (flat.len() + 1).saturating_mul(4).saturating_add(payload);
10735 let Some(out) = settling.text(&values, payload)? else {
10736 return Ok(None);
10737 };
10738 Ok((out.len() < plain).then_some(out))
10739}
10740
10741fn encoded_codes(codes: &[u32]) -> Result<Option<Vec<u8>>> {
10742 let wide: Vec<i64> = codes.iter().map(|code| i64::from(*code)).collect();
10743 let coded = integer::encode_with(&wide, &Codes)?;
10744 let plain = codes.len().saturating_mul(size_of::<u32>());
10745 Ok((coded.len() < plain).then_some(coded))
10746}
10747
10748fn push_validity(out: &mut Vec<u8>, flat: &Vector) {
10751 let flag = match flat.validity() {
10752 Validity::AllValid => 0,
10753 Validity::AllInvalid => 1,
10754 Validity::Mask(_) => 2,
10755 };
10756 out.push(flag);
10757 if flag == 2 {
10758 for group in (0..flat.len()).step_by(8) {
10759 let mut bits = 0_u8;
10760 for bit in 0..8 {
10761 if group + bit < flat.len() && !flat.is_null_at(group + bit) {
10762 bits |= 1 << bit;
10763 }
10764 }
10765 out.push(bits);
10766 }
10767 }
10768}
10769
10770fn coded_page(codes: &[u32], validity: &[u8]) -> Result<Vec<u8>> {
10777 let coded = encoded_codes(codes)?;
10778 let mut out = Vec::with_capacity(
10779 1 + validity.len() + coded.as_ref().map_or(size_of_val(codes), Vec::len),
10780 );
10781 out.push(if coded.is_some() { 4 } else { 3 });
10782 out.extend_from_slice(validity);
10783 match coded {
10784 Some(coded) => out.extend_from_slice(&coded),
10785 None => {
10786 for &code in codes {
10787 put_u32(&mut out, code);
10788 }
10789 }
10790 }
10791 Ok(out)
10792}
10793
10794fn encode(vector: &Vector, settling: &mut Settling) -> Result<Vec<u8>> {
10797 let ty = vector.logical_type();
10798 let flat = vector.flatten()?;
10800 let mut out = Vec::new();
10801 let dictionary = if coded_type(ty) { string_dictionary(&flat)? } else { None };
10802 let compressed_text = if dictionary.is_none() && coded_type(ty) {
10803 text_compressed(&flat, settling)?
10804 } else {
10805 None
10806 };
10807 let packed_vector = if dictionary.is_none() { Some(flat.bit_packed()?) } else { None };
10808 let packed = packed_vector.as_ref().and_then(Vector::packed_parts);
10809 let cascade =
10813 if dictionary.is_none() { cascaded(&flat, ty, packed.as_ref(), settling)? } else { None };
10814 out.push(if cascade.is_some() {
10815 5
10816 } else if dictionary.is_some() {
10817 1
10818 } else if compressed_text.is_some() {
10819 6
10820 } else if packed.is_some() {
10821 2
10822 } else {
10823 0
10824 });
10825 push_validity(&mut out, &flat);
10826 if let Some(cascade) = cascade {
10827 out.extend_from_slice(&cascade);
10828 return Ok(out);
10829 }
10830 if let Some(dictionary) = dictionary {
10831 out.extend_from_slice(&dictionary);
10832 return Ok(out);
10833 }
10834 if let Some(compressed_text) = compressed_text {
10835 out.extend_from_slice(&compressed_text);
10836 return Ok(out);
10837 }
10838 if let Some(packed) = packed {
10839 if packed.offset() != 0 {
10840 return Err(invalid("writer received a sliced packed vector"));
10841 }
10842 out.push(u8::try_from(packed.width()).map_err(|_| invalid("packed width overflow"))?);
10843 out.extend_from_slice(&packed.base().to_le_bytes());
10844 put_u32(
10845 &mut out,
10846 u32::try_from(packed.words().len()).map_err(|_| invalid("too many packed words"))?,
10847 );
10848 for word in packed.words() {
10849 put_u64(&mut out, *word);
10850 }
10851 return Ok(out);
10852 }
10853 let data = flat.data().ok_or_else(|| invalid("scalar column did not flatten"))?;
10854 match (ty, data) {
10855 (LogicalType::TinyInt, Data::Int8(values)) => {
10856 for value in &**values {
10857 out.extend_from_slice(&value.to_le_bytes());
10858 }
10859 }
10860 (LogicalType::UTinyInt, Data::UInt8(values)) => {
10861 for value in &**values {
10862 out.extend_from_slice(&value.to_le_bytes());
10863 }
10864 }
10865 (LogicalType::SmallInt, Data::Int16(values)) => {
10866 for value in &**values {
10867 out.extend_from_slice(&value.to_le_bytes());
10868 }
10869 }
10870 (LogicalType::USmallInt, Data::UInt16(values)) => {
10871 for value in &**values {
10872 out.extend_from_slice(&value.to_le_bytes());
10873 }
10874 }
10875 (LogicalType::UInteger, Data::UInt32(values)) => {
10876 for value in &**values {
10877 out.extend_from_slice(&value.to_le_bytes());
10878 }
10879 }
10880 (LogicalType::UBigInt, Data::UInt64(values)) => {
10881 for value in &**values {
10882 out.extend_from_slice(&value.to_le_bytes());
10883 }
10884 }
10885 (LogicalType::Integer | LogicalType::Date, Data::Int32(values)) => {
10886 for value in &**values {
10887 out.extend_from_slice(&value.to_le_bytes());
10888 }
10889 }
10890 (
10891 LogicalType::BigInt
10892 | LogicalType::Timestamp
10893 | LogicalType::Time
10894 | LogicalType::TimeTz
10895 | LogicalType::TimestampTz
10896 | LogicalType::TimestampS
10897 | LogicalType::TimestampMs
10898 | LogicalType::TimestampNs,
10899 Data::Int64(values),
10900 ) => {
10901 for value in &**values {
10902 out.extend_from_slice(&value.to_le_bytes());
10903 }
10904 }
10905 (LogicalType::HugeInt | LogicalType::Uuid, Data::Int128(values)) => {
10908 for value in &**values {
10909 out.extend_from_slice(&value.to_le_bytes());
10910 }
10911 }
10912 (LogicalType::UHugeInt, Data::UInt128(values)) => {
10913 for value in &**values {
10914 out.extend_from_slice(&value.to_le_bytes());
10915 }
10916 }
10917 (LogicalType::Float, Data::Float32(values)) => {
10920 for value in &**values {
10921 out.extend_from_slice(&value.to_le_bytes());
10922 }
10923 }
10924 (LogicalType::Double, Data::Float64(values)) => {
10925 for value in &**values {
10926 out.extend_from_slice(&value.to_le_bytes());
10927 }
10928 }
10929 (LogicalType::Interval, Data::Interval(values)) => {
10933 for (months, days, micros) in &**values {
10934 out.extend_from_slice(&months.to_le_bytes());
10935 out.extend_from_slice(&days.to_le_bytes());
10936 out.extend_from_slice(µs.to_le_bytes());
10937 }
10938 }
10939 (LogicalType::Boolean, Data::Bool(values)) => {
10940 for value in &**values {
10941 out.push(u8::from(*value));
10942 }
10943 }
10944 (LogicalType::Decimal { .. }, Data::Int16(values)) => {
10947 for value in &**values {
10948 out.extend_from_slice(&value.to_le_bytes());
10949 }
10950 }
10951 (LogicalType::Decimal { .. }, Data::Int32(values)) => {
10952 for value in &**values {
10953 out.extend_from_slice(&value.to_le_bytes());
10954 }
10955 }
10956 (LogicalType::Decimal { .. }, Data::Int64(values)) => {
10957 for value in &**values {
10958 out.extend_from_slice(&value.to_le_bytes());
10959 }
10960 }
10961 (LogicalType::Decimal { .. }, Data::Int128(values)) => {
10962 for value in &**values {
10963 out.extend_from_slice(&value.to_le_bytes());
10964 }
10965 }
10966 (LogicalType::Varchar | LogicalType::Blob | LogicalType::Bit, Data::Varlen(values)) => {
10971 let mut bytes = Vec::new();
10972 put_u32(&mut out, 0);
10973 for row in 0..vector.len() {
10974 let value = values.bytes(row).ok_or_else(|| invalid("string view is invalid"))?;
10975 bytes.extend_from_slice(value);
10976 put_u32(
10977 &mut out,
10978 u32::try_from(bytes.len())
10979 .map_err(|_| invalid("string payload exceeds 4GiB"))?,
10980 );
10981 }
10982 out.extend_from_slice(&bytes);
10983 }
10984 _ => return Err(Error::not_implemented(format!("native page for {ty}"))),
10985 }
10986 Ok(out)
10987}
10988
10989fn put_varint(out: &mut Vec<u8>, mut value: u32) {
10990 while value >= 0x80 {
10991 out.push((value as u8 & 0x7f) | 0x80);
10992 value >>= 7;
10993 }
10994 out.push(value as u8);
10995}
10996
10997fn unique_codes(codes: &[u32]) -> Vec<u32> {
10999 let mut unique = codes.to_vec();
11000 unique.sort_unstable();
11001 unique.dedup();
11002 unique
11003}
11004
11005fn merged_codes(lists: Vec<Vec<u32>>) -> Vec<u32> {
11011 let mut lists = lists;
11012 while lists.len() > 1 {
11013 let mut next = Vec::with_capacity(lists.len().div_ceil(2));
11014 for pair in lists.chunks(2) {
11015 match pair {
11016 [left, right] => next.push(merged_pair(left, right)),
11017 [only] => next.push(only.clone()),
11018 _ => {}
11019 }
11020 }
11021 lists = next;
11022 }
11023 lists.pop().unwrap_or_default()
11024}
11025
11026fn merged_pair(left: &[u32], right: &[u32]) -> Vec<u32> {
11027 let mut out = Vec::with_capacity(left.len().saturating_add(right.len()));
11028 let mut at = 0;
11029 let mut to = 0;
11030 while at < left.len() && to < right.len() {
11031 match left[at].cmp(&right[to]) {
11032 Ordering::Less => {
11033 out.push(left[at]);
11034 at += 1;
11035 }
11036 Ordering::Greater => {
11037 out.push(right[to]);
11038 to += 1;
11039 }
11040 Ordering::Equal => {
11041 out.push(left[at]);
11042 at += 1;
11043 to += 1;
11044 }
11045 }
11046 }
11047 out.extend_from_slice(&left[at..]);
11048 out.extend_from_slice(&right[to..]);
11049 out
11050}
11051
11052fn merged_range(ranges: impl Iterator<Item = Range>) -> Range {
11057 let mut merged = Range::default();
11058 let mut first = true;
11059 for range in ranges {
11060 merged.nulls = merged.nulls.saturating_add(range.nulls);
11061 merged.sum = match (merged.sum.take(), range.sum) {
11065 (Some(held), Some(next)) if !first => held.checked_add(next),
11066 (_, next) if first => next,
11067 _ => None,
11068 };
11069 merged.exact = if first { range.exact } else { merged.exact && range.exact };
11070 if first {
11071 merged.low = range.low;
11072 merged.high = range.high;
11073 first = false;
11074 continue;
11075 }
11076 merged.low = match (merged.low.take(), range.low) {
11077 (Some(held), Some(next)) => Some(held.smaller(next)),
11078 _ => None,
11079 };
11080 merged.high = match (merged.high.take(), range.high) {
11081 (Some(held), Some(next)) => Some(held.larger(next)),
11082 _ => None,
11083 };
11084 }
11085 merged
11086}
11087
11088fn shortened(bound: Option<Bound>, high: bool) -> Option<Bound> {
11101 match bound {
11102 Some(Bound::Bytes(mut value)) if value.len() > PART_BOUND_BYTES => {
11103 value.truncate(PART_BOUND_BYTES);
11104 if !high {
11105 return Some(Bound::Bytes(value));
11106 }
11107 while let Some(last) = value.pop() {
11108 if last < u8::MAX {
11109 value.push(last + 1);
11110 return Some(Bound::Bytes(value));
11111 }
11112 }
11113 None
11114 }
11115 other => other,
11116 }
11117}
11118
11119fn encode_part_ranges(ranges: &[Range]) -> Result<Vec<u8>> {
11127 let mut out = Vec::new();
11128 put_u32(
11129 &mut out,
11130 u32::try_from(ranges.len()).map_err(|_| invalid("too many parts in a stripe"))?,
11131 );
11132 for range in ranges {
11133 put_bound(&mut out, shortened(range.low.clone(), false).as_ref())?;
11134 put_bound(&mut out, shortened(range.high.clone(), true).as_ref())?;
11135 put_u32(&mut out, u32::try_from(range.nulls).map_err(|_| invalid("null count overflow"))?);
11136 }
11137 Ok(out)
11138}
11139
11140fn decode_part_ranges(bytes: &[u8]) -> Result<Vec<Range>> {
11142 let mut cur = Cursor::new(bytes);
11143 let parts = cur.u32()? as usize;
11144 let mut out = Vec::new();
11145 for _ in 0..parts {
11146 let low = cur.bound()?;
11147 let high = cur.bound()?;
11148 let nulls = cur.u32()? as usize;
11149 out.push(Range { low, high, nulls, exact: false, sum: None });
11150 }
11151 Ok(out)
11152}
11153
11154fn encode_sieves<'a>(sieves: impl Iterator<Item = &'a Option<Sieve>>) -> Result<Vec<u8>> {
11155 let held: Vec<&Option<Sieve>> = sieves.collect();
11156 let mut out = Vec::new();
11157 put_u32(
11158 &mut out,
11159 u32::try_from(held.len()).map_err(|_| invalid("too many parts in a stripe"))?,
11160 );
11161 for sieve in &held {
11162 let length = sieve.as_ref().map_or(0, Sieve::len);
11163 put_u32(&mut out, u32::try_from(length).map_err(|_| invalid("sieve length overflow"))?);
11164 }
11165 for sieve in held.into_iter().flatten() {
11167 out.extend_from_slice(&sieve.to_bytes());
11168 }
11169 Ok(out)
11170}
11171
11172fn decode_sieves(bytes: &[u8]) -> Result<Vec<Option<Sieve>>> {
11178 let parts = u32::from_le_bytes(
11179 bytes
11180 .get(..4)
11181 .ok_or_else(|| invalid("sieve page is truncated"))?
11182 .try_into()
11183 .map_err(|_| invalid("sieve page is truncated"))?,
11184 ) as usize;
11185 let mut lengths = Vec::with_capacity(parts);
11186 for part in 0..parts {
11187 let at = 4 + part * 4;
11188 let field = bytes.get(at..at + 4).ok_or_else(|| invalid("sieve page is truncated"))?;
11189 lengths.push(u32::from_le_bytes(
11190 field.try_into().map_err(|_| invalid("sieve page is truncated"))?,
11191 ) as usize);
11192 }
11193 let mut at = 4 + parts * 4;
11194 let mut out = Vec::with_capacity(parts);
11195 for length in lengths {
11196 if length == 0 {
11197 out.push(None);
11198 continue;
11199 }
11200 let end = at.checked_add(length).ok_or_else(|| invalid("sieve page is truncated"))?;
11201 let field = bytes.get(at..end).ok_or_else(|| invalid("sieve page is truncated"))?;
11202 out.push(Sieve::from_bytes(field));
11203 at = end;
11204 }
11205 if at != bytes.len() {
11206 return Err(invalid("sieve page has trailing bytes"));
11207 }
11208 Ok(out)
11209}
11210
11211fn encode_membership(unique: &[u32]) -> Vec<u8> {
11217 let mut out = Vec::with_capacity(unique.len().saturating_mul(2).saturating_add(5));
11218 put_varint(&mut out, u32::try_from(unique.len()).unwrap_or(u32::MAX));
11219 let mut previous = 0;
11220 for (at, &code) in unique.iter().enumerate() {
11221 put_varint(&mut out, if at == 0 { code } else { code - previous });
11222 previous = code;
11223 }
11224 out
11225}
11226
11227fn take_varint(bytes: &[u8], at: &mut usize) -> Result<u32> {
11228 let mut value = 0_u32;
11229 for shift in (0..35).step_by(7) {
11230 let byte = *bytes.get(*at).ok_or_else(|| invalid("membership varint is truncated"))?;
11231 *at += 1;
11232 let part = u32::from(byte & 0x7f);
11233 if shift == 28 && part > 0x0f {
11234 return Err(invalid("membership varint overflow"));
11235 }
11236 value = value
11237 .checked_add(
11238 part.checked_shl(shift).ok_or_else(|| invalid("membership varint overflow"))?,
11239 )
11240 .ok_or_else(|| invalid("membership varint overflow"))?;
11241 if byte & 0x80 == 0 {
11242 return Ok(value);
11243 }
11244 }
11245 Err(invalid("membership varint is too long"))
11246}
11247
11248fn decode_membership(bytes: &[u8]) -> Result<Vec<u32>> {
11249 let mut at = 0;
11250 let count = take_varint(bytes, &mut at)? as usize;
11251 let mut codes = Vec::with_capacity(count);
11252 let mut previous = 0_u32;
11253 for index in 0..count {
11254 let delta = take_varint(bytes, &mut at)?;
11255 let code = if index == 0 {
11256 delta
11257 } else {
11258 previous.checked_add(delta).ok_or_else(|| invalid("membership code overflow"))?
11259 };
11260 if index > 0 && code <= previous {
11261 return Err(invalid("membership codes are not increasing"));
11262 }
11263 codes.push(code);
11264 previous = code;
11265 }
11266 if at != bytes.len() {
11267 return Err(invalid("membership page has trailing bytes"));
11268 }
11269 Ok(codes)
11270}
11271
11272fn string_dictionary(vector: &Vector) -> Result<Option<Vec<u8>>> {
11280 let mut by_text: HashMap<&[u8], u32, Spread> =
11281 HashMap::with_capacity_and_hasher(vector.len(), Spread);
11282 let mut values = Vec::new();
11283 let mut codes = Vec::with_capacity(vector.len());
11284 let mut plain_bytes = 0_usize;
11285 for row in 0..vector.len() {
11286 let text = vector.bytes_at(row).unwrap_or(b"");
11287 plain_bytes = plain_bytes.saturating_add(text.len());
11288 let code = match by_text.get(text) {
11289 Some(&code) => code,
11290 None => {
11291 let code = u32::try_from(values.len())
11292 .map_err(|_| invalid("too many dictionary values"))?;
11293 by_text.insert(text, code);
11294 values.push(text);
11295 code
11296 }
11297 };
11298 codes.push(code);
11299 }
11300 let dictionary_bytes = values.iter().map(|value| value.len()).sum::<usize>();
11301 let encoded = 8_usize
11302 .saturating_add((values.len() + 1).saturating_mul(4))
11303 .saturating_add(dictionary_bytes)
11304 .saturating_add(codes.len().saturating_mul(4));
11305 let plain = (vector.len() + 1).saturating_mul(4).saturating_add(plain_bytes);
11306 if encoded >= plain {
11307 return Ok(None);
11308 }
11309 let mut out = Vec::with_capacity(encoded);
11310 put_u32(
11311 &mut out,
11312 u32::try_from(values.len()).map_err(|_| invalid("too many dictionary values"))?,
11313 );
11314 put_u32(
11315 &mut out,
11316 u32::try_from(dictionary_bytes).map_err(|_| invalid("dictionary payload exceeds 4GiB"))?,
11317 );
11318 let mut offset = 0_u32;
11319 put_u32(&mut out, offset);
11320 for value in &values {
11321 offset = offset
11322 .checked_add(
11323 u32::try_from(value.len()).map_err(|_| invalid("dictionary value is too long"))?,
11324 )
11325 .ok_or_else(|| invalid("dictionary payload exceeds 4GiB"))?;
11326 put_u32(&mut out, offset);
11327 }
11328 for value in values {
11329 out.extend_from_slice(value);
11330 }
11331 for code in codes {
11332 put_u32(&mut out, code);
11333 }
11334 Ok(Some(out))
11335}
11336
11337struct Room<'a, T> {
11339 state: &'a Mutex<(T, usize)>,
11340 finished: &'a Condvar,
11341 bytes: usize,
11342}
11343
11344impl<T> Drop for Room<'_, T> {
11345 fn drop(&mut self) {
11346 let mut held = self.state.lock().unwrap_or_else(PoisonError::into_inner);
11347 held.1 -= self.bytes;
11348 drop(held);
11349 self.finished.notify_all();
11350 }
11351}
11352
11353enum Closing<'a> {
11355 Numeric {
11358 column: usize,
11359 counted: bool,
11360 dense: Option<(u64, usize)>,
11361 },
11362 Dictionary {
11363 index: usize,
11364 dictionary: &'a GlobalDictionary,
11365 },
11366}
11367
11368enum Closed {
11370 Numeric(usize, (Option<FrequencySummary>, Option<u64>)),
11371 Dictionary(usize, ClosedDictionary),
11372}
11373
11374struct ClosedDictionary {
11376 distinct: Option<u64>,
11378 frequencies: Option<FrequencySummary>,
11379 texts: Vec<Option<Vec<u8>>>,
11380 hosts: Option<host::HostSummary>,
11381 encoded: EncodedDictionary,
11382 payload: u64,
11384}
11385
11386struct EncodedDictionary {
11387 index: Vec<u8>,
11388 ranks: Vec<u8>,
11389 grams: Vec<u8>,
11390}
11391
11392fn sort_by_value<'a>(codes: &mut [u32], values: impl Fn(u32) -> &'a [u8]) {
11433 let mut work = vec![(0, codes.len(), 0)];
11434 let mut keyed: Vec<(u64, u8, u32)> = Vec::new();
11435 while let Some((from, to, depth)) = work.pop() {
11436 let part = &mut codes[from..to];
11437 keyed.clear();
11438 keyed.extend(part.iter().map(|&code| {
11439 let value = values(code);
11440 let rest = value.get(depth..).unwrap_or_default();
11441 (head(rest), rest.len().min(8) as u8, code)
11442 }));
11443 keyed.sort_unstable();
11444 for (slot, entry) in part.iter_mut().zip(keyed.iter()) {
11445 *slot = entry.2;
11446 }
11447 let mut start = 0;
11448 while start < keyed.len() {
11449 let (key, taken, _) = keyed[start];
11450 let mut end = start + 1;
11451 while end < keyed.len() && keyed[end].0 == key && keyed[end].1 == taken {
11452 end += 1;
11453 }
11454 if taken == 8 && end - start > 1 {
11455 work.push((from + start, from + end, depth + 8));
11456 }
11457 start = end;
11458 }
11459 }
11460}
11461
11462const PARALLEL_SORT_MIN: usize = 1 << 16;
11464
11465const BUCKETS_PER_WORKER: usize = 4;
11468
11469const SAMPLES_PER_BUCKET: usize = 32;
11471
11472fn sort_by_value_across<'a>(
11490 codes: &mut [u32],
11491 values: impl Fn(u32) -> &'a [u8] + Sync,
11492 workers: usize,
11493) {
11494 if workers <= 1 || codes.len() < PARALLEL_SORT_MIN {
11495 sort_by_value(codes, values);
11496 return;
11497 }
11498 let buckets = workers * BUCKETS_PER_WORKER;
11499 let wanted = buckets * SAMPLES_PER_BUCKET;
11500 let mut sample = (0..wanted).map(|at| codes[at * codes.len() / wanted]).collect::<Vec<_>>();
11501 sort_by_value(&mut sample, &values);
11502 let splitters =
11503 (1..buckets).map(|cut| values(sample[cut * sample.len() / buckets])).collect::<Vec<_>>();
11504 let values = &values;
11505 let splitters = &splitters;
11506 let per = codes.len().div_ceil(workers);
11507 let places = std::thread::scope(|scope| {
11509 codes
11510 .chunks(per)
11511 .map(|run| {
11512 scope.spawn(move || {
11513 run.iter()
11514 .map(|&code| {
11515 let value = values(code);
11516 splitters.partition_point(|splitter| *splitter <= value) as u32
11517 })
11518 .collect::<Vec<_>>()
11519 })
11520 })
11521 .collect::<Vec<_>>()
11522 .into_iter()
11523 .flat_map(|handle| {
11524 handle.join().unwrap_or_else(|panic| std::panic::resume_unwind(panic))
11525 })
11526 .collect::<Vec<_>>()
11527 });
11528 let mut starts = vec![0_usize; buckets + 1];
11529 for &place in &places {
11530 starts[place as usize + 1] += 1;
11531 }
11532 for bucket in 0..buckets {
11533 starts[bucket + 1] += starts[bucket];
11534 }
11535 let mut laid = vec![0_u32; codes.len()];
11536 let mut next = starts.clone();
11537 for (&code, &place) in codes.iter().zip(&places) {
11538 laid[next[place as usize]] = code;
11539 next[place as usize] += 1;
11540 }
11541 drop(places);
11542 let mut runs = Vec::with_capacity(buckets);
11543 let mut rest = laid.as_mut_slice();
11544 for bucket in 0..buckets {
11545 let (run, after) = rest.split_at_mut(starts[bucket + 1] - starts[bucket]);
11546 runs.push(run);
11547 rest = after;
11548 }
11549 runs.sort_by_key(|run| run.len());
11551 let queue = Mutex::new(runs);
11552 std::thread::scope(|scope| {
11553 for _ in 0..workers {
11554 scope.spawn(|| {
11555 loop {
11556 let taken = queue.lock().unwrap_or_else(PoisonError::into_inner).pop();
11557 let Some(run) = taken else { break };
11558 sort_by_value(run, values);
11559 }
11560 });
11561 }
11562 });
11563 codes.copy_from_slice(&laid);
11564}
11565
11566fn head(bytes: &[u8]) -> u64 {
11568 let mut word = [0; 8];
11569 let take = bytes.len().min(8);
11570 word[..take].copy_from_slice(&bytes[..take]);
11571 u64::from_be_bytes(word)
11572}
11573
11574fn encode_global_dictionary(
11585 dictionary: &GlobalDictionary,
11586 order: &[(u64, u32)],
11587 places: &[Placed],
11588 scattered: bool,
11589) -> Result<EncodedDictionary> {
11590 let values = dictionary.values();
11591 if order.len() != values {
11592 return Err(invalid("global dictionary order does not cover its values"));
11593 }
11594 let blocks = values.div_ceil(TEXT_PAYLOAD_VALUES);
11595 if places.len() != blocks {
11596 return Err(invalid("global dictionary payload is not the blocks it says it is"));
11597 }
11598 if dictionary.grams.len() != blocks {
11599 return Err(invalid("global dictionary signatures do not cover its blocks"));
11600 }
11601 let (ranks, rank_ends) = encode_ranks(order, code_width(values))?;
11602 let rank_blocks = values.div_ceil(TEXT_RANK_BLOCK);
11603 let offset_bits = offset_width(&dictionary.ends);
11604 let payload_words = if scattered { 3 } else { 2 };
11605 let index_len = DICTIONARY_HEADER
11606 .checked_add(offset_bytes(values, offset_bits))
11607 .and_then(|len| len.checked_add(blocks.checked_mul(payload_words * 8)?))
11608 .and_then(|len| len.checked_add(rank_blocks.checked_mul(16)?))
11609 .and_then(|len| len.checked_add(8))
11610 .ok_or_else(|| invalid("global dictionary index length overflow"))?;
11611 let mut index = Vec::with_capacity(index_len);
11612 put_u32(
11613 &mut index,
11614 u32::try_from(values).map_err(|_| invalid("global dictionary has too many values"))?,
11615 );
11616 put_u32(&mut index, TEXT_PAYLOAD_VALUES as u32);
11617 put_u32(
11618 &mut index,
11619 u32::try_from(blocks).map_err(|_| invalid("global dictionary has too many blocks"))?,
11620 );
11621 let flag = (if scattered { DICTIONARY_SCATTERED } else { 0 })
11622 | DICTIONARY_GRAMS
11623 | DICTIONARY_WIDE_GRAMS;
11624 put_u32(&mut index, offset_bits as u32 | flag);
11625 encode_offsets(&dictionary.ends, offset_bits, &mut index)?;
11626 let mut end = 0_u64;
11631 for place in places {
11632 if scattered {
11633 put_u64(&mut index, place.start);
11634 put_u64(&mut index, place.length);
11635 } else {
11636 end = end
11637 .checked_add(place.length)
11638 .ok_or_else(|| invalid("global dictionary payload overflow"))?;
11639 put_u64(&mut index, end);
11640 }
11641 }
11642 for place in places {
11643 put_u64(&mut index, place.hash);
11644 }
11645 if rank_ends.len() != rank_blocks {
11648 return Err(invalid("global dictionary order is not the blocks it says it is"));
11649 }
11650 for end in &rank_ends {
11651 put_u64(&mut index, *end);
11652 }
11653 let mut at = 0_usize;
11654 for end in &rank_ends {
11655 let end = usize::try_from(*end).map_err(|_| invalid("global dictionary order overflow"))?;
11656 put_u64(&mut index, checksum(&ranks[at..end]));
11657 at = end;
11658 }
11659 let gram_len = blocks
11660 .checked_mul(TEXT_GRAM_BYTES)
11661 .ok_or_else(|| invalid("global dictionary signature count overflow"))?;
11662 let mut grams = Vec::with_capacity(gram_len);
11663 for block in &dictionary.grams {
11664 grams.extend_from_slice(block);
11665 }
11666 put_u64(&mut index, checksum(&grams));
11667 if index.len() != index_len {
11668 return Err(invalid("global dictionary index is not the length it was laid out for"));
11669 }
11670 Ok(EncodedDictionary { index, ranks, grams })
11671}
11672
11673const PAYLOAD_SAMPLE_BLOCKS: usize = 8;
11680
11681fn payload_shapes() -> Vec<chooser::Settled> {
11707 let integers = vec![integer::Kind::Packed];
11708 [
11709 vec![string::Kind::Front, string::Kind::Lz],
11710 vec![string::Kind::Lz, string::Kind::Fsst],
11711 vec![string::Kind::Lz, string::Kind::Plain],
11712 vec![string::Kind::Fsst],
11713 vec![string::Kind::Plain],
11714 ]
11715 .into_iter()
11716 .map(|strings| chooser::Settled::new(strings, integers.clone()))
11717 .collect()
11718}
11719
11720fn synced(file: &dyn rudb_io::File, profile: Option<&LoadProfile>) -> Result<()> {
11727 let started = profile.map(|_| std::time::Instant::now());
11728 file.sync()?;
11729 if let (Some(profile), Some(started)) = (profile, started) {
11730 profile.waited(
11731 Stage::Publish,
11732 u64::try_from(started.elapsed().as_nanos()).unwrap_or(u64::MAX),
11733 );
11734 }
11735 Ok(())
11736}
11737
11738#[derive(Debug)]
11743pub(crate) struct Unencoded {
11744 column: usize,
11745 at: usize,
11746 ends: Vec<u32>,
11747 bytes: Vec<u8>,
11748 shape: chooser::Settled,
11749}
11750
11751impl Unencoded {
11752 pub(crate) fn encode(&self) -> Result<EncodedBlock> {
11754 let values = block_values(&self.ends, &self.bytes);
11755 Ok((string::encode_with(&values, &self.shape)?, block_grams(&values)))
11756 }
11757
11758 pub(crate) fn place(&self) -> (usize, usize) {
11760 (self.column, self.at)
11761 }
11762}
11763
11764pub(crate) type EncodedBlock = (Vec<u8>, Box<[u8; TEXT_GRAM_BYTES]>);
11768
11769fn block_grams(values: &[&[u8]]) -> Box<[u8; TEXT_GRAM_BYTES]> {
11771 let mut grams = Box::new([0_u8; TEXT_GRAM_BYTES]);
11772 for value in values {
11773 for gram in value.windows(4) {
11774 for bit in gram_bits(gram, TEXT_GRAM_BYTES) {
11775 grams[bit / 8] |= 1 << (bit % 8);
11776 }
11777 }
11778 }
11779 grams
11780}
11781
11782fn block_values<'a>(ends: &[u32], bytes: &'a [u8]) -> Vec<&'a [u8]> {
11784 let mut out = Vec::with_capacity(ends.len());
11785 let mut from = 0;
11786 for &to in ends {
11787 out.push(&bytes[from..to as usize]);
11788 from = to as usize;
11789 }
11790 out
11791}
11792
11793fn finish_dictionaries(dictionaries: &mut [Option<GlobalDictionary>]) -> Result<()> {
11800 for dictionary in dictionaries.iter_mut().flatten() {
11801 if !dictionary.early.is_empty() {
11802 return Err(Error::internal("a dictionary block handed out never came back"));
11803 }
11804 dictionary.seal_rest();
11805 dictionary.settle_rest()?;
11806 }
11807 encode_waiting(dictionaries)?;
11808 if dictionaries
11811 .iter()
11812 .flatten()
11813 .any(|dictionary| dictionary.encoded() != dictionary.values().div_ceil(TEXT_PAYLOAD_VALUES))
11814 {
11815 return Err(Error::internal("a dictionary block handed out never came back"));
11816 }
11817 Ok(())
11818}
11819
11820fn encode_waiting(dictionaries: &mut [Option<GlobalDictionary>]) -> Result<()> {
11823 let jobs = dictionaries
11824 .iter()
11825 .enumerate()
11826 .flat_map(|(column, held)| {
11827 (0..held.as_ref().map_or(0, |held| held.waiting.len())).map(move |at| (column, at))
11828 })
11829 .collect::<Vec<_>>();
11830 if jobs.is_empty() {
11831 return Ok(());
11832 }
11833 let one = |column: usize, at: usize| -> Result<(usize, usize, EncodedBlock)> {
11834 let held = dictionaries[column].as_ref().ok_or_else(|| Error::internal("no dictionary"))?;
11835 Ok((column, at, held.encode_waiting(at)?))
11836 };
11837 let workers = std::thread::available_parallelism()
11838 .map_or(1, usize::from)
11839 .min(MAX_FREQUENCY_WORKERS)
11840 .min(jobs.len());
11841 let made = if workers <= 1 {
11842 jobs.iter().map(|&(column, at)| one(column, at)).collect::<Result<Vec<_>>>()?
11843 } else {
11844 let next = AtomicUsize::new(0);
11845 let jobs = &jobs;
11846 let pieces = std::thread::scope(|scope| {
11847 (0..workers)
11848 .map(|_| {
11849 scope.spawn(|| {
11850 let mut mine = Vec::new();
11851 loop {
11852 let job = next.fetch_add(1, Atomic::Relaxed);
11853 let Some(&(column, at)) = jobs.get(job) else { break };
11854 mine.push(one(column, at)?);
11855 }
11856 Ok(mine)
11857 })
11858 })
11859 .collect::<Vec<_>>()
11860 .into_iter()
11861 .map(|handle| {
11862 handle
11863 .join()
11864 .map_err(|_| Error::internal("a dictionary encode worker panicked"))?
11865 })
11866 .collect::<Result<Vec<_>>>()
11867 })?;
11868 pieces.into_iter().flatten().collect()
11869 };
11870 let mut done: Vec<Vec<(usize, EncodedBlock)>> =
11871 (0..dictionaries.len()).map(|_| Vec::new()).collect();
11872 for (column, at, bytes) in made {
11873 done[column].push((at, bytes));
11874 }
11875 for (column, mut made) in done.into_iter().enumerate() {
11876 if made.is_empty() {
11877 continue;
11878 }
11879 let Some(held) = dictionaries[column].as_mut() else { continue };
11880 made.sort_by_key(|(at, _)| *at);
11881 let waiting = std::mem::take(&mut held.waiting);
11882 for ((at, _), (_, block)) in waiting.into_iter().zip(made) {
11883 if held.encoded() != at {
11884 return Err(Error::internal("a dictionary block was encoded out of order"));
11885 }
11886 held.push_block(block);
11887 }
11888 }
11889 Ok(())
11890}
11891
11892fn settle_shape(sample: &[Vec<&[u8]>]) -> Result<chooser::Settled> {
11902 let mut best: Option<(chooser::Settled, usize)> = None;
11903 for shape in payload_shapes() {
11904 let mut size = 0;
11905 for block in sample {
11906 size += string::encode_with(block, &shape)?.len();
11907 }
11908 if best.as_ref().is_none_or(|(_, smallest)| size < *smallest) {
11909 best = Some((shape, size));
11910 }
11911 }
11912 best.map(|(shape, _)| shape)
11913 .ok_or_else(|| invalid("no shape applies to a global dictionary payload"))
11914}
11915
11916fn encode_ranks(order: &[(u64, u32)], code_bits: usize) -> Result<(Vec<u8>, Vec<u64>)> {
11923 let mut out = Vec::with_capacity(order.len() * 4);
11924 let mut ends = Vec::with_capacity(order.len().div_ceil(TEXT_RANK_BLOCK));
11925 let mut heads = Vec::with_capacity(TEXT_RANK_BLOCK);
11926 let mut codes = Vec::with_capacity(TEXT_RANK_BLOCK);
11927 for block in order.chunks(TEXT_RANK_BLOCK) {
11928 let base = block.first().map_or(0, |&(head, _)| head);
11931 let span = block.last().map_or(0, |&(head, _)| head.wrapping_sub(base));
11932 let width = (u64::BITS - span.leading_zeros()) as usize;
11933 heads.clear();
11934 codes.clear();
11935 for &(head, code) in block {
11936 heads.push(head.wrapping_sub(base));
11937 codes.push(u64::from(code));
11938 }
11939 put_u64(&mut out, base);
11940 out.push(width as u8);
11941 bitpack::pack_tail(&heads, width, &mut out)
11942 .map_err(|_| invalid("global dictionary heads do not pack"))?;
11943 bitpack::pack_tail(&codes, code_bits, &mut out)
11944 .map_err(|_| invalid("global dictionary codes do not pack"))?;
11945 ends.push(out.len() as u64);
11946 }
11947 Ok((out, ends))
11948}
11949
11950fn open_global_dictionary(
11957 file: Arc<File>,
11958 page: Page,
11959 ty: &LogicalType,
11960 keep_budget: usize,
11961) -> Result<Vector> {
11962 if !coded_type(ty) {
11963 return Err(invalid("global dictionary belongs to a non-string column"));
11964 }
11965 let mut header = [0; DICTIONARY_HEADER];
11966 read_at(&file, page.offset, &mut header)?;
11967 let count = u32::from_le_bytes(header[0..4].try_into().expect("four bytes")) as usize;
11968 let per_block = u32::from_le_bytes(header[4..8].try_into().expect("four bytes")) as usize;
11969 let blocks = u32::from_le_bytes(header[8..12].try_into().expect("four bytes")) as usize;
11970 let width = u32::from_le_bytes(header[12..16].try_into().expect("four bytes"));
11971 let scattered = width & DICTIONARY_SCATTERED != 0;
11972 let has_grams = width & DICTIONARY_GRAMS != 0;
11973 let gram_width =
11974 if width & DICTIONARY_WIDE_GRAMS != 0 { TEXT_GRAM_BYTES } else { NARROW_GRAM_BYTES };
11975 let offset_bits = (width & !DICTIONARY_FLAGS) as usize;
11976 if per_block != TEXT_PAYLOAD_VALUES {
11977 return Err(invalid("global dictionary block width differs"));
11978 }
11979 if blocks != count.div_ceil(TEXT_PAYLOAD_VALUES) {
11980 return Err(invalid("global dictionary block count differs from its value count"));
11981 }
11982 if offset_bits > u32::BITS as usize {
11983 return Err(invalid("global dictionary packs offsets past a payload"));
11984 }
11985 let offset_len = offset_bytes(count, offset_bits);
11986 let ranks = count;
11991 let rank_blocks = ranks.div_ceil(TEXT_RANK_BLOCK);
11992 let payload_words = if scattered { 3 } else { 2 };
11996 let hash_len = blocks
11997 .checked_mul(payload_words * 8)
11998 .and_then(|len| len.checked_add(rank_blocks.checked_mul(16)?))
11999 .and_then(|len| len.checked_add(usize::from(has_grams) * 8))
12000 .ok_or_else(|| invalid("global dictionary block count overflow"))?;
12001 let gram_len = if has_grams {
12002 blocks
12003 .checked_mul(gram_width)
12004 .ok_or_else(|| invalid("global dictionary signature count overflow"))?
12005 } else {
12006 0
12007 };
12008 let index_len = DICTIONARY_HEADER
12009 .checked_add(offset_len)
12010 .and_then(|len| len.checked_add(hash_len))
12011 .ok_or_else(|| invalid("global dictionary header overflow"))?;
12012 if index_len > page.length as usize {
12013 return Err(invalid("global dictionary offset index exceeds its page"));
12014 }
12015 let mut index = vec![0; index_len];
12016 index[..DICTIONARY_HEADER].copy_from_slice(&header);
12017 read_at(&file, page.offset + DICTIONARY_HEADER as u64, &mut index[DICTIONARY_HEADER..])?;
12018 if checksum(&index) != page.hash {
12019 return Err(invalid("global dictionary index checksum differs"));
12020 }
12021 let word_end = index_len - usize::from(has_grams) * 8;
12022 let gram_hash = has_grams
12023 .then(|| u64::from_le_bytes(index[word_end..index_len].try_into().expect("eight bytes")));
12024 let mut words = index[DICTIONARY_HEADER + offset_len..word_end]
12025 .chunks_exact(8)
12026 .map(|part| u64::from_le_bytes(part.try_into().expect("eight bytes")))
12027 .collect::<Vec<_>>();
12028 let mut rest = words.split_off(blocks * payload_words);
12029 let rank_hashes = rest.split_off(rank_blocks);
12030 let rank_ends = rest;
12031 if rank_ends.windows(2).any(|pair| pair[0] >= pair[1]) {
12034 return Err(invalid("global dictionary order blocks do not rise"));
12035 }
12036 let rank_len = usize::try_from(rank_ends.last().copied().unwrap_or_default())
12037 .map_err(|_| invalid("global dictionary rank overflow"))?;
12038 let body_len = index_len
12039 .checked_add(rank_len)
12040 .ok_or_else(|| invalid("global dictionary header overflow"))?;
12041 if body_len > page.length as usize {
12042 return Err(invalid("global dictionary order exceeds its page"));
12043 }
12044 let gram_end = body_len
12045 .checked_add(gram_len)
12046 .ok_or_else(|| invalid("global dictionary signature length overflow"))?;
12047 if gram_end > page.length as usize {
12048 return Err(invalid("global dictionary signatures exceed their page"));
12049 }
12050 let grams = gram_hash.map(|hash| NativeGrams {
12051 start: page.offset + body_len as u64,
12052 length: gram_len,
12053 width: gram_width,
12054 hash,
12055 verdicts: Mutex::new(Vec::new()),
12056 });
12057 let mut offsets = index;
12061 offsets.truncate(DICTIONARY_HEADER + offset_len);
12062 let hashes = words.split_off(blocks * (payload_words - 1));
12063 let (starts, lengths) = if scattered {
12064 let mut starts = Vec::with_capacity(blocks);
12065 let mut lengths = Vec::with_capacity(blocks);
12066 for pair in words.chunks_exact(2) {
12067 starts.push(pair[0]);
12068 lengths.push(pair[1]);
12069 }
12070 (starts, lengths)
12071 } else {
12072 let base = page.offset + gram_end as u64;
12076 let mut starts = Vec::with_capacity(blocks);
12077 let mut lengths = Vec::with_capacity(blocks);
12078 let mut at = 0_u64;
12079 for &end in &words {
12080 let len = end
12081 .checked_sub(at)
12082 .ok_or_else(|| invalid("global dictionary block ends before it starts"))?;
12083 starts.push(base + at);
12084 lengths.push(len);
12085 at = end;
12086 }
12087 (starts, lengths)
12088 };
12089 let stored_len = page.length as u64 - gram_end as u64;
12095 if scattered && stored_len == 0 {
12096 let size = file.metadata().map_err(io)?.len();
12097 let inside = starts.iter().zip(&lengths).all(|(&start, &len)| {
12098 start >= HEADER && start.checked_add(len).is_some_and(|end| end <= size)
12099 });
12100 if !inside {
12101 return Err(invalid("global dictionary block lies outside the file"));
12102 }
12103 } else if lengths.iter().try_fold(0_u64, |sum, len| sum.checked_add(*len)) != Some(stored_len) {
12104 return Err(invalid("global dictionary blocks do not bound the payload"));
12105 }
12106 Vector::external_text(
12107 ty.clone(),
12108 Arc::new(NativeText {
12109 file,
12110 values: count,
12111 offsets,
12112 offset_bits,
12113 value_ends: OnceLock::new(),
12114 value_lens: OnceLock::new(),
12115 ends_asked: AtomicUsize::new(0),
12116 ranks,
12117 rank_at: page.offset + index_len as u64,
12118 rank_ends,
12119 rank_hashes,
12120 rank_blocks: (0..rank_blocks).map(|_| OnceLock::new()).collect(),
12121 code_bits: code_width(count),
12122 code_ranks: OnceLock::new(),
12123 starts,
12124 lengths,
12125 hashes,
12126 grams,
12127 blocks: (0..blocks).map(|_| OnceLock::new()).collect(),
12128 char_lens: (0..blocks).map(|_| OnceLock::new()).collect(),
12129 keep_budget,
12130 payload_kept: AtomicUsize::new(0),
12131 swept: (0..blocks).map(|_| AtomicBool::new(false)).collect(),
12132 visit_dropped: AtomicUsize::new(0),
12133 searched: Mutex::new(HashMap::new()),
12134 }),
12135 )
12136}
12137
12138fn page_encoding(ty: &LogicalType, rows: usize, bytes: &[u8]) -> String {
12151 fn cascade_at(rows: usize, bytes: &[u8]) -> Result<(u8, usize)> {
12153 let mut cur = Cursor::new(bytes);
12154 let codec = cur.u8()?;
12155 if cur.u8()? == 2 {
12156 cur.take(rows.div_ceil(8))?;
12157 }
12158 Ok((codec, cur.at))
12159 }
12160 let Ok((codec, at)) = cascade_at(rows, bytes) else {
12161 return "UNREADABLE".to_string();
12162 };
12163 let tail = &bytes[at..];
12164 let described = |described: Result<String>| described.unwrap_or_else(|_| "UNREADABLE".into());
12165 match codec {
12166 0 => match ty {
12167 LogicalType::Varchar | LogicalType::Blob => "PLAIN".to_string(),
12168 _ => "FIXED".to_string(),
12169 },
12170 1 => "DICT(PLAIN)".to_string(),
12171 2 => "FOR+BITPACK".to_string(),
12172 3 => "TABLE DICT".to_string(),
12173 4 => format!("TABLE DICT({})", described(integer::describe(tail))),
12174 5 => described(integer::describe(tail)),
12175 6 => described(string::describe(tail)),
12176 other => format!("CODEC {other}"),
12177 }
12178}
12179
12180fn decode_selected_stable_codes(
12185 rows: usize,
12186 bytes: &[u8],
12187 positions: &[usize],
12188 out: &mut Vec<Option<u32>>,
12189) -> Result<bool> {
12190 if positions.windows(2).any(|pair| pair[0] >= pair[1])
12191 || positions.last().is_some_and(|&position| position >= rows)
12192 {
12193 return Err(invalid("selected code positions are not sorted and in range"));
12194 }
12195 let mut cur = Cursor::new(bytes);
12196 let codec = cur.u8()?;
12197 if codec != 3 && codec != 4 {
12198 return Ok(false);
12199 }
12200 let flag = cur.u8()?;
12201 let mask = match flag {
12202 0 | 1 => None,
12203 2 => {
12204 let at = cur.at;
12205 let len = rows.div_ceil(8);
12206 cur.take(len)?;
12207 Some((at, len))
12208 }
12209 _ => return Err(invalid("page validity tag differs")),
12210 };
12211 let valid = |row: usize| match flag {
12212 0 => true,
12213 1 => false,
12214 2 => mask.is_some_and(|(at, _)| bytes[at + row / 8] >> (row % 8) & 1 == 1),
12215 _ => unreachable!("the validity tag was checked"),
12216 };
12217 if codec == 4 {
12218 let wide = integer::decode_selected(&bytes[cur.at..], positions)?;
12219 for (&row, code) in positions.iter().zip(wide) {
12220 let code = u32::try_from(code).map_err(|_| invalid("code is not a code"))?;
12221 out.push(valid(row).then_some(code));
12222 }
12223 return Ok(true);
12224 }
12225 let codes_at = cur.at;
12226 let codes_len = rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?;
12227 cur.take(codes_len)?;
12228 if cur.at != bytes.len() {
12229 return Err(invalid("global code page has trailing bytes"));
12230 }
12231 let codes = &bytes[codes_at..codes_at + codes_len];
12232 for &row in positions {
12233 let at = row.checked_mul(4).ok_or_else(|| invalid("dictionary code offset overflow"))?;
12234 let code = u32::from_le_bytes(
12235 codes[at..at + 4].try_into().map_err(|_| invalid("dictionary code is truncated"))?,
12236 );
12237 out.push(valid(row).then_some(code));
12238 }
12239 Ok(true)
12240}
12241
12242fn decode_at(
12248 ty: &LogicalType,
12249 rows: usize,
12250 bytes: &[u8],
12251 global: Option<Arc<Vector>>,
12252 positions: &[u32],
12253) -> Result<Vector> {
12254 if positions.last().is_some_and(|&last| last as usize >= rows) {
12255 return Err(invalid("a position is past the end of the part"));
12256 }
12257 if bytes.first() == Some(&5)
12260 && positions.len().saturating_mul(8) <= rows
12261 && bytes
12263 .get(2 + if bytes.get(1) == Some(&2) { rows.div_ceil(8) } else { 0 }..)
12264 .is_some_and(integer::pointed)
12265 {
12266 return cascade_at(ty, rows, bytes, positions);
12267 }
12268 if bytes.first() != Some(&6) {
12269 return decode(ty, rows, bytes, global)?.gather(positions);
12270 }
12271 if !coded_type(ty) {
12272 return Err(invalid("compressed text codec belongs to a non-string page"));
12273 }
12274 let mut cur = Cursor::new(bytes);
12275 cur.u8()?;
12276 let validity = match cur.u8()? {
12277 0 => Validity::AllValid,
12278 1 => Validity::AllInvalid,
12279 2 => {
12280 let mask = cur.take(rows.div_ceil(8))?;
12281 Validity::from_iter(positions.len(), |at| {
12282 let row = positions[at] as usize;
12283 mask[row / 8] >> (row % 8) & 1 == 1
12284 })
12285 }
12286 _ => return Err(invalid("page validity tag differs")),
12287 };
12288 let (payload, ends) = string::decode_flat_at(&bytes[cur.at..], positions)?.into_parts();
12289 let mut values = StringColumn::over(Buffer::from_vec(payload).into_page());
12290 push_values(&mut values, ty, &ends)?;
12291 Ok(Vector::flat(ty.clone(), Data::Varlen(values))?.with_validity(validity))
12292}
12293
12294fn cascade_at(ty: &LogicalType, rows: usize, bytes: &[u8], positions: &[u32]) -> Result<Vector> {
12300 fn wanted<T: integer::Lane>(values: &[i64]) -> Result<Vec<T>> {
12301 values
12302 .iter()
12303 .map(|&value| T::fit(value).ok_or_else(|| invalid("page value is not of its type")))
12304 .collect()
12305 }
12306 let mut cur = Cursor::new(bytes);
12307 cur.u8()?;
12308 let validity = match cur.u8()? {
12309 0 => Validity::AllValid,
12310 1 => Validity::AllInvalid,
12311 2 => {
12312 let mask = cur.take(rows.div_ceil(8))?;
12313 Validity::from_iter(positions.len(), |at| {
12314 let row = positions[at] as usize;
12315 mask[row / 8] >> (row % 8) & 1 == 1
12316 })
12317 }
12318 _ => return Err(invalid("page validity tag differs")),
12319 };
12320 let at: Vec<usize> = positions.iter().map(|&row| row as usize).collect();
12321 let values = integer::decode_selected(&bytes[cur.at..], &at)
12322 .map_err(|error| invalid(&format!("page value is not of its type: {error}")))?;
12323 if values.len() != positions.len() {
12324 return Err(invalid("cascade page holds the wrong number of rows"));
12325 }
12326 let data = match ty {
12327 LogicalType::TinyInt => Data::Int8(wanted::<i8>(&values)?.into()),
12328 LogicalType::UTinyInt => Data::UInt8(wanted::<u8>(&values)?.into()),
12329 LogicalType::SmallInt => Data::Int16(wanted::<i16>(&values)?.into()),
12330 LogicalType::USmallInt => Data::UInt16(wanted::<u16>(&values)?.into()),
12331 LogicalType::Integer | LogicalType::Date => Data::Int32(wanted::<i32>(&values)?.into()),
12332 LogicalType::UInteger => Data::UInt32(wanted::<u32>(&values)?.into()),
12333 LogicalType::BigInt
12334 | LogicalType::Timestamp
12335 | LogicalType::Time
12336 | LogicalType::TimeTz
12337 | LogicalType::TimestampTz
12338 | LogicalType::TimestampS
12339 | LogicalType::TimestampMs
12340 | LogicalType::TimestampNs => Data::Int64(values.into()),
12341 LogicalType::Decimal { .. } => match ty.physical() {
12342 PhysicalType::Int16 => Data::Int16(wanted::<i16>(&values)?.into()),
12343 PhysicalType::Int32 => Data::Int32(wanted::<i32>(&values)?.into()),
12344 PhysicalType::Int64 => Data::Int64(values.into()),
12345 _ => return Err(invalid("cascade codec belongs to a decimal that is not an integer")),
12346 },
12347 _ => return Err(invalid("cascade codec belongs to a page that is not integers")),
12348 };
12349 Ok(Vector::flat(ty.clone(), data)?.with_validity(validity))
12350}
12351
12352fn push_values(values: &mut StringColumn, ty: &LogicalType, ends: &[usize]) -> Result<()> {
12356 if ty == &LogicalType::Varchar {
12357 return values.push_run_in_place(0, ends);
12358 }
12359 let mut start = 0;
12360 for &end in ends {
12361 let len = end
12362 .checked_sub(start)
12363 .ok_or_else(|| invalid("a string value ends before it starts"))?;
12364 values.push_bytes_in_place(start, len)?;
12365 start = end;
12366 }
12367 Ok(())
12368}
12369
12370fn decode(
12371 ty: &LogicalType,
12372 rows: usize,
12373 bytes: &[u8],
12374 global: Option<Arc<Vector>>,
12375) -> Result<Vector> {
12376 let mut cur = Cursor::new(bytes);
12377 let codec = cur.u8()?;
12378 let flag = cur.u8()?;
12379 let validity = match flag {
12380 0 => Validity::AllValid,
12381 1 => Validity::AllInvalid,
12382 2 => {
12383 let mask = cur.take(rows.div_ceil(8))?;
12384 Validity::from_iter(rows, |row| mask[row / 8] >> (row % 8) & 1 == 1)
12385 }
12386 _ => return Err(invalid("page validity tag differs")),
12387 };
12388 if codec == 1 {
12389 if !coded_type(ty) {
12390 return Err(invalid("dictionary codec belongs to a non-string page"));
12391 }
12392 let count = cur.u32()? as usize;
12393 let payload_len = cur.u32()? as usize;
12394 let offset_bytes = cur.take(
12395 (count + 1)
12396 .checked_mul(4)
12397 .ok_or_else(|| invalid("dictionary offset count overflow"))?,
12398 )?;
12399 let offsets = offset_bytes
12400 .chunks_exact(4)
12401 .map(|part| u32::from_le_bytes(part.try_into().expect("four bytes")))
12402 .collect::<Vec<_>>();
12403 let payload = cur.take(payload_len)?.to_vec();
12404 if offsets.first() != Some(&0)
12405 || offsets.last().copied().map(|last| last as usize) != Some(payload.len())
12406 || offsets.windows(2).any(|pair| pair[0] > pair[1])
12407 {
12408 return Err(invalid("dictionary offsets do not bound the payload"));
12409 }
12410 let mut strings = StringColumn::over(Buffer::from_vec(payload).into_page());
12413 let ends: Vec<usize> = offsets[1..].iter().map(|&end| end as usize).collect();
12414 push_values(&mut strings, ty, &ends)?;
12415 let mut codes = Vec::with_capacity(rows);
12416 for _ in 0..rows {
12417 codes.push(cur.u32()?);
12418 }
12419 if codes.iter().any(|code| *code as usize >= count) {
12420 return Err(invalid("dictionary code is out of range"));
12421 }
12422 if cur.at != bytes.len() {
12423 return Err(invalid("dictionary page has trailing bytes"));
12424 }
12425 let dictionary = Vector::flat(ty.clone(), Data::Varlen(strings))?;
12426 return Ok(Vector::dictionary(codes, dictionary)?.with_validity(validity));
12427 }
12428 if codec == 3 || codec == 4 {
12429 let dictionary = global.ok_or_else(|| invalid("global code page has no dictionary"))?;
12430 let codes = if codec == 4 {
12431 let codes = integer::decode_as::<u32>(&bytes[cur.at..])
12436 .map_err(|error| invalid(&format!("code is not a code: {error}")))?;
12437 if codes.len() != rows {
12438 return Err(invalid("encoded code page holds the wrong number of rows"));
12439 }
12440 codes
12441 } else {
12442 let mut codes = Vec::with_capacity(rows);
12443 for _ in 0..rows {
12444 codes.push(cur.u32()?);
12445 }
12446 if cur.at != bytes.len() {
12447 return Err(invalid("global code page has trailing bytes"));
12448 }
12449 codes
12450 };
12451 let highest = codes.iter().copied().max();
12452 return Ok(Vector::stable_dictionary_validated(codes, dictionary, highest)?
12453 .with_validity(validity));
12454 }
12455 if codec == 6 {
12456 if !coded_type(ty) {
12457 return Err(invalid("compressed text codec belongs to a non-string page"));
12458 }
12459 let (payload, ends) = string::decode_flat(&bytes[cur.at..])?.into_parts();
12463 if ends.len() != rows {
12464 return Err(invalid("compressed text page holds the wrong number of rows"));
12465 }
12466 let mut values = StringColumn::over(Buffer::from_vec(payload).into_page());
12469 push_values(&mut values, ty, &ends)?;
12470 return Ok(Vector::flat(ty.clone(), Data::Varlen(values))?.with_validity(validity));
12471 }
12472 if codec == 5 {
12473 let data = cascade(ty, &bytes[cur.at..], rows)?;
12475 return Ok(Vector::flat(ty.clone(), data)?.with_validity(validity));
12476 }
12477 if codec == 2 {
12478 let width = u32::from(cur.u8()?);
12479 let base = i128::from_le_bytes(cur.take(16)?.try_into().expect("sixteen bytes"));
12480 let count = cur.u32()? as usize;
12481 let length = count.checked_mul(8).ok_or_else(|| invalid("packed page is too long"))?;
12482 let words: Vec<u64> = cur
12483 .take(length)?
12484 .chunks_exact(8)
12485 .map(|word| u64::from_le_bytes(word.try_into().expect("eight bytes")))
12486 .collect();
12487 if cur.at != bytes.len() {
12488 return Err(invalid("packed page has trailing bytes"));
12489 }
12490 return Ok(Vector::packed(ty.clone(), words, width, base, rows)?.with_validity(validity));
12491 }
12492 if codec != 0 {
12493 return Err(invalid("page codec is unknown"));
12494 }
12495 let data = match ty {
12496 LogicalType::TinyInt => {
12497 let values = cur.take(rows)?;
12498 Data::Int8(values.iter().map(|item| *item as i8).collect::<Vec<_>>().into())
12499 }
12500 LogicalType::UTinyInt => Data::UInt8(cur.take(rows)?.to_vec().into()),
12501 LogicalType::SmallInt => {
12502 let values =
12503 cur.take(rows.checked_mul(2).ok_or_else(|| invalid("page size overflow"))?)?;
12504 Data::Int16(
12505 values
12506 .chunks_exact(2)
12507 .map(|item| i16::from_le_bytes(item.try_into().expect("two bytes")))
12508 .collect::<Vec<_>>()
12509 .into(),
12510 )
12511 }
12512 LogicalType::USmallInt => {
12513 let values =
12514 cur.take(rows.checked_mul(2).ok_or_else(|| invalid("page size overflow"))?)?;
12515 Data::UInt16(
12516 values
12517 .chunks_exact(2)
12518 .map(|item| u16::from_le_bytes(item.try_into().expect("two bytes")))
12519 .collect::<Vec<_>>()
12520 .into(),
12521 )
12522 }
12523 LogicalType::UInteger => {
12524 let values =
12525 cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
12526 Data::UInt32(
12527 values
12528 .chunks_exact(4)
12529 .map(|item| u32::from_le_bytes(item.try_into().expect("four bytes")))
12530 .collect::<Vec<_>>()
12531 .into(),
12532 )
12533 }
12534 LogicalType::UBigInt => {
12535 let values =
12536 cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
12537 Data::UInt64(
12538 values
12539 .chunks_exact(8)
12540 .map(|item| u64::from_le_bytes(item.try_into().expect("eight bytes")))
12541 .collect::<Vec<_>>()
12542 .into(),
12543 )
12544 }
12545 LogicalType::Integer | LogicalType::Date => {
12546 let values =
12547 cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
12548 Data::Int32(
12549 values
12550 .chunks_exact(4)
12551 .map(|item| i32::from_le_bytes(item.try_into().expect("four bytes")))
12552 .collect::<Vec<_>>()
12553 .into(),
12554 )
12555 }
12556 LogicalType::BigInt
12557 | LogicalType::Timestamp
12558 | LogicalType::Time
12559 | LogicalType::TimeTz
12560 | LogicalType::TimestampTz
12561 | LogicalType::TimestampS
12562 | LogicalType::TimestampMs
12563 | LogicalType::TimestampNs => {
12564 let values =
12565 cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
12566 Data::Int64(
12567 values
12568 .chunks_exact(8)
12569 .map(|item| i64::from_le_bytes(item.try_into().expect("eight bytes")))
12570 .collect::<Vec<_>>()
12571 .into(),
12572 )
12573 }
12574 LogicalType::HugeInt | LogicalType::Uuid => {
12575 let values =
12576 cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
12577 Data::Int128(
12578 values
12579 .chunks_exact(16)
12580 .map(|item| i128::from_le_bytes(item.try_into().expect("sixteen bytes")))
12581 .collect::<Vec<_>>()
12582 .into(),
12583 )
12584 }
12585 LogicalType::UHugeInt => {
12586 let values =
12587 cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
12588 Data::UInt128(
12589 values
12590 .chunks_exact(16)
12591 .map(|item| u128::from_le_bytes(item.try_into().expect("sixteen bytes")))
12592 .collect::<Vec<_>>()
12593 .into(),
12594 )
12595 }
12596 LogicalType::Float => {
12597 let values =
12598 cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
12599 Data::Float32(
12600 values
12601 .chunks_exact(4)
12602 .map(|item| f32::from_le_bytes(item.try_into().expect("four bytes")))
12603 .collect::<Vec<_>>()
12604 .into(),
12605 )
12606 }
12607 LogicalType::Double => {
12608 let values =
12609 cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
12610 Data::Float64(
12611 values
12612 .chunks_exact(8)
12613 .map(|item| f64::from_le_bytes(item.try_into().expect("eight bytes")))
12614 .collect::<Vec<_>>()
12615 .into(),
12616 )
12617 }
12618 LogicalType::Interval => {
12619 let values =
12620 cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
12621 Data::Interval(
12622 values
12623 .chunks_exact(16)
12624 .map(|item| {
12625 (
12626 i32::from_le_bytes(item[..4].try_into().expect("four bytes")),
12627 i32::from_le_bytes(item[4..8].try_into().expect("four bytes")),
12628 i64::from_le_bytes(item[8..].try_into().expect("eight bytes")),
12629 )
12630 })
12631 .collect::<Vec<_>>()
12632 .into(),
12633 )
12634 }
12635 LogicalType::Boolean => {
12636 let values = cur.take(rows)?;
12637 if values.iter().any(|value| *value > 1) {
12638 return Err(invalid("boolean page has another value"));
12639 }
12640 Data::Bool(values.iter().map(|value| *value == 1).collect::<Vec<_>>().into())
12641 }
12642 LogicalType::Decimal { .. } => match ty.physical() {
12645 PhysicalType::Int16 => {
12646 let values =
12647 cur.take(rows.checked_mul(2).ok_or_else(|| invalid("page size overflow"))?)?;
12648 Data::Int16(
12649 values
12650 .chunks_exact(2)
12651 .map(|item| i16::from_le_bytes(item.try_into().expect("two bytes")))
12652 .collect::<Vec<_>>()
12653 .into(),
12654 )
12655 }
12656 PhysicalType::Int32 => {
12657 let values =
12658 cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
12659 Data::Int32(
12660 values
12661 .chunks_exact(4)
12662 .map(|item| i32::from_le_bytes(item.try_into().expect("four bytes")))
12663 .collect::<Vec<_>>()
12664 .into(),
12665 )
12666 }
12667 PhysicalType::Int64 => {
12668 let values =
12669 cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
12670 Data::Int64(
12671 values
12672 .chunks_exact(8)
12673 .map(|item| i64::from_le_bytes(item.try_into().expect("eight bytes")))
12674 .collect::<Vec<_>>()
12675 .into(),
12676 )
12677 }
12678 _ => {
12679 let values =
12680 cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
12681 Data::Int128(
12682 values
12683 .chunks_exact(16)
12684 .map(|item| i128::from_le_bytes(item.try_into().expect("sixteen bytes")))
12685 .collect::<Vec<_>>()
12686 .into(),
12687 )
12688 }
12689 },
12690 LogicalType::Varchar | LogicalType::Blob | LogicalType::Bit => {
12691 let offset_bytes = cur
12692 .take((rows + 1).checked_mul(4).ok_or_else(|| invalid("offset count overflow"))?)?;
12693 let offsets = offset_bytes
12694 .chunks_exact(4)
12695 .map(|part| u32::from_le_bytes(part.try_into().expect("four bytes")))
12696 .collect::<Vec<_>>();
12697 let payload = cur.take(bytes.len() - cur.at)?.to_vec();
12698 if offsets.first() != Some(&0)
12699 || offsets.last().copied().map(|last| last as usize) != Some(payload.len())
12700 || offsets.windows(2).any(|pair| pair[0] > pair[1])
12701 {
12702 return Err(invalid("string offsets do not bound the payload"));
12703 }
12704 let mut values = StringColumn::over(Buffer::from_vec(payload).into_page());
12712 let ends: Vec<usize> = offsets[1..].iter().map(|&end| end as usize).collect();
12713 push_values(&mut values, ty, &ends)?;
12714 Data::Varlen(values)
12715 }
12716 _ => return Err(Error::not_implemented(format!("native page for {ty}"))),
12717 };
12718 if cur.at != bytes.len() {
12719 return Err(invalid("page has trailing bytes"));
12720 }
12721 Ok(Vector::flat(ty.clone(), data)?.with_validity(validity))
12722}
12723
12724#[cfg(test)]
12725mod tests {
12726 use std::fs::{self, OpenOptions};
12727 use std::io::{Seek, SeekFrom, Write};
12728 use std::path::PathBuf;
12729 use std::time::{SystemTime, UNIX_EPOCH};
12730
12731 use rudb_common::Stat;
12732 use rudb_common::Value;
12733 use rudb_common::bounds::{Frequencies, Op, Remainder, Zones};
12734 use rudb_common::stat::Provenance;
12735
12736 use super::*;
12737
12738 #[test]
12739 fn spanned_frequency_header_rejects_missing_or_out_of_bounds_payloads() {
12740 for (length, entries) in [(0_u32, 1_u32), (9, 0), (1, FREQUENCY_ENTRIES as u32 + 1)] {
12741 let mut bytes = Vec::new();
12742 put_u32(&mut bytes, length);
12743 put_u32(&mut bytes, entries);
12744 bytes.push(1);
12745 assert!(summary_span(&mut Cursor::new(&bytes)).is_err());
12746 }
12747 let mut bytes = Vec::new();
12748 put_u32(&mut bytes, 1);
12749 put_u32(&mut bytes, 0);
12750 bytes.push(1);
12751 assert_eq!(summary_span(&mut Cursor::new(&bytes)).expect("one byte"), Some((1, 0)));
12752 }
12753
12754 #[test]
12755 fn a_name_taken_in_pieces_is_the_name_of_the_pieces_joined() {
12756 let bytes: Vec<u8> =
12757 (0..300_u32).map(|at| (at.wrapping_mul(2_654_435_761) >> 13) as u8).collect();
12758 for length in [0, 1, 7, 31, 32, 33, 63, 64, 65, 100, 300] {
12759 let whole = content_name(&bytes[..length]);
12760 for step in [1, 3, 8, 31, 32, 33, 64, 301] {
12761 let mut namer = ContentNamer::default();
12762 bytes[..length].chunks(step).for_each(|piece| namer.update(piece));
12763 assert_eq!(namer.finish(), whole, "{length} bytes in pieces of {step}");
12764 }
12765 }
12766 }
12767
12768 #[derive(Debug)]
12771 struct TestsEverything<'a>(&'a dyn chooser::Chooser);
12772
12773 impl chooser::Chooser for TestsEverything<'_> {
12774 fn name(&self) -> &'static str {
12775 "tests everything"
12776 }
12777
12778 fn narrow_strings(
12779 &self,
12780 values: &[&[u8]],
12781 offered: &[string::Kind],
12782 depth: u8,
12783 ) -> Vec<string::Kind> {
12784 self.0.narrow_strings(values, offered, depth)
12785 }
12786
12787 fn narrow_integers(
12788 &self,
12789 values: &[i64],
12790 offered: &[integer::Kind],
12791 depth: u8,
12792 ) -> Vec<integer::Kind> {
12793 self.0.narrow_integers(values, offered, depth)
12794 }
12795 }
12796
12797 #[test]
12798 fn ruling_kinds_out_before_testing_for_them_writes_the_same_bytes() {
12799 let columns: Vec<Vec<i64>> = vec![
12800 vec![],
12801 vec![5; 1000],
12802 (0..1000).collect(),
12803 (0..1000).map(|row| 1_600_000_000_000_000 + row * 1_000_000).collect(),
12804 (0..1000).map(|row| row / 50).collect(),
12805 (0..1000).map(|row| if row % 97 == 0 { row } else { 0 }).collect(),
12806 (0..1000).map(|row| (row * 7919) % 13).collect(),
12807 (0..1000).map(|row| (row * 2_654_435_761) % 1_000_003).collect(),
12808 (0..1000).map(|row| [3, 3, 3, 9, 9, 1][row as usize % 6]).collect(),
12809 (0..1000).map(|row| i64::MIN + row % 3).collect(),
12810 ];
12811 let choosers: [&dyn chooser::Chooser; 2] = [&Fixed, &Codes];
12812 for column in &columns {
12813 for chooser in choosers {
12814 let quick = integer::encode_with(column, chooser).unwrap();
12815 let full = integer::encode_with(column, &TestsEverything(chooser)).unwrap();
12816 assert_eq!(
12817 quick,
12818 full,
12819 "{} on {:?}",
12820 chooser.name(),
12821 &column[..column.len().min(8)]
12822 );
12823 }
12824 }
12825 }
12826
12827 #[test]
12830 fn parts_that_look_alike_replay_to_the_bytes_a_search_writes() {
12831 let mut settling = Settling::default();
12832 for part in 0..STRIPE_PARTS as i64 {
12833 let values: Vec<i64> = (0..2048)
12834 .map(|row| 1_600_000_000_000_000 + (part * 2048 + row) * 1_000_000 + row % 7)
12835 .collect();
12836 let searched = integer::encode_with(&values, &Fixed).unwrap();
12837 assert_eq!(settling.encode(&values).unwrap(), searched, "part {part}");
12838 }
12839 }
12840
12841 #[test]
12845 fn text_pages_share_a_table_until_the_text_changes() {
12846 let words = ["carefully", "final", "deposits", "sleep", "furiously", "quickly", "among"];
12847 let english: Vec<Vec<u8>> = (0..1024)
12848 .map(|row: usize| {
12849 let pick = |at: usize| words[(row * 7 + at * 3) % words.len()];
12850 format!("{} {} {} the {}", pick(0), pick(1), pick(2), pick(3)).into_bytes()
12851 })
12852 .collect();
12853 let digits: Vec<Vec<u8>> =
12854 (0..1024_u64).map(|row| format!("{:020}", row * 7_919_993).into_bytes()).collect();
12855 let mut settling = Settling::default();
12856 for page in 0..8 {
12857 let values: Vec<&[u8]> =
12858 if page < 4 { &english } else { &digits }.iter().map(Vec::as_slice).collect();
12859 let payload = values.iter().map(|value| value.len()).sum();
12860 let out = settling.text(&values, payload).unwrap().unwrap();
12861 assert_eq!(string::decode(&out).unwrap(), values, "page {page}");
12862 let alone = string::encode_only(string::Kind::Fsst, &values).unwrap().unwrap();
12863 assert!(
12864 out.len() * 4 <= alone.len() * 5,
12865 "page {page}: {} against {}",
12866 out.len(),
12867 alone.len()
12868 );
12869 let since = settling.symbols.as_ref().unwrap().since;
12870 assert_eq!(since, page % 4, "page {page}");
12871 }
12872 }
12873
12874 #[test]
12878 fn a_column_that_changes_under_the_shape_is_searched_again() {
12879 let mut state = 0x9e37_79b9_7f4a_7c15_u64;
12880 let mut noise = move || {
12881 state ^= state << 13;
12882 state ^= state >> 7;
12883 state ^= state << 17;
12884 (state % 1_000_000) as i64
12885 };
12886 let mut settling = Settling::default();
12887 for part in 0..STRIPE_PARTS as i64 {
12888 let values: Vec<i64> = match part / 16 {
12889 0 => (0..2048).map(|row| (part * 2048 + row) / 300).collect(),
12890 1 => (0..2048).map(|_| noise()).collect(),
12891 2 => (0..2048).map(|row| if row % 97 == 0 { row } else { 42 }).collect(),
12892 _ => (0..2048).map(|row| 5 + (part * 2048 + row) * 1_000_000).collect(),
12893 };
12894 let settled = settling.encode(&values).unwrap();
12895 assert_eq!(integer::decode(&settled).unwrap(), values, "part {part}");
12896 let searched = integer::encode_with(&values, &Fixed).unwrap();
12897 assert!(
12898 settled.len() * 4 <= searched.len() * 5,
12899 "part {part}: {} settled against {} searched, {} against {}",
12900 settled.len(),
12901 searched.len(),
12902 integer::describe(&settled).unwrap(),
12903 integer::describe(&searched).unwrap(),
12904 );
12905 }
12906 }
12907
12908 #[test]
12909 fn checksum_matches_fixed_vectors() {
12910 assert_eq!(checksum(b""), 0xef46_db37_51d8_e999);
12911 assert_eq!(checksum(b"a"), 0xd24e_c4f1_a98c_6e5b);
12912 assert_eq!(checksum(b"abc"), 0x44bc_2cf5_ad77_0999);
12913 }
12914
12915 #[test]
12916 fn sorting_across_threads_matches_sorting_on_one() {
12917 let mut state = 0x9e37_79b9_7f4a_7c15_u64;
12918 let mut next = move || {
12919 state ^= state << 13;
12920 state ^= state >> 7;
12921 state ^= state << 17;
12922 state
12923 };
12924 let mut values = Vec::new();
12925 for at in 0..150_000_u64 {
12926 let value = match next() % 6 {
12927 0 => Vec::new(),
12928 1 => format!("https://example.com/{}", next() % 5_000).into_bytes(),
12929 2 => format!("https://example.com/path/{at}").into_bytes(),
12930 3 => b"same".to_vec(),
12931 4 => vec![0xff; (next() % 12) as usize],
12932 _ => (0..next() % 20).map(|_| (next() % 3) as u8).collect(),
12933 };
12934 values.push(value);
12935 }
12936 let value = |code: u32| values[code as usize].as_slice();
12937 for workers in [1, 2, 3, 8, 32] {
12938 let mut one = (0..values.len() as u32).rev().collect::<Vec<_>>();
12939 let mut across = one.clone();
12940 sort_by_value(&mut one, value);
12941 sort_by_value_across(&mut across, value, workers);
12942 assert_eq!(one, across, "{workers} workers");
12943 }
12944 let mut sorted = (0..values.len() as u32).collect::<Vec<_>>();
12945 sort_by_value_across(&mut sorted, value, 8);
12946 assert!(sorted.windows(2).all(|pair| value(pair[0]) <= value(pair[1])));
12947 }
12948
12949 fn path(label: &str) -> PathBuf {
12950 let stamp = SystemTime::now().duration_since(UNIX_EPOCH).expect("time advances").as_nanos();
12951 std::env::temp_dir().join(format!("rudb-native-{label}-{}-{stamp}.rdb", std::process::id()))
12952 }
12953
12954 fn dictionary_values(dictionary: &GlobalDictionary) -> Vec<Vec<u8>> {
12959 let (flat, bases) = dictionary.decoded(None).expect("the blocks decode");
12960 (0..dictionary.values())
12961 .map(|code| {
12962 let (from, to) = GlobalDictionary::value_span(&dictionary.ends, &bases, code);
12963 flat[from..to].to_vec()
12964 })
12965 .collect()
12966 }
12967
12968 fn attached(table: &Table) -> Vec<&Section> {
12975 table.sections().iter().filter(|held| !held.among(section::STATISTICS_KINDS)).collect()
12976 }
12977
12978 #[test]
12980 fn a_read_at_an_offset_ignores_where_another_thread_left_the_cursor() {
12981 const SPANS: usize = 64;
12982 const SPAN: usize = 512;
12983 let path = path("positional");
12984 let content: Vec<u8> =
12985 (0..SPANS).flat_map(|span| std::iter::repeat_n(span as u8, SPAN)).collect();
12986 fs::write(&path, &content).expect("the file is written");
12987 let file = Arc::new(File::open(&path).expect("the file opens"));
12988 std::thread::scope(|scope| {
12989 for _ in 0..8 {
12990 let file = Arc::clone(&file);
12991 scope.spawn(move || {
12992 for _ in 0..64 {
12993 for span in 0..SPANS {
12994 let mut bytes = [0_u8; SPAN];
12995 read_at(&file, (span * SPAN) as u64, &mut bytes)
12996 .expect("the span reads");
12997 assert!(
12998 bytes.iter().all(|byte| *byte == span as u8),
12999 "span {span} came back as {}",
13000 bytes[0],
13001 );
13002 }
13003 }
13004 });
13005 }
13006 });
13007 let mut past = [0_u8; SPAN];
13008 let end = (SPANS * SPAN) as u64;
13009 let error = read_at(&file, end, &mut past).expect_err("a read past the end is refused");
13010 assert!(error.message().contains("ends before its declared length"), "{error}");
13011 drop(file);
13012 let _ = fs::remove_file(&path);
13013 }
13014
13015 #[test]
13022 fn a_writer_puts_a_page_where_it_said_it_did_wherever_the_cursor_has_got_to() {
13023 let path = path("cursor");
13024 let mut writer = Writer::create(
13025 &path,
13026 "items",
13027 vec![
13028 Field::required("id", LogicalType::Integer),
13029 Field::new("text", LogicalType::Varchar),
13030 ],
13031 )
13032 .expect("new file");
13033 writer.append(&sample()).expect("first part");
13034 writer.append(&sample()).expect("second part");
13035 writer.finish().expect("commit");
13036 let reader = Reader::open(&path).expect("reopen from disk");
13037 assert_eq!(reader.table().rows(), 6);
13038 let ids = reader.read(0, &[0]).expect("the integer page reads back");
13039 assert_eq!(ids.value_at(0, 0), Value::Integer(4));
13040 assert_eq!(ids.value_at(2, 0), Value::Integer(-2));
13041 let text = reader.read(1, &[1]).expect("the text page reads back");
13042 assert_eq!(text.value_at(1, 0), Value::Null);
13043 assert_eq!(text.value_at(2, 0), Value::Varchar("long text after a slash".into()));
13044 let end = reader.table().stripes().iter().flat_map(|stripe| {
13047 stripe
13048 .pages
13049 .iter()
13050 .map(|page| page.offset + u64::from(page.length))
13051 .chain(std::iter::once(stripe.index.offset + u64::from(stripe.index.length)))
13052 });
13053 let last = end.fold(HEADER, u64::max);
13054 let directory = fs::metadata(&path).expect("the file is there").len();
13055 assert!(last <= directory, "a page runs to {last} in a file of {directory} bytes");
13056 fs::remove_file(path).expect("remove scratch file");
13057 }
13058
13059 fn dictionary_index_len(header: &[u8; DICTIONARY_HEADER]) -> u64 {
13065 let count = u64::from(u32::from_le_bytes(header[0..4].try_into().expect("four bytes")));
13066 let blocks = u64::from(u32::from_le_bytes(header[8..12].try_into().expect("four bytes")));
13067 let width = u32::from_le_bytes(header[12..16].try_into().expect("four bytes"));
13068 let bits = (width & !DICTIONARY_FLAGS) as usize;
13069 let payload_words = if width & DICTIONARY_SCATTERED == 0 { 2 } else { 3 };
13070 let rank_blocks = count.div_ceil(TEXT_RANK_BLOCK as u64);
13071 DICTIONARY_HEADER as u64
13072 + offset_bytes(count as usize, bits) as u64
13073 + blocks * payload_words * 8
13074 + rank_blocks * 16
13075 + if width & DICTIONARY_GRAMS == 0 { 0 } else { 8 }
13076 }
13077
13078 fn sample() -> Chunk {
13079 Chunk::new(vec![
13080 Vector::from_values(
13081 LogicalType::Integer,
13082 &[Value::Integer(4), Value::Integer(9), Value::Integer(-2)],
13083 )
13084 .expect("integers"),
13085 Vector::from_values(
13086 LogicalType::Varchar,
13087 &[
13088 Value::Varchar("alpha".into()),
13089 Value::Null,
13090 Value::Varchar("long text after a slash".into()),
13091 ],
13092 )
13093 .expect("strings"),
13094 ])
13095 .expect("matching rows")
13096 }
13097
13098 fn sample_ids() -> Chunk {
13099 Chunk::new(vec![
13100 Vector::flat(LogicalType::Integer, Data::Int32(vec![7, 8, 9].into()))
13101 .expect("integers"),
13102 ])
13103 .expect("one column")
13104 }
13105
13106 #[test]
13107 fn the_planner_gets_the_null_count_off_the_same_directory_the_bounds_are_in() {
13108 let path = path("nulls_for_the_planner");
13111 let mut writer =
13112 Writer::create(&path, "items", vec![Field::new("a", LogicalType::Integer)])
13113 .expect("new file");
13114 let rows = Chunk::new(vec![
13115 Vector::from_values(
13116 LogicalType::Integer,
13117 &[
13118 Value::Integer(4),
13119 Value::Null,
13120 Value::Integer(9),
13121 Value::Null,
13122 Value::Integer(1),
13123 Value::Integer(2),
13124 ],
13125 )
13126 .expect("integers"),
13127 ])
13128 .expect("one column");
13129 writer.append(&rows).expect("the only part");
13130 writer.finish().expect("commit");
13131 let reader = Reader::open(&path).expect("reopen from disk");
13132 let stripes = Stripes::new(reader);
13133 let column = stripes.column("a").expect("the file has that column");
13134 assert_eq!(stripes.nulls(column), Stat::exact(2, Provenance::NullCount));
13135 assert_eq!(stripes.nulls(column + 1), Stat::Unknown);
13138 fs::remove_file(&path).expect("clean up");
13139 }
13140
13141 #[test]
13142 fn the_planner_gets_a_leading_count_without_a_complete_numeric_synopsis() {
13143 let path = path("frequencies_for_the_planner");
13146 let mut writer =
13147 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
13148 .expect("new file");
13149 let rows = Chunk::new(vec![
13150 Vector::from_values(
13151 LogicalType::Integer,
13152 &[
13153 Value::Integer(4),
13154 Value::Integer(4),
13155 Value::Integer(4),
13156 Value::Integer(9),
13157 Value::Integer(9),
13158 Value::Integer(1),
13159 ],
13160 )
13161 .expect("integers"),
13162 ])
13163 .expect("one column");
13164 writer.append(&rows).expect("the only part");
13165 writer.finish().expect("commit");
13166 let reader = Reader::open(&path).expect("reopen from disk");
13167 let common = Common::new(reader);
13168 assert_eq!(common.rows(), 6);
13169 let column = common.column("id").expect("the file has that column");
13170 assert_eq!(common.column("nothing"), None);
13171 assert_eq!(
13172 common.rows_with(column, &Bound::Int(4)),
13173 Stat::exact(3, Provenance::FrequencySynopsis)
13174 );
13175 assert_eq!(common.rows_with(column, &Bound::Int(7)), Stat::Unknown);
13177 assert_eq!(common.rows_with(column, &Bound::Bytes(b"four".to_vec())), Stat::Unknown);
13180 assert!(common.remainder(column).is_some());
13181 fs::remove_file(&path).expect("clean up");
13182 }
13183
13184 #[test]
13185 fn string_frequency_estimates_do_not_open_the_global_dictionary() {
13186 let path = path("string_frequencies_for_the_planner");
13187 let mut writer =
13188 Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
13189 .expect("new file");
13190 let rows = Chunk::new(vec![
13191 Vector::from_values(
13192 LogicalType::Varchar,
13193 &[
13194 Value::Varchar(String::new()),
13195 Value::Varchar("alpha".into()),
13196 Value::Varchar(String::new()),
13197 Value::Varchar("beta".into()),
13198 Value::Varchar(String::new()),
13199 ],
13200 )
13201 .expect("strings"),
13202 ])
13203 .expect("one column");
13204 writer.append(&rows).expect("the only part");
13205 writer.finish().expect("commit");
13206
13207 let reader = Reader::open(&path).expect("reopen from disk");
13208 assert_eq!(reader.reads().dictionaries, 0, "open reads only the directory");
13209 let common = Common::new(reader.clone());
13210 let column = common.column("text").expect("the file has that column");
13211 assert_eq!(
13212 common.rows_with(column, &Bound::Bytes(Vec::new())),
13213 Stat::exact(3, Provenance::FrequencySynopsis)
13214 );
13215 assert_eq!(
13216 common.rows_with(column, &Bound::Bytes(b"missing".to_vec())),
13217 Stat::exact(0, Provenance::FrequencySynopsis)
13218 );
13219 assert_eq!(
13220 reader.reads().dictionaries,
13221 0,
13222 "the bounded spellings answer without opening the dictionary index"
13223 );
13224 fs::remove_file(&path).expect("clean up");
13225 }
13226
13227 #[test]
13228 fn host_groups_certify_omitted_hosts_and_keep_exact_aggregates() {
13229 let path = path("certified_host_groups");
13230 let mut writer =
13231 Writer::create(&path, "hits", vec![Field::required("Referer", LogicalType::Varchar)])
13232 .expect("new file");
13233 let mut values = vec![Value::Varchar("http://www.example.com/a".into()); 150];
13234 values.extend(vec![Value::Varchar("https://example.com/b".into()); 70]);
13235 values.extend((0..550).map(|at| Value::Varchar(format!("https://site{at}.test/x"))));
13236 values.push(Value::Varchar(String::new()));
13237 for part in values.chunks(512) {
13238 writer
13239 .append(
13240 &Chunk::new(vec![
13241 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
13242 ])
13243 .expect("one column"),
13244 )
13245 .expect("part written");
13246 }
13247 writer.finish().expect("commit");
13248 let reader = Reader::open(&path).expect("reopen");
13249 assert!(reader.table.host_groups.is_none(), "no query-specific host result is stored");
13250 fs::remove_file(&path).expect("clean up");
13251 }
13252
13253 fn bare_table(sections: Vec<Section>) -> Table {
13258 Table {
13259 name: "linked".to_owned(),
13260 fields: vec![Field::required("id", LogicalType::Integer)],
13261 stripes: Vec::new(),
13262 rows: 0,
13263 dictionaries: vec![None],
13264 dictionary_payloads: Vec::new(),
13265 demoted: Vec::new(),
13266 distincts: vec![None],
13267 frequencies: vec![None],
13268 pair_frequencies: Vec::new(),
13269 frequency_texts: Vec::new(),
13270 host_groups: None,
13271 clustering: None,
13272 constraints: Constraints::default(),
13273 generation: 1,
13274 sections,
13275 }
13276 }
13277
13278 fn a_key_map_section() -> Section {
13279 Section {
13280 kind: *section::KEY_MAP,
13281 id: 1,
13282 generation: 3,
13283 extents: 1,
13284 extent_page: HEADER,
13285 extent_bytes: section::EXTENT_BYTES as u32,
13286 hash: 0x1234_5678_9abc_def0,
13287 flags: 0,
13288 header_bytes: 24,
13289 }
13290 }
13291
13292 #[test]
13293 fn a_section_table_round_trips_through_a_directory() {
13294 let mut later = a_key_map_section();
13295 later.kind = *b"RUDBZZ9\0";
13296 later.id = 2;
13297 let table = bare_table(vec![a_key_map_section(), later]);
13298 let directory = encode_directory(&table).expect("directory");
13299 let decoded = decode_directory(&directory, 1 << 20).expect("reopen");
13300 assert_eq!(decoded.sections(), &[a_key_map_section(), later]);
13301 assert!(decoded.sections()[0].known());
13305 assert!(!decoded.sections()[1].known());
13306 }
13307
13308 #[test]
13309 fn a_directory_written_before_the_section_table_reads_as_a_table_with_none() {
13310 let directory = encode_directory(&bare_table(Vec::new())).expect("directory");
13314 let block = SECTIONS.len() + size_of::<u64>() + size_of::<u16>();
13315 let older = &directory[..directory.len() - block];
13316 let decoded = decode_directory(older, 1 << 20).expect("a directory from before sections");
13317 assert!(decoded.sections().is_empty());
13318 assert_eq!(decoded.generation(), 0, "a format 22 table recorded no generation");
13319 assert_eq!(decoded.name(), "linked");
13320 assert_eq!(decoded.fields().len(), 1, "everything before the block still decodes");
13321 }
13322
13323 #[test]
13324 fn a_file_stamped_with_the_previous_format_still_opens_and_reads() {
13325 let path = path("format_twenty_two");
13332 let mut writer =
13333 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
13334 .expect("new file");
13335 let rows = Chunk::new(vec![
13336 Vector::from_values(
13337 LogicalType::Integer,
13338 &[Value::Integer(1), Value::Integer(2), Value::Integer(3)],
13339 )
13340 .expect("integers"),
13341 ])
13342 .expect("one column");
13343 writer.append(&rows).expect("the only part");
13344 writer.finish().expect("commit");
13345
13346 let file = OpenOptions::new().write(true).open(&path).expect("reopen to patch");
13347 write_at(&file, 8, &22_u32.to_le_bytes()).expect("stamp the older format");
13348 drop(file);
13349
13350 let reader = Reader::open(&path).expect("a format 22 file opens unchanged");
13351 assert_eq!(reader.table().rows(), 3);
13352 assert_eq!(reader.read(0, &[0]).expect("the part still reads").len(), 3);
13357
13358 let file = OpenOptions::new().write(true).open(&path).expect("reopen to patch");
13361 write_at(&file, 8, &21_u32.to_le_bytes()).expect("stamp an unreadable format");
13362 drop(file);
13363 let error = Reader::open(&path).expect_err("format 21 is not readable");
13364 assert!(error.to_string().contains("format 21"), "{error}");
13365
13366 fs::remove_file(&path).expect("clean up");
13367 }
13368
13369 #[test]
13370 fn a_section_whose_extent_table_is_outside_the_file_is_refused() {
13371 let mut past = a_key_map_section();
13376 past.extent_page = 1 << 30;
13377 let directory = encode_directory(&bare_table(vec![past])).expect("directory");
13378 let error = decode_directory(&directory, 1 << 20).expect_err("refused");
13379 assert!(error.to_string().contains("outside the file"), "{error}");
13380
13381 let mut inside_the_header = a_key_map_section();
13382 inside_the_header.extent_page = 8;
13383 let directory = encode_directory(&bare_table(vec![inside_the_header])).expect("directory");
13384 assert!(
13385 decode_directory(&directory, 1 << 20).is_err(),
13386 "a section may not overlap a header"
13387 );
13388 }
13389
13390 #[test]
13391 fn a_section_recorded_as_not_built_is_legal_and_names_no_bytes() {
13392 let not_built = Section {
13396 kind: *section::FORWARD_LINK,
13397 id: 9,
13398 generation: 3,
13399 extents: 0,
13400 extent_page: 0,
13401 extent_bytes: 0,
13402 hash: 0,
13403 flags: 0,
13404 header_bytes: 0,
13405 };
13406 let directory = encode_directory(&bare_table(vec![not_built])).expect("directory");
13407 let decoded = decode_directory(&directory, 1 << 20).expect("reopen");
13408 assert_eq!(decoded.sections(), &[not_built]);
13409
13410 let mut incoherent = not_built;
13413 incoherent.extent_bytes = 28;
13414 incoherent.extent_page = HEADER;
13415 let directory = encode_directory(&bare_table(vec![incoherent])).expect("directory");
13416 assert!(decode_directory(&directory, 1 << 20).is_err());
13417 }
13418
13419 #[test]
13420 fn a_directory_naming_more_sections_than_the_bound_is_refused() {
13421 let directory = encode_directory(&bare_table(Vec::new())).expect("directory");
13422 let mut torn = directory.clone();
13423 let count_at = torn.len() - size_of::<u16>();
13424 torn[count_at..].copy_from_slice(&u16::MAX.to_le_bytes());
13425 assert!(decode_directory(&torn, 1 << 20).is_err());
13428 }
13429
13430 fn linked_file(label: &str, rows: i32) -> PathBuf {
13432 let path = path(label);
13433 let mut writer =
13434 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
13435 .expect("new file");
13436 let values = (0..rows).map(Value::Integer).collect::<Vec<_>>();
13437 let chunk =
13438 Chunk::new(vec![Vector::from_values(LogicalType::Integer, &values).expect("integers")])
13439 .expect("one column");
13440 writer.append(&chunk).expect("the only part");
13441 writer.finish().expect("commit");
13442 path
13443 }
13444
13445 fn a_key_map_payload() -> Vec<u8> {
13446 (0..512_u32).flat_map(u32::to_le_bytes).collect()
13449 }
13450
13451 #[test]
13452 fn a_section_attached_to_a_committed_file_reads_back_byte_for_byte() {
13453 let path = linked_file("attach", 64);
13454 let payload = a_key_map_payload();
13455 let table = attach(
13456 &path,
13457 "items",
13458 &[section::Attachment {
13459 kind: *section::KEY_MAP,
13460 id: 0,
13461 flags: 2,
13462 header_bytes: 40,
13463 bytes: &payload,
13464 }],
13465 )
13466 .expect("attach a key map");
13467 assert_eq!(attached(&table).len(), 1);
13468
13469 let reader = Reader::open(&path).expect("reopen after the attach");
13470 let held = attached(reader.table());
13471 assert_eq!(held.len(), 1);
13472 assert_eq!(held[0].kind, *section::KEY_MAP);
13473 assert_eq!(held[0].flags, 2, "the form a reader must not have to guess");
13474 assert_eq!(held[0].header_bytes, 40);
13475 assert_eq!(held[0].generation, 1);
13479 assert!(held[0].usable(reader.table().generation()));
13480 assert_eq!(reader.payload(held[0]).expect("read the payload"), payload);
13481 assert_eq!(reader.extents(held[0]).expect("extent table").len(), 1);
13482
13483 fs::remove_file(&path).expect("clean up");
13484 }
13485
13486 #[test]
13487 fn attaching_a_section_answers_every_row_exactly_as_before() {
13488 let path = linked_file("attach_changes_nothing", 300);
13493 let before = Reader::open(&path).expect("open before");
13494 let rows = before.table().rows();
13495 let first = before.read(0, &[0]).expect("read before");
13496 let values = (0..rows).map(|at| first.value_at(at, 0)).collect::<Vec<_>>();
13497 let layout = before.layout().columns_total();
13498 drop(before);
13499
13500 let payload = a_key_map_payload();
13501 attach(
13502 &path,
13503 "items",
13504 &[section::Attachment {
13505 kind: *section::KEY_MAP,
13506 id: 0,
13507 flags: 0,
13508 header_bytes: 0,
13509 bytes: &payload,
13510 }],
13511 )
13512 .expect("attach");
13513
13514 let after = Reader::open(&path).expect("open after");
13515 assert_eq!(after.table().rows(), rows);
13516 let read = after.read(0, &[0]).expect("read after");
13517 for (at, value) in values.iter().enumerate() {
13518 assert_eq!(&read.value_at(at, 0), value, "row {at} moved");
13519 }
13520 assert_eq!(
13521 after.layout().columns_total(),
13522 layout,
13523 "an attach appends and does not rewrite a column page"
13524 );
13525
13526 fs::remove_file(&path).expect("clean up");
13527 }
13528
13529 #[test]
13530 fn a_rebuilt_section_replaces_the_one_it_supersedes() {
13531 let path = linked_file("attach_twice", 32);
13535 let one = a_key_map_payload();
13536 let two = vec![7_u8; 1024];
13537 let entry = |bytes| section::Attachment {
13538 kind: *section::KEY_MAP,
13539 id: 4,
13540 flags: 1,
13541 header_bytes: 0,
13542 bytes,
13543 };
13544 attach(&path, "items", &[entry(&one)]).expect("first build");
13545 attach(&path, "items", &[entry(&two)]).expect("rebuild");
13546
13547 let reader = Reader::open(&path).expect("reopen");
13548 let held = attached(reader.table());
13549 assert_eq!(held.len(), 1, "one map per column and not one per build");
13550 assert_eq!(reader.payload(held[0]).expect("payload"), two);
13551
13552 fs::remove_file(&path).expect("clean up");
13553 }
13554
13555 #[test]
13556 fn an_attach_carries_through_a_kind_it_does_not_know() {
13557 let path = linked_file("attach_unknown", 16);
13561 let payload = vec![3_u8; 96];
13562 attach(
13563 &path,
13564 "items",
13565 &[section::Attachment {
13566 kind: *b"RUDBZZ9\0",
13567 id: 1,
13568 flags: 0,
13569 header_bytes: 0,
13570 bytes: &payload,
13571 }],
13572 )
13573 .expect("a kind this build does not know still writes");
13574 let key_map = a_key_map_payload();
13575 attach(
13576 &path,
13577 "items",
13578 &[section::Attachment {
13579 kind: *section::KEY_MAP,
13580 id: 0,
13581 flags: 0,
13582 header_bytes: 0,
13583 bytes: &key_map,
13584 }],
13585 )
13586 .expect("attach beside it");
13587
13588 let reader = Reader::open(&path).expect("reopen");
13589 let held = attached(reader.table());
13590 assert_eq!(held.len(), 2, "the unfamiliar entry survived a directory rewrite");
13591 let unknown = held.iter().find(|one| !one.known()).expect("the unfamiliar one");
13592 assert_eq!(reader.payload(unknown).expect("its bytes are still there"), payload);
13593
13594 fs::remove_file(&path).expect("clean up");
13595 }
13596
13597 #[test]
13598 fn a_payload_of_nothing_is_a_relationship_recorded_as_not_built() {
13599 let path = linked_file("attach_not_built", 8);
13600 attach(
13601 &path,
13602 "items",
13603 &[section::Attachment {
13604 kind: *section::FORWARD_LINK,
13605 id: 2,
13606 flags: 0,
13607 header_bytes: 0,
13608 bytes: &[],
13609 }],
13610 )
13611 .expect("record a link that did not fit the budget");
13612
13613 let reader = Reader::open(&path).expect("reopen");
13614 let held = attached(reader.table());
13615 assert_eq!(held.len(), 1);
13616 assert_eq!(held[0].extents, 0);
13617 assert_eq!(held[0].extent_page, 0, "an entry that names no bytes points at none");
13618 assert!(reader.extents(held[0]).expect("no extent table").is_empty());
13619 assert!(reader.payload(held[0]).expect("no payload").is_empty());
13620
13621 fs::remove_file(&path).expect("clean up");
13622 }
13623
13624 #[test]
13625 fn a_payload_past_one_extent_is_split_and_joined_back() {
13626 let path = linked_file("attach_two_extents", 8);
13630 let payload = vec![0x5a_u8; section::MAX_EXTENT as usize + 1];
13631 attach(
13632 &path,
13633 "items",
13634 &[section::Attachment {
13635 kind: *section::KEY_MAP,
13636 id: 0,
13637 flags: 0,
13638 header_bytes: 0,
13639 bytes: &payload,
13640 }],
13641 )
13642 .expect("attach a payload past the bound");
13643
13644 let reader = Reader::open(&path).expect("reopen");
13645 let held = attached(reader.table());
13646 let extents = reader.extents(held[0]).expect("extent table");
13647 assert_eq!(extents.len(), 2, "one byte past the bound is two extents");
13648 assert_eq!(extents[0].length, section::MAX_EXTENT);
13649 assert_eq!(extents[1].length, 1);
13650 assert_eq!(extents[1].first, u64::from(section::MAX_EXTENT));
13651 assert_eq!(reader.extent(&extents[1]).expect("the last extent"), vec![0x5a]);
13653 assert_eq!(reader.payload(held[0]).expect("the whole payload").len(), payload.len());
13654
13655 fs::remove_file(&path).expect("clean up");
13656 }
13657
13658 #[test]
13659 fn a_torn_extent_is_refused_rather_than_decoded() {
13660 let path = linked_file("attach_torn", 8);
13661 let payload = a_key_map_payload();
13662 attach(
13663 &path,
13664 "items",
13665 &[section::Attachment {
13666 kind: *section::KEY_MAP,
13667 id: 0,
13668 flags: 0,
13669 header_bytes: 0,
13670 bytes: &payload,
13671 }],
13672 )
13673 .expect("attach");
13674
13675 let reader = Reader::open(&path).expect("reopen");
13676 let extent = reader.extents(&reader.table().sections()[0]).expect("extent table")[0];
13677 let file = OpenOptions::new().write(true).open(&path).expect("reopen to corrupt");
13678 write_at(&file, extent.offset + 7, &[0xff]).expect("flip a byte of the payload");
13679 drop(file);
13680
13681 let reader = Reader::open(&path).expect("the table still opens");
13682 let error = reader
13683 .payload(&reader.table().sections()[0])
13684 .expect_err("a corrupt payload is not handed out");
13685 assert!(error.to_string().contains("checksum"), "{error}");
13686 assert_eq!(reader.read(0, &[0]).expect("the column is untouched").width(), 1);
13689
13690 fs::remove_file(&path).expect("clean up");
13691 }
13692
13693 #[test]
13694 fn attaching_to_a_file_of_the_previous_format_is_refused_rather_than_done() {
13695 let path = linked_file("attach_old_format", 8);
13698 let file = OpenOptions::new().write(true).open(&path).expect("reopen to patch");
13699 write_at(&file, 8, &22_u32.to_le_bytes()).expect("stamp the older format");
13700 drop(file);
13701
13702 let payload = a_key_map_payload();
13703 let error = attach(
13704 &path,
13705 "items",
13706 &[section::Attachment {
13707 kind: *section::KEY_MAP,
13708 id: 0,
13709 flags: 0,
13710 header_bytes: 0,
13711 bytes: &payload,
13712 }],
13713 )
13714 .expect_err("format 22 cannot gain a section");
13715 assert!(error.to_string().contains("format 22"), "{error}");
13716 assert!(Reader::open(&path).expect("and the file is untouched").table().rows() == 8);
13717
13718 fs::remove_file(&path).expect("clean up");
13719 }
13720
13721 #[test]
13722 fn a_section_header_longer_than_its_payload_is_refused_at_the_write() {
13723 let path = linked_file("attach_bad_header", 8);
13724 let error = attach(
13725 &path,
13726 "items",
13727 &[section::Attachment {
13728 kind: *section::KEY_MAP,
13729 id: 0,
13730 flags: 0,
13731 header_bytes: 40,
13732 bytes: &[1, 2, 3],
13733 }],
13734 )
13735 .expect_err("a writer's bug stops at the write");
13736 assert!(error.to_string().contains("header is longer"), "{error}");
13737
13738 fs::remove_file(&path).expect("clean up");
13739 }
13740
13741 #[test]
13742 fn attaching_to_a_name_the_file_does_not_hold_says_so() {
13743 let path = linked_file("attach_wrong_name", 8);
13744 let error = attach(&path, "orders", &[]).expect_err("no such table");
13745 assert!(error.to_string().contains("orders"), "{error}");
13746 fs::remove_file(&path).expect("clean up");
13747 }
13748
13749 #[test]
13750 fn the_planner_gets_an_exact_count_for_a_leading_value_of_an_incomplete_synopsis() {
13751 let path = path("frequency_prefix_for_the_planner");
13758 let mut writer =
13759 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
13760 .expect("new file");
13761 let mut values = vec![Value::Integer(1); 10_000];
13762 for _ in 0..10 {
13763 values.extend((0..600).map(|tail| Value::Integer(1_000 + tail)));
13764 }
13765 for part in values.chunks(8_000) {
13768 let rows = Chunk::new(vec![
13769 Vector::from_values(LogicalType::Integer, part).expect("integers"),
13770 ])
13771 .expect("one column");
13772 writer.append(&rows).expect("a part");
13773 }
13774 writer.finish().expect("commit");
13775 let reader = Reader::open(&path).expect("reopen from disk");
13776 let prefix =
13777 reader.frequency_prefix(0).expect("a readable synopsis").expect("the column has one");
13778 assert_eq!(prefix.entries.len(), 512);
13781 assert_eq!(prefix.omitted_max, 10);
13782 let common = Common::new(reader);
13783 assert_eq!(common.rows(), 16_000);
13784 let column = common.column("id").expect("the file has that column");
13785 assert_eq!(
13786 common.rows_with(column, &Bound::Int(1)),
13787 Stat::exact(10_000, Provenance::FrequencySynopsis)
13788 );
13789 assert_eq!(
13791 common.rows_with(column, &Bound::Int(1_100)),
13792 Stat::exact(10, Provenance::FrequencySynopsis)
13793 );
13794 assert_eq!(common.rows_with(column, &Bound::Int(1_550)), Stat::Unknown);
13797 assert_eq!(common.rows_with(column, &Bound::Int(9_999)), Stat::Unknown);
13800 let remainder = common.remainder(column).expect("the list is a prefix");
13804 assert_eq!(remainder, Remainder { rows: 890, listed: 512, most: 10 });
13805 assert_eq!(remainder.rows / (601 - remainder.listed), 10);
13806 fs::remove_file(&path).expect("clean up");
13807 }
13808
13809 #[test]
13811 fn a_file_holding_no_table_commits_and_opens_and_a_table_can_be_added_to_it() {
13812 let path = path("empty");
13813 Writer::empty(&path, &[]).expect("a file with nothing in it");
13814 let catalog = Catalog::open(&path).expect("the empty file opens");
13815 assert_eq!(catalog.len(), 0);
13816 assert!(catalog.is_empty());
13817 assert_eq!(catalog.names().count(), 0);
13818 let mut writer =
13821 Writer::open(&path, "items", vec![Field::required("id", LogicalType::Integer)])
13822 .expect("a table goes into the empty file");
13823 writer.append(&sample_ids()).expect("rows");
13824 writer.finish().expect("commit");
13825 let catalog = Catalog::open(&path).expect("the file opens again");
13826 assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
13827 fs::remove_file(&path).expect("clean up");
13828 }
13829
13830 #[test]
13840 fn a_committed_empty_table_gives_up_its_name_and_one_with_rows_does_not() {
13841 let path = path("empty-name");
13842 let field = || vec![Field::required("id", LogicalType::Integer)];
13843 Writer::create(&path, "items", field()).expect("new file").finish().expect("commit");
13844 let catalog = Catalog::open(&path).expect("the file opens");
13845 assert_eq!(catalog.rows().collect::<Vec<_>>(), vec![("items", 0)]);
13846
13847 let mut writer = Writer::open(&path, "items", field()).expect("the empty name is free");
13848 writer.append(&sample_ids()).expect("rows");
13849 writer.finish().expect("commit");
13850 let catalog = Catalog::open(&path).expect("the file opens again");
13851 assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
13853 let held = catalog.rows().collect::<Vec<_>>();
13854 assert_eq!(held.len(), 1);
13855 assert!(held[0].1 > 0, "the rows that were appended are the ones the catalog counts");
13856
13857 let error = Writer::open(&path, "items", field()).expect_err("a name with rows is taken");
13859 assert!(error.to_string().contains("same name"), "{error}");
13860 fs::remove_file(&path).expect("clean up");
13861 }
13862
13863 fn sample_view(name: &str) -> ViewEntry {
13865 ViewEntry {
13866 name: name.to_string(),
13867 sql: "SELECT id FROM items WHERE id > 0".to_string(),
13868 statement: format!("CREATE VIEW {name} AS SELECT id FROM items WHERE (id > 0);"),
13869 aliases: vec!["n".to_string()],
13870 columns: vec![Field::new("n", LogicalType::Integer)],
13871 }
13872 }
13873
13874 #[test]
13875 fn a_view_written_into_the_catalog_comes_back_whole() {
13876 let path = path("views");
13877 let mut writer =
13878 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
13879 .expect("new file");
13880 writer.append(&sample_ids()).expect("rows");
13881 writer.with_views(vec![sample_view("v")]).finish().expect("commit");
13882 let catalog = Catalog::open(&path).expect("reopen");
13883 assert_eq!(catalog.views().cloned().collect::<Vec<_>>(), vec![sample_view("v")]);
13884 assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
13887 fs::remove_file(&path).expect("clean up");
13888 }
13889
13890 #[test]
13892 fn appending_a_table_carries_the_views_forward() {
13893 let path = path("viewscarry");
13894 let mut writer =
13895 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
13896 .expect("new file");
13897 writer.append(&sample_ids()).expect("rows");
13898 writer.with_views(vec![sample_view("v")]).finish().expect("commit");
13899 let mut writer =
13900 Writer::open(&path, "other", vec![Field::required("id", LogicalType::Integer)])
13901 .expect("a second table");
13902 writer.append(&sample_ids()).expect("rows");
13903 writer.finish().expect("commit");
13904 let catalog = Catalog::open(&path).expect("reopen");
13905 assert_eq!(catalog.views().count(), 1);
13906 assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items", "other"]);
13907 fs::remove_file(&path).expect("clean up");
13908 }
13909
13910 #[test]
13912 fn restating_the_views_leaves_every_table_where_it_was() {
13913 let path = path("restate");
13914 let mut writer =
13915 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
13916 .expect("new file");
13917 writer.append(&sample_ids()).expect("rows");
13918 writer.finish().expect("commit");
13919 let before = fs::metadata(&path).expect("the file is there").len();
13920 Writer::restate(&path, &[sample_view("v"), sample_view("w")]).expect("two views");
13921 let catalog = Catalog::open(&path).expect("reopen");
13922 assert_eq!(catalog.views().count(), 2);
13923 assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
13924 let after = fs::metadata(&path).expect("the file is there").len();
13927 assert!(after > before, "a generation was written");
13928 assert!(after - before < before, "the table was not written again");
13929 let reader = Catalog::open(&path).expect("reopen").table("items").expect("the table");
13932 assert_eq!(reader.table().rows, 3);
13933 Writer::restate(&path, &[]).expect("no views at all");
13936 assert_eq!(Catalog::open(&path).expect("reopen").views().count(), 0);
13937 fs::remove_file(&path).expect("clean up");
13938 }
13939
13940 #[test]
13942 fn a_view_named_after_a_table_is_refused_when_the_catalog_is_read() {
13943 let bytes = encode_catalog(
13944 &[Entry {
13945 name: "items".to_string(),
13946 fields: vec![Field::required("id", LogicalType::Integer)],
13947 rows: 1,
13948 directory: Page { offset: HEADER, length: 8, hash: 0 },
13949 nonzero: vec![None],
13950 aggregates: vec![None],
13951 distincts: vec![None],
13952 extremes: vec![None],
13953 frequencies: vec![None],
13954 }],
13955 &[sample_view("items")],
13956 )
13957 .expect("it encodes, because encoding does not look");
13958 let error = decode_catalog(&bytes, HEADER + 8).expect_err("and decoding does");
13959 assert!(error.to_string().contains("same name"), "{error}");
13960 }
13961
13962 #[test]
13965 fn a_compressed_text_page_read_at_some_rows_is_those_rows_of_the_whole() {
13966 let rows: usize = 300;
13967 let text: Vec<String> =
13968 (0..rows).map(|row| format!("a street named after number {}", row * 7)).collect();
13969 let values: Vec<&[u8]> = text.iter().map(String::as_bytes).collect();
13970 let mut page = vec![6, 2];
13971 page.extend((0..rows.div_ceil(8)).map(|byte| {
13972 (0..8).filter(|bit| (byte * 8 + bit) % 5 != 3).fold(0_u8, |mask, bit| mask | 1 << bit)
13973 }));
13974 let compressed = string::encode_only(string::Kind::Fsst, &values)
13975 .expect("encoded")
13976 .expect("text this repetitive compresses");
13977 page.extend_from_slice(&compressed);
13978 let whole = decode(&LogicalType::Varchar, rows, &page, None).expect("the whole page");
13979 let positions = [0_u32, 3, 8, 13, 200, 299];
13980 let some =
13981 decode_at(&LogicalType::Varchar, rows, &page, None, &positions).expect("some rows");
13982 assert_eq!(some.len(), positions.len());
13983 for (at, &row) in positions.iter().enumerate() {
13984 assert_eq!(some.value_at(at), whole.value_at(row as usize), "row {row}");
13985 }
13986 assert_eq!(some.value_at(1), Value::Null, "row 3 is null");
13987 assert!(decode_at(&LogicalType::Varchar, rows, &page, None, &[300]).is_err());
13988 assert!(decode_at(&LogicalType::Varchar, rows, &page, None, &[8, 3]).is_err());
13989 }
13990
13991 #[test]
13994 fn a_part_read_at_some_rows_is_the_part_read_whole_and_gathered() {
13995 let path = path("rows");
13996 let mut writer = Writer::create(
13997 &path,
13998 "items",
13999 vec![
14000 Field::required("id", LogicalType::Integer),
14001 Field::new("text", LogicalType::Varchar),
14002 ],
14003 )
14004 .expect("new file");
14005 let rows = 2_000;
14006 let chunk = Chunk::new(vec![
14007 Vector::from_values(
14008 LogicalType::Integer,
14009 &(0..rows).map(Value::Integer).collect::<Vec<_>>(),
14010 )
14011 .expect("integers"),
14012 Vector::from_values(
14013 LogicalType::Varchar,
14014 &(0..rows)
14015 .map(|row| {
14016 if row % 7 == 2 {
14017 Value::Null
14018 } else {
14019 Value::Varchar(format!("a comment about order {}", row * 13))
14020 }
14021 })
14022 .collect::<Vec<_>>(),
14023 )
14024 .expect("strings"),
14025 ])
14026 .expect("matching rows");
14027 writer.append(&chunk).expect("one part");
14028 writer.finish().expect("commit");
14029 let reader = Reader::open(&path).expect("reopen from disk");
14030 let positions = [1_u32, 2, 9, 1_000, 1_999];
14031 for whole in [true, false] {
14032 let some = reader.read_rows(0, &[0, 1], &positions, whole).expect("some rows");
14033 let all = reader.read(0, &[0, 1]).expect("the whole part");
14034 assert_eq!(some.len(), positions.len());
14035 for column in 0..2 {
14036 for (at, &row) in positions.iter().enumerate() {
14037 assert_eq!(some.value_at(at, column), all.value_at(row as usize, column));
14038 }
14039 }
14040 }
14041 assert!(reader.read_rows(0, &[1], &[2_000], true).is_err());
14042 }
14043
14044 #[test]
14045 fn committed_file_reopens_and_reads_only_requested_columns() {
14046 let path = path("reopen");
14047 let mut writer = Writer::create(
14048 &path,
14049 "items",
14050 vec![
14051 Field::required("id", LogicalType::Integer),
14052 Field::new("text", LogicalType::Varchar),
14053 ],
14054 )
14055 .expect("new file");
14056 writer.append(&sample()).expect("first part");
14057 writer.append(&sample()).expect("second part");
14058 writer.finish().expect("commit");
14059 let reader = Reader::open(&path).expect("reopen from disk");
14060 assert_eq!(reader.table().rows(), 6);
14061 assert_eq!(reader.table().stripes().len(), 1);
14064 assert_eq!(reader.parts(), 2);
14065 assert_eq!(reader.part_rows(0), 3);
14066 assert_eq!(reader.part_rows(1), 3);
14067 let text = reader.read(1, &[1]).expect("only text page");
14068 assert_eq!(text.width(), 1);
14069 assert_eq!(text.value_at(1, 0), Value::Null);
14070 assert_eq!(text.value_at(2, 0), Value::Varchar("long text after a slash".into()));
14071 let sparse = reader.read_sparse(1, &[1]).expect("one part without its whole page");
14072 assert_eq!(sparse.width(), 1);
14073 assert_eq!(sparse.value_at(1, 0), Value::Null);
14074 assert_eq!(sparse.value_at(2, 0), Value::Varchar("long text after a slash".into()));
14075 assert!(!reader.skips_codes(0, 1, &[0]).expect("alpha is in the stripe"));
14076 assert!(!reader.skips_codes(0, 1, &[2]).expect("long text is in the stripe"));
14077 assert!(reader.skips_codes(0, 1, &[3]).expect("unknown code is absent"));
14078 let count = reader.read(0, &[]).expect("no page is needed for count");
14079 assert_eq!(count.len(), 3);
14080 assert!(reader.skips(0, &[Probe { column: 0, op: Op::Greater, value: Bound::Int(100) }]));
14081 assert!(!reader.skips(0, &[Probe { column: 0, op: Op::Greater, value: Bound::Int(0) }]));
14082 assert_eq!(reader.top_frequencies(0, 1).expect("valid integer synopsis"), None);
14083 let integers = reader.frequency_prefix(0).expect("valid integer synopsis").expect("kept");
14084 assert_eq!(integers.entries, vec![(Value::Integer(-2), 2), (Value::Integer(4), 2)]);
14085 assert_eq!(integers.omitted_max, 2);
14086 let strings = reader.top_frequencies(1, 1).expect("valid string synopsis").expect("kept");
14087 assert_eq!(strings.len(), 3);
14088 assert!(strings.contains(&(Value::Null, 2)));
14089 assert!(strings.contains(&(Value::Varchar("alpha".into()), 2)));
14090 assert!(strings.contains(&(Value::Varchar("long text after a slash".into()), 2)));
14091 fs::remove_file(path).expect("remove scratch file");
14092 }
14093
14094 #[test]
14102 fn runs_handed_over_out_of_order_still_read_back_in_source_order() {
14103 let path = path("interleaved-runs");
14104 let mut writer =
14105 Writer::create(&path, "interleaved", vec![Field::new("v", LogicalType::BigInt)])
14106 .expect("new file");
14107 for morsel in [2_u64, 0, 3, 1] {
14108 let parts = (0..4_u64)
14109 .map(|chunk| {
14110 let first = i64::try_from(morsel * 32 + chunk * 8).expect("small");
14111 let values =
14112 (0..8_i64).map(|row| Value::BigInt(first + row)).collect::<Vec<_>>();
14113 let column =
14114 Vector::from_values(LogicalType::BigInt, &values).expect("a column");
14115 ((morsel, chunk), Chunk::new(vec![column]).expect("one column"))
14116 })
14117 .collect::<Vec<_>>();
14118 writer.append_stripe(parts).expect("a stripe");
14119 }
14120 writer.finish().expect("commit");
14121
14122 let reader = Reader::open(&path).expect("valid directory");
14123 assert_eq!(reader.table().stripes().len(), 4, "a run is a stripe of its own");
14124 assert_eq!(reader.table().rows(), 128);
14125 for part in 0..16_usize {
14126 let read = reader.read(part, &[0]).expect("a part back");
14127 for row in 0..8_usize {
14128 let want = i64::try_from(part * 8 + row).expect("small");
14129 assert_eq!(read.value_at(row, 0), Value::BigInt(want), "part {part} row {row}");
14130 }
14131 }
14132 fs::remove_file(path).expect("remove scratch file");
14133 }
14134
14135 #[test]
14138 fn runs_that_overlap_each_other_are_refused_at_commit() {
14139 let path = path("overlapping-runs");
14140 let mut writer =
14141 Writer::create(&path, "overlapping", vec![Field::new("v", LogicalType::BigInt)])
14142 .expect("new file");
14143 let one = |order: (u64, u64)| {
14144 let column =
14145 Vector::from_values(LogicalType::BigInt, &[Value::BigInt(1)]).expect("a column");
14146 (order, Chunk::new(vec![column]).expect("one column"))
14147 };
14148 writer.append_stripe(vec![one((0, 0)), one((0, 2))]).expect("a stripe");
14151 writer.append_stripe(vec![one((0, 1))]).expect("a stripe");
14152 let error = writer.finish().expect_err("the runs overlap");
14153 assert!(error.message().contains("source order"), "{error}");
14154 fs::remove_file(path).expect("remove scratch file");
14155 }
14156
14157 #[test]
14160 fn a_run_longer_than_a_stripe_is_refused() {
14161 let path = path("overlong-run");
14162 let mut writer =
14163 Writer::create(&path, "overlong", vec![Field::new("v", LogicalType::BigInt)])
14164 .expect("new file");
14165 let parts = (0..=STRIPE_PARTS)
14166 .map(|at| {
14167 let column = Vector::from_values(LogicalType::BigInt, &[Value::BigInt(1)])
14168 .expect("a column");
14169 let chunk = Chunk::new(vec![column]).expect("one column");
14170 ((0, u64::try_from(at).expect("small")), chunk)
14171 })
14172 .collect::<Vec<_>>();
14173 let error = writer.append_stripe(parts).expect_err("one part too many");
14174 assert!(error.message().contains("more parts than it holds"), "{error}");
14175 fs::remove_file(path).expect("remove scratch file");
14176 }
14177
14178 #[test]
14184 fn parts_past_the_stripe_bound_start_a_new_stripe() {
14185 let path = path("stripe-bound");
14186 let mut writer = Writer::create(
14187 &path,
14188 "items",
14189 vec![
14190 Field::required("id", LogicalType::Integer),
14191 Field::new("text", LogicalType::Varchar),
14192 ],
14193 )
14194 .expect("new file");
14195 let parts = STRIPE_PARTS * 2 + 3;
14196 for part in 0..parts {
14197 let id = part as i32;
14198 let chunk = Chunk::new(vec![
14199 Vector::from_values(
14200 LogicalType::Integer,
14201 &[Value::Integer(id), Value::Integer(-id)],
14202 )
14203 .expect("integers"),
14204 Vector::from_values(
14205 LogicalType::Varchar,
14206 &[Value::Varchar(format!("value {part}")), Value::Null],
14207 )
14208 .expect("strings"),
14209 ])
14210 .expect("matching rows");
14211 writer.append(&chunk).expect("one part");
14212 }
14213 writer.finish().expect("commit");
14214
14215 let reader = Reader::open(&path).expect("reopen from disk");
14216 assert_eq!(reader.parts(), parts);
14217 assert_eq!(reader.table().rows(), parts * 2);
14218 assert_eq!(reader.table().stripes().len(), parts.div_ceil(STRIPE_PARTS));
14219 assert_eq!(reader.table().stripes()[0].parts(), STRIPE_PARTS);
14220 assert_eq!(reader.table().stripes()[0].rows(), STRIPE_PARTS * 2);
14221 assert_eq!(reader.table().stripes()[2].parts(), 3);
14222 for part in (0..parts).rev() {
14225 let dense = reader.read(part, &[0, 1]).expect("a whole page read");
14226 let sparse = reader.read_sparse(part, &[0, 1]).expect("one part read");
14227 for chunk in [&dense, &sparse] {
14228 assert_eq!(chunk.len(), 2, "part {part} has its own row count");
14229 assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
14230 assert_eq!(chunk.value_at(1, 0), Value::Integer(-(part as i32)));
14231 assert_eq!(chunk.value_at(0, 1), Value::Varchar(format!("value {part}")));
14232 assert_eq!(chunk.value_at(1, 1), Value::Null);
14233 }
14234 }
14235 let above = [Probe { column: 0, op: Op::Greater, value: Bound::Int(100) }];
14238 assert!(reader.skips(0, &above), "the first stripe stops at 63");
14239 assert!(!reader.skips(STRIPE_PARTS * 2, &above), "the third stripe reaches 130");
14240 fs::remove_file(path).expect("remove scratch file");
14241 }
14242
14243 fn scattered(n: i64) -> i64 {
14245 n.wrapping_mul(-7_046_029_254_386_353_131)
14246 }
14247
14248 #[test]
14254 fn a_part_is_skipped_when_its_sieve_does_not_hold_the_constant() {
14255 let path = path("sieve-skip");
14256 let mut writer =
14257 Writer::create(&path, "hits", vec![Field::required("id", LogicalType::BigInt)])
14258 .expect("new file");
14259 let parts = STRIPE_PARTS + 3;
14260 let per_part = 128;
14264 for part in 0..parts {
14265 let held: Vec<Value> = (0..per_part)
14266 .map(|row| Value::BigInt(scattered((part * per_part + row) as i64)))
14267 .collect();
14268 let chunk =
14269 Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
14270 .expect("one column");
14271 writer.append(&chunk).expect("one part");
14272 }
14273 writer.finish().expect("commit");
14274
14275 let reader = Reader::open(&path).expect("reopen from disk");
14276 let probe = |value: i64| Probe {
14277 column: 0,
14278 op: Op::Equal,
14279 value: Bound::Int(i128::from(scattered(value))),
14280 };
14281 for wanted in [0_i64, (per_part + 1) as i64, (parts * per_part - 1) as i64] {
14282 let tests = [probe(wanted)];
14283 let kept: Vec<usize> = (0..parts).filter(|&part| !reader.skips(part, &tests)).collect();
14284 let home = wanted as usize / per_part;
14285 assert!(kept.contains(&home), "the part holding {wanted} is read");
14286 assert!(kept.len() <= 2, "{wanted} keeps {kept:?}, which is more than one stray part");
14290 }
14291 let absent = [probe((parts * per_part) as i64 + 1)];
14292 let kept = (0..parts).filter(|&part| !reader.skips(part, &absent)).count();
14293 assert!(kept <= 1, "{kept} parts of {parts} kept a value no part holds");
14294 let tests = [probe(0)];
14297 assert!(
14298 reader.table().stripes().iter().all(|stripe| !stripe.zone.skips(&tests)),
14299 "the bounds rule out no stripe at all"
14300 );
14301 fs::remove_file(path).expect("remove scratch file");
14302 }
14303
14304 #[test]
14310 fn a_part_is_skipped_when_its_own_bounds_rule_out_a_comparison_the_stripe_keeps() {
14311 let path = path("part-range-skip");
14312 let mut writer =
14313 Writer::create(&path, "hits", vec![Field::required("at", LogicalType::BigInt)])
14314 .expect("new file");
14315 let parts = STRIPE_PARTS + 3;
14316 let per_part = 128;
14317 for part in 0..parts {
14318 let held: Vec<Value> = (0..per_part)
14322 .map(|row| {
14323 Value::BigInt((part * 1_000) as i64 + (scattered(row as i64).rem_euclid(900)))
14324 })
14325 .collect();
14326 let chunk =
14327 Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
14328 .expect("one column");
14329 writer.append(&chunk).expect("one part");
14330 }
14331 writer.finish().expect("commit");
14332
14333 let reader = Reader::open(&path).expect("reopen from disk");
14334 let under = [Probe { column: 0, op: Op::Less, value: Bound::Int(3_000) }];
14335 let kept: Vec<usize> = (0..parts).filter(|&part| !reader.skips(part, &under)).collect();
14336 assert_eq!(kept, vec![0, 1, 2], "only the three parts that start under three thousand");
14337 assert!(!reader.stripe_skips(0, &under), "the stripe reaches from zero and keeps itself");
14339 fs::remove_file(path).expect("remove scratch file");
14340 }
14341
14342 #[test]
14346 fn a_part_is_waved_through_when_its_own_bounds_pass_a_comparison_the_stripe_cannot() {
14347 let path = path("part-range-certain");
14348 let mut writer =
14349 Writer::create(&path, "hits", vec![Field::required("at", LogicalType::BigInt)])
14350 .expect("new file");
14351 let parts = STRIPE_PARTS + 3;
14352 let per_part = 128;
14353 for part in 0..parts {
14354 let held: Vec<Value> = (0..per_part)
14355 .map(|row| {
14356 Value::BigInt((part * 1_000) as i64 + (scattered(row as i64).rem_euclid(900)))
14357 })
14358 .collect();
14359 let chunk =
14360 Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
14361 .expect("one column");
14362 writer.append(&chunk).expect("one part");
14363 }
14364 writer.finish().expect("commit");
14365
14366 let reader = Reader::open(&path).expect("reopen from disk");
14367 let under = [Probe { column: 0, op: Op::Less, value: Bound::Int(3_000) }];
14368 let waved: Vec<usize> = (0..parts).filter(|&part| reader.certain(part, &under)).collect();
14369 assert_eq!(waved, vec![0, 1, 2], "the three parts that end under three thousand");
14370 assert!(!reader.stripe_skips(0, &under), "the stripe straddles the comparison");
14373 fs::remove_file(path).expect("remove scratch file");
14374 }
14375
14376 #[test]
14379 fn a_stripe_of_one_part_writes_no_range_page_and_a_stripe_of_many_does() {
14380 for (parts, wanted) in [(1_usize, false), (STRIPE_PARTS, true)] {
14381 let path = path("part-range-page");
14382 let mut writer =
14383 Writer::create(&path, "hits", vec![Field::required("at", LogicalType::BigInt)])
14384 .expect("new file");
14385 for part in 0..parts {
14386 let held: Vec<Value> = (0..128)
14387 .map(|row| {
14388 Value::BigInt((part * 1_000) as i64 + scattered(row as i64).rem_euclid(900))
14389 })
14390 .collect();
14391 let chunk = Chunk::new(vec![
14392 Vector::from_values(LogicalType::BigInt, &held).expect("numbers"),
14393 ])
14394 .expect("one column");
14395 writer.append(&chunk).expect("one part");
14396 }
14397 writer.finish().expect("commit");
14398 let reader = Reader::open(&path).expect("reopen from disk");
14399 let bytes = reader.layout().columns[0].part_ranges;
14400 assert_eq!(bytes > 0, wanted, "{parts} parts wrote {bytes} bytes of ranges");
14401 fs::remove_file(path).expect("remove scratch file");
14402 }
14403 }
14404
14405 #[test]
14408 fn a_string_end_that_is_cut_down_still_covers_the_value_it_came_from() {
14409 let long = vec![b'a'; PART_BOUND_BYTES * 2];
14410 let low = shortened(Some(Bound::Bytes(long.clone())), false).expect("a low end");
14411 let high = shortened(Some(Bound::Bytes(long.clone())), true).expect("a high end");
14412 let Bound::Bytes(low) = low else { panic!("a string stays a string") };
14413 let Bound::Bytes(high) = high else { panic!("a string stays a string") };
14414 assert!(low.len() <= PART_BOUND_BYTES && high.len() <= PART_BOUND_BYTES);
14415 assert!(low.as_slice() <= long.as_slice(), "the low end is at or under the value");
14416 assert!(high.as_slice() >= long.as_slice(), "the high end is at or over the value");
14417 }
14418
14419 #[test]
14422 fn a_string_end_with_no_room_to_step_up_gives_up_the_bound() {
14423 let long = vec![u8::MAX; PART_BOUND_BYTES * 2];
14424 assert_eq!(shortened(Some(Bound::Bytes(long.clone())), true), None);
14425 let low = shortened(Some(Bound::Bytes(long)), false).expect("a low end is still a prefix");
14426 assert_eq!(low, Bound::Bytes(vec![u8::MAX; PART_BOUND_BYTES]));
14427 }
14428
14429 #[test]
14441 fn what_a_column_is_stored_as_follows_the_order_the_rows_were_written_in() {
14442 let parts = 4;
14443 let per_part = 1024;
14444 let rows = parts * per_part;
14445 let written = |name: &str, keys: &[i64]| {
14446 let path = path(name);
14447 let fields = vec![Field::required("key", LogicalType::BigInt)];
14448 let mut writer = Writer::create(&path, "keys", fields).expect("new file");
14449 for part in 0..parts {
14450 let values: Vec<Value> = keys[part * per_part..(part + 1) * per_part]
14451 .iter()
14452 .map(|key| Value::BigInt(*key))
14453 .collect();
14454 let chunk = Chunk::new(vec![
14455 Vector::from_values(LogicalType::BigInt, &values).expect("numbers"),
14456 ])
14457 .expect("one column");
14458 writer.append(&chunk).expect("one part");
14459 }
14460 writer.finish().expect("commit");
14461 path
14462 };
14463 let climbing = |step: &dyn Fn(usize) -> i64| {
14466 let mut key = 0;
14467 (0..rows)
14468 .map(|row| {
14469 key += step(row);
14470 key
14471 })
14472 .collect::<Vec<i64>>()
14473 };
14474 let ascending = climbing(&|row| (row % 3) as i64);
14475 let sparse = climbing(&|row| ((row * 2_654_435_761) % 4096) as i64);
14479 let near_path = written("stored-near", &ascending);
14480 let far_path = written("stored-far", &sparse);
14481
14482 let one = Reader::open(&near_path).expect("reopen from disk");
14483 let other = Reader::open(&far_path).expect("reopen from disk");
14484 let near = one.stored(0).expect("the column is stored");
14485 let far = other.stored(0).expect("the column is stored");
14486 assert_eq!(near.len(), parts, "one row per part");
14487 assert_eq!(far.len(), parts);
14488 let total = |stored: &[StoredPart]| stored.iter().map(|part| part.bytes).sum::<u64>();
14491 assert_eq!(total(&near), one.layout().columns[0].pages);
14492 assert_eq!(total(&far), other.layout().columns[0].pages);
14493 assert!(
14494 total(&near) * 2 < total(&far),
14495 "the sparse keys cost more, {} against {}",
14496 total(&far),
14497 total(&near)
14498 );
14499 for (at, part) in near.iter().enumerate() {
14501 assert_eq!(part.part, at);
14502 assert_eq!(part.row, at * per_part);
14503 assert_eq!(part.rows, per_part);
14504 let held = &ascending[at * per_part..(at + 1) * per_part];
14505 assert_eq!(part.low, Some(Value::BigInt(held[0])));
14506 assert_eq!(part.high, Some(Value::BigInt(held[per_part - 1])));
14507 assert_eq!(part.nulls, Some(0));
14508 }
14509 assert!(near[0].encoding.contains("DELTA"), "{}", near[0].encoding);
14512 assert!(far[0].encoding.contains("DELTA"), "{}", far[0].encoding);
14513 assert_ne!(near[0].encoding, far[0].encoding);
14514 fs::remove_file(near_path).expect("remove scratch file");
14515 fs::remove_file(far_path).expect("remove scratch file");
14516 }
14517
14518 #[test]
14528 fn a_sieve_larger_than_the_part_it_indexes_is_not_written() {
14529 let path = path("sieve-pays");
14530 let fields = vec![
14531 Field::required("spread", LogicalType::BigInt),
14532 Field::required("repeated", LogicalType::BigInt),
14533 ];
14534 let mut writer = Writer::create(&path, "hits", fields).expect("new file");
14535 let parts = 3;
14536 let per_part = 1024;
14537 for part in 0..parts {
14538 let base = (part * per_part) as i64;
14539 let spread: Vec<Value> =
14540 (0..per_part).map(|row| Value::BigInt(scattered(base + row as i64))).collect();
14541 let repeated: Vec<Value> =
14542 (0..per_part).map(|row| Value::BigInt(scattered((row / 256) as i64))).collect();
14543 let chunk = Chunk::new(vec![
14544 Vector::from_values(LogicalType::BigInt, &spread).expect("numbers"),
14545 Vector::from_values(LogicalType::BigInt, &repeated).expect("numbers"),
14546 ])
14547 .expect("two columns");
14548 writer.append(&chunk).expect("one part");
14549 }
14550 writer.finish().expect("commit");
14551
14552 let reader = Reader::open(&path).expect("reopen from disk");
14553 let layout = reader.layout();
14554 let spread = &layout.columns[0];
14555 let repeated = &layout.columns[1];
14556 assert!(spread.sieves > 0, "a column whose parts are worth a filter keeps one");
14557 assert_eq!(
14558 repeated.sieves, 0,
14559 "a column whose filter costs more than its parts keeps none"
14560 );
14561 for column in &layout.columns {
14564 assert!(
14565 column.sieves < column.pages,
14566 "{} spends {} on sieves over {} of data",
14567 column.name,
14568 column.sieves,
14569 column.pages
14570 );
14571 }
14572 let absent = [Probe {
14574 column: 0,
14575 op: Op::Equal,
14576 value: Bound::Int(i128::from(scattered((parts * per_part) as i64 + 1))),
14577 }];
14578 assert!((0..parts).all(|part| reader.skips(part, &absent)), "no part holds it");
14579 fs::remove_file(path).expect("remove scratch file");
14580 }
14581
14582 #[test]
14588 fn a_damaged_sieve_page_is_read_through_rather_than_refused() {
14589 let path = path("sieve-damaged");
14590 let mut writer =
14591 Writer::create(&path, "hits", vec![Field::required("id", LogicalType::BigInt)])
14592 .expect("new file");
14593 let rows = 128;
14594 let held: Vec<Value> = (0..rows).map(|row| Value::BigInt(scattered(row))).collect();
14595 let chunk =
14596 Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
14597 .expect("one column");
14598 writer.append(&chunk).expect("one part");
14599 writer.finish().expect("commit");
14600
14601 let page = Reader::open(&path).expect("reopen").table.stripes[0]
14602 .sieves
14603 .get(0)
14604 .expect("a sieve page");
14605 let mut file = OpenOptions::new().write(true).open(&path).expect("open the sieve page");
14606 file.seek(SeekFrom::Start(page.offset + u64::from(page.length) - 1)).expect("seek");
14607 file.write_all(&[0xff]).expect("damage one byte");
14608 drop(file);
14609
14610 let reader = Reader::open(&path).expect("reopen the damaged file");
14611 let absent =
14612 [Probe { column: 0, op: Op::Equal, value: Bound::Int(i128::from(scattered(99))) }];
14613 assert!(!reader.skips(0, &absent), "a sieve that cannot be read skips nothing");
14614 assert_eq!(
14615 reader.read(0, &[0]).expect("the rows are untouched").len(),
14616 usize::try_from(rows).expect("a small count")
14617 );
14618 fs::remove_file(path).expect("remove scratch file");
14619 }
14620
14621 #[test]
14632 fn workers_that_want_the_same_stripe_read_it_once() {
14633 let path = path("single-flight");
14634 let mut writer =
14635 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
14636 .expect("new file");
14637 for part in 0..STRIPE_PARTS {
14638 let id = part as i32;
14639 let chunk = Chunk::new(vec![
14640 Vector::from_values(
14641 LogicalType::Integer,
14642 &[Value::Integer(id), Value::Integer(-id)],
14643 )
14644 .expect("integers"),
14645 ])
14646 .expect("matching rows");
14647 writer.append(&chunk).expect("one part");
14648 }
14649 writer.finish().expect("commit");
14650
14651 let reader = Reader::open(&path).expect("reopen from disk");
14652 assert_eq!(reader.table().stripes().len(), 1, "one stripe is the point of the test");
14653 let barrier = std::sync::Barrier::new(8);
14654 std::thread::scope(|scope| {
14655 for worker in 0..8 {
14656 let reader = &reader;
14657 let barrier = &barrier;
14658 scope.spawn(move || {
14659 barrier.wait();
14660 for part in (worker..STRIPE_PARTS).step_by(8) {
14661 let chunk = reader.read(part, &[0]).expect("a whole page read");
14662 assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
14663 assert_eq!(chunk.value_at(1, 0), Value::Integer(-(part as i32)));
14664 }
14665 });
14666 }
14667 });
14668 assert_eq!(reader.pages.load(Atomic::Relaxed), 1, "one stripe, one page read, whoever won");
14669 fs::remove_file(path).expect("remove scratch file");
14670 }
14671
14672 #[test]
14685 fn opening_costs_the_same_over_a_thousand_times_the_rows() {
14686 let opened = |label: &str, rows_per_part: i32| {
14687 let path = path(label);
14688 let mut writer =
14689 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
14690 .expect("new file");
14691 for part in 0..STRIPE_PARTS * 3 {
14692 let values = (0..rows_per_part)
14696 .map(|row| {
14697 Value::Integer((part as i32 * rows_per_part + row).wrapping_mul(2_654_435))
14698 })
14699 .collect::<Vec<_>>();
14700 let chunk = Chunk::new(vec![
14701 Vector::from_values(LogicalType::Integer, &values).expect("integers"),
14702 ])
14703 .expect("matching rows");
14704 writer.append(&chunk).expect("one part");
14705 }
14706 writer.finish().expect("commit");
14707 let reader = Reader::open(&path).expect("reopen from disk");
14708 let size = fs::metadata(&path).expect("the file is there").len();
14709 let out = (reader.reads(), reader.table().stripes().len(), size);
14710 fs::remove_file(path).expect("remove scratch file");
14711 out
14712 };
14713
14714 let (thin, thin_stripes, thin_size) = opened("open-thin", 1);
14715 let (fat, fat_stripes, fat_size) = opened("open-fat", 1000);
14716 assert_eq!(
14717 thin_stripes, fat_stripes,
14718 "the same stripe count is what makes this a fair ask"
14719 );
14720 assert!(
14721 fat_size > thin_size * 50,
14722 "the fat file has to actually be larger, and it is {fat_size} against {thin_size}"
14723 );
14724
14725 assert_eq!(thin.opening.reads, fat.opening.reads, "the same reads either way");
14726 assert_eq!(thin.pages, 0, "opening read a page");
14727 assert_eq!(fat.pages, 0, "opening read a page");
14728 assert_eq!(thin.indexes, 0, "opening read an index");
14729 assert_eq!(fat.indexes, 0, "opening read an index");
14730 assert!(
14733 fat.opening.bytes < thin.opening.bytes * 2,
14734 "opening the thin file read {} bytes and the fat one read {}",
14735 thin.opening.bytes,
14736 fat.opening.bytes
14737 );
14738 }
14739
14740 #[test]
14748 fn two_opens_of_one_file_cost_the_same_and_the_second_is_not_cheaper() {
14749 let path = path("open-twice");
14750 let mut writer =
14751 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
14752 .expect("new file");
14753 for part in 0..STRIPE_PARTS * 3 {
14754 let chunk = Chunk::new(vec![
14755 Vector::from_values(LogicalType::Integer, &[Value::Integer(part as i32)])
14756 .expect("integers"),
14757 ])
14758 .expect("matching rows");
14759 writer.append(&chunk).expect("one part");
14760 }
14761 writer.finish().expect("commit");
14762
14763 let first = Reader::open(&path).expect("open");
14764 for part in 0..first.parts() {
14767 first.read(part, &[0]).expect("a part");
14768 }
14769 assert!(first.reads().pages > 0, "the scan has to have read something");
14770 let second = Reader::open(&path).expect("open again");
14771
14772 assert_eq!(first.reads().opening, second.reads().opening);
14773 assert_eq!(
14774 second.reads().pages,
14775 0,
14776 "the second open read a page off the back of the first"
14777 );
14778 assert_eq!(second.reads().indexes, 0, "the second open read an index it inherited");
14779 fs::remove_file(path).expect("remove scratch file");
14780 }
14781
14782 #[test]
14790 fn an_index_is_read_once_per_stripe_however_often_the_page_is_evicted() {
14791 let path = path("index-cache");
14792 let mut writer =
14793 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
14794 .expect("new file");
14795 let parts = STRIPE_PARTS * (CACHED_STRIPES_PER_COLUMN + 2);
14796 for part in 0..parts {
14797 let id = part as i32;
14798 let chunk = Chunk::new(vec![
14799 Vector::from_values(LogicalType::Integer, &[Value::Integer(id)]).expect("integers"),
14800 ])
14801 .expect("matching rows");
14802 writer.append(&chunk).expect("one part");
14803 }
14804 writer.finish().expect("commit");
14805
14806 let reader = Reader::open(&path).expect("reopen from disk");
14807 let stripes = reader.table().stripes().len();
14808 assert!(stripes > CACHED_STRIPES_PER_COLUMN, "the page cache has to be too small for this");
14809 for _ in 0..2 {
14811 for part in 0..parts {
14812 let chunk = reader.read(part, &[0]).expect("a part");
14813 assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
14814 }
14815 }
14816 assert_eq!(reader.indexes.load(Atomic::Relaxed), stripes, "one index read per stripe");
14817 assert!(
14818 reader.pages.load(Atomic::Relaxed) > stripes,
14819 "the pages are the ones that get read again, which is what makes the index count mean \
14820 something"
14821 );
14822 fs::remove_file(path).expect("remove scratch file");
14823 }
14824
14825 #[test]
14832 fn a_pool_keeps_pages_between_scans_and_gives_them_to_the_table_being_read() {
14833 let path = path("page-pool");
14834 let parts = STRIPE_PARTS * (CACHED_STRIPES_PER_COLUMN * 2 + 2);
14835 let fields = || vec![Field::required("id", LogicalType::Integer)];
14836 let mut writer = Writer::create(&path, "a", fields()).expect("new file");
14837 for table in ["a", "b"] {
14838 if table == "b" {
14839 writer = writer.next("b".to_string(), fields()).expect("a second table");
14840 }
14841 for part in 0..parts {
14842 let chunk = Chunk::new(vec![
14843 Vector::from_values(LogicalType::Integer, &[Value::Integer(part as i32)])
14844 .expect("integers"),
14845 ])
14846 .expect("matching rows");
14847 writer.append(&chunk).expect("one part");
14848 }
14849 }
14850 writer.finish().expect("commit");
14851
14852 let pool = PagePool::new(usize::MAX);
14853 let catalog = Catalog::open_in(&path, &pool).expect("the file opens");
14854 let (a, b) = (catalog.table("a").expect("a"), catalog.table("b").expect("b"));
14855 let stripes = a.table().stripes().len();
14856 assert!(
14857 stripes > CACHED_STRIPES_PER_COLUMN * 2,
14858 "the floor has to be smaller than a table"
14859 );
14860 let scan = |reader: &Reader| {
14861 for part in 0..parts {
14862 let chunk = reader.read(part, &[0]).expect("a part");
14863 assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
14864 }
14865 };
14866 scan(&a);
14869 assert_eq!(pool.bytes(), 0, "a page read once is not the pool's");
14870 scan(&a);
14871 let twice = stripes * 2 - CACHED_STRIPES_PER_COLUMN;
14872 assert_eq!(a.pages.load(Atomic::Relaxed), twice, "the second scan reads the rest again");
14873 scan(&a);
14874 assert_eq!(a.pages.load(Atomic::Relaxed), twice, "the third scan reads nothing");
14875 let one = pool.bytes();
14876 assert!(one > 0, "the pool counts what the reader holds");
14877
14878 pool.budget.store(one, Atomic::Relaxed);
14880 scan(&b);
14881 scan(&b);
14882 assert_eq!(b.pages.load(Atomic::Relaxed), twice, "a page is never let go while in use");
14883 assert_eq!(a.cache.held[0].load(Atomic::Relaxed), CACHED_STRIPES_PER_COLUMN);
14884 let column = a.cache.columns[0].lock().expect("the column");
14885 let held = column.pages.iter().flatten().count();
14886 assert_eq!(
14887 held,
14888 CACHED_STRIPES_PER_COLUMN + column.passing.len(),
14889 "the count and the slots agree"
14890 );
14891 drop(column);
14892
14893 drop((a, b, catalog));
14895 let c = Catalog::open_in(&path, &pool).expect("again").table("a").expect("a");
14896 scan(&c);
14897 scan(&c);
14898 assert!(pool.bytes() <= one, "only what the live reader holds is counted");
14899 fs::remove_file(path).expect("remove scratch file");
14900 }
14901
14902 #[test]
14911 fn a_worker_per_stripe_reads_its_page_once_when_the_cache_was_told_to_expect_it() {
14912 let workers = CACHED_STRIPES_PER_COLUMN + 4;
14913 let path = path("stripe-per-worker");
14914 let mut writer =
14915 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
14916 .expect("new file");
14917 for part in 0..STRIPE_PARTS * workers {
14918 let chunk = Chunk::new(vec![
14919 Vector::from_values(LogicalType::Integer, &[Value::Integer(part as i32)])
14920 .expect("integers"),
14921 ])
14922 .expect("matching rows");
14923 writer.append(&chunk).expect("one part");
14924 }
14925 writer.finish().expect("commit");
14926
14927 let read = |told: bool| {
14928 let reader = Reader::open(&path).expect("reopen from disk");
14929 assert_eq!(reader.table().stripes().len(), workers, "a stripe per worker");
14930 if told {
14931 reader.keep_stripes(workers);
14932 }
14933 let barrier = std::sync::Barrier::new(workers);
14934 std::thread::scope(|scope| {
14935 for (worker, run) in reader.stripe_parts().into_iter().enumerate() {
14936 let reader = &reader;
14937 let barrier = &barrier;
14938 scope.spawn(move || {
14939 for part in run {
14940 barrier.wait();
14941 let chunk = reader.read(part, &[0]).expect("a part of my own stripe");
14942 assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
14943 }
14944 assert!(worker < workers);
14945 });
14946 }
14947 });
14948 reader.pages.load(Atomic::Relaxed)
14949 };
14950
14951 assert_eq!(read(true), workers, "one page read per stripe and no more");
14952 assert!(read(false) > workers, "a cache that small is read again on every part");
14953 fs::remove_file(path).expect("remove scratch file");
14954 }
14955
14956 #[test]
14961 fn a_damaged_index_page_is_an_error() {
14962 let path = path("damaged-index");
14963 let mut writer =
14964 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
14965 .expect("new file");
14966 writer.append(&sample_ids()).expect("first part");
14967 writer.append(&sample_ids()).expect("second part");
14968 writer.finish().expect("commit");
14969
14970 let reader = Reader::open(&path).expect("valid directory");
14971 let index = reader.table.stripes[0].index;
14972 let mut byte = [0; 1];
14973 read_at(&reader.file, index.offset, &mut byte).expect("the first part length");
14974 let mut file = OpenOptions::new().write(true).open(&path).expect("open index page");
14975 file.seek(SeekFrom::Start(index.offset)).expect("index start");
14976 file.write_all(&[!byte[0]]).expect("damage the first part length");
14977 let error = reader.read(1, &[0]).expect_err("a damaged index must not be used");
14978 assert!(error.message().contains("index page section checksum differs"), "{error}");
14979 fs::remove_file(path).expect("remove scratch file");
14980 }
14981
14982 #[test]
14989 fn every_integer_width_round_trips_through_a_page() {
14990 let path = path("integer-widths");
14991 let columns = [
14992 (LogicalType::TinyInt, vec![Value::TinyInt(i8::MIN), Value::TinyInt(i8::MAX)]),
14993 (LogicalType::UTinyInt, vec![Value::UTinyInt(0), Value::UTinyInt(u8::MAX)]),
14994 (LogicalType::SmallInt, vec![Value::SmallInt(i16::MIN), Value::SmallInt(i16::MAX)]),
14995 (LogicalType::USmallInt, vec![Value::USmallInt(0), Value::USmallInt(u16::MAX)]),
14996 (LogicalType::Integer, vec![Value::Integer(i32::MIN), Value::Integer(i32::MAX)]),
14997 (LogicalType::UInteger, vec![Value::UInteger(0), Value::UInteger(u32::MAX)]),
14998 (LogicalType::BigInt, vec![Value::BigInt(i64::MIN), Value::BigInt(i64::MAX)]),
14999 (LogicalType::UBigInt, vec![Value::UBigInt(0), Value::UBigInt(u64::MAX)]),
15000 ];
15001 let fields = columns
15002 .iter()
15003 .enumerate()
15004 .map(|(at, (ty, _))| Field::required(format!("c{at}"), ty.clone()))
15005 .collect::<Vec<_>>();
15006 let vectors = columns
15007 .iter()
15008 .map(|(ty, values)| Vector::from_values(ty.clone(), values).expect("a vector"))
15009 .collect::<Vec<_>>();
15010 let mut writer = Writer::create(&path, "widths", fields).expect("new file");
15011 writer.append(&Chunk::new(vectors).expect("matching rows")).expect("one stripe");
15012 writer.finish().expect("commit");
15013
15014 let reader = Reader::open(&path).expect("reopen from disk");
15015 let wanted = (0..columns.len()).collect::<Vec<_>>();
15016 let read = reader.read(0, &wanted).expect("every column");
15017 assert_eq!(read.len(), 2);
15018 for (at, (ty, values)) in columns.iter().enumerate() {
15020 assert_eq!(read.value_at(0, at), values[0], "the low end of {ty}");
15021 assert_eq!(read.value_at(1, at), values[1], "the high end of {ty}");
15022 }
15023 fs::remove_file(path).expect("remove scratch file");
15024 }
15025
15026 #[test]
15037 fn every_other_type_the_format_knows_round_trips_through_a_page() {
15038 let path = path("other-types");
15039 let columns = [
15040 (LogicalType::Float, vec![Value::Float(f32::MIN), Value::Float(-0.0)]),
15041 (LogicalType::Double, vec![Value::Double(f64::MIN), Value::Double(f64::MAX)]),
15042 (LogicalType::HugeInt, vec![Value::HugeInt(i128::MIN), Value::HugeInt(i128::MAX)]),
15043 (LogicalType::UHugeInt, vec![Value::UHugeInt(0), Value::UHugeInt(u128::MAX)]),
15044 (LogicalType::Time, vec![Value::Time(0), Value::Time(86_399_999_999)]),
15045 (LogicalType::TimeTz, vec![Value::TimeTz(-50_400_000_000), Value::TimeTz(0)]),
15046 (
15047 LogicalType::TimestampTz,
15048 vec![Value::TimestampTz(i64::MIN + 1), Value::TimestampTz(i64::MAX)],
15049 ),
15050 (
15051 LogicalType::Interval,
15052 vec![
15053 Value::Interval { months: i32::MIN, days: i32::MAX, micros: i64::MIN },
15054 Value::Interval { months: 13, days: -1, micros: 1 },
15055 ],
15056 ),
15057 (
15058 LogicalType::Blob,
15059 vec![Value::Blob(vec![0, 0xff, 0x80, 0xfe]), Value::Blob(Vec::new())],
15060 ),
15061 ];
15062 let fields = columns
15063 .iter()
15064 .enumerate()
15065 .map(|(at, (ty, _))| Field::required(format!("c{at}"), ty.clone()))
15066 .collect::<Vec<_>>();
15067 let vectors = columns
15068 .iter()
15069 .map(|(ty, values)| Vector::from_values(ty.clone(), values).expect("a vector"))
15070 .collect::<Vec<_>>();
15071 let mut writer = Writer::create(&path, "others", fields).expect("new file");
15072 writer.append(&Chunk::new(vectors).expect("matching rows")).expect("one stripe");
15073 writer.finish().expect("commit");
15074
15075 let reader = Reader::open(&path).expect("reopen from disk");
15076 let wanted = (0..columns.len()).collect::<Vec<_>>();
15077 let read = reader.read(0, &wanted).expect("every column");
15078 assert_eq!(read.len(), 2);
15079 for (at, (ty, values)) in columns.iter().enumerate() {
15080 assert_eq!(read.value_at(0, at), values[0], "the low end of {ty}");
15081 assert_eq!(read.value_at(1, at), values[1], "the high end of {ty}");
15082 }
15083 let Value::Float(zero) = read.value_at(1, 0) else { panic!("a float stays a float") };
15086 assert!(zero.is_sign_negative(), "a negative zero came back as {zero}");
15087
15088 fs::remove_file(path).expect("remove scratch file");
15089 }
15090
15091 #[test]
15097 fn a_nan_survives_being_written_down() {
15098 let path = path("nan");
15099 let nan = Vector::from_values(LogicalType::Double, &[Value::Double(f64::NAN)])
15100 .expect("a NaN vector");
15101 let mut writer =
15102 Writer::create(&path, "nan", vec![Field::required("d", LogicalType::Double)])
15103 .expect("new file");
15104 writer.append(&Chunk::new(vec![nan]).expect("one column")).expect("one stripe");
15105 writer.finish().expect("commit");
15106 let read = Reader::open(&path).expect("reopen").read(0, &[0]).expect("the column");
15107 let Value::Double(back) = read.value_at(0, 0) else { panic!("a double stays a double") };
15108 assert!(back.is_nan(), "a NaN came back as {back}");
15109 fs::remove_file(path).expect("remove scratch file");
15110 }
15111
15112 #[test]
15119 fn a_uuid_and_a_bit_string_come_back_as_the_bits_that_went_in() {
15120 let path = path("uuid-and-bit");
15121 let uuids = vec![0_i128, i128::MIN, -1];
15122 let mut bits = StringColumn::new();
15123 for value in [&b"\x02\xff"[..], &b""[..], &b"\x00\x01\x02\x03\x04\x05"[..]] {
15124 bits.push_bytes(value);
15125 }
15126 let expected = bits.clone();
15127 let fields =
15128 vec![Field::required("u", LogicalType::Uuid), Field::required("b", LogicalType::Bit)];
15129 let vectors = vec![
15130 Vector::flat(LogicalType::Uuid, Data::Int128(uuids.clone().into())).expect("uuids"),
15131 Vector::flat(LogicalType::Bit, Data::Varlen(bits)).expect("bit strings"),
15132 ];
15133 let mut writer = Writer::create(&path, "ids", fields).expect("new file");
15134 writer.append(&Chunk::new(vectors).expect("matching rows")).expect("one stripe");
15135 writer.finish().expect("commit");
15136
15137 let reader = Reader::open(&path).expect("reopen from disk");
15138 let read = reader.read(0, &[0, 1]).expect("both columns").flatten().expect("flat");
15139 let Some(Data::Int128(back)) = read.column(0).expect("the uuids").data() else {
15140 panic!("a uuid column is the 128 bit lane")
15141 };
15142 assert_eq!(back.as_slice(), uuids.as_slice());
15143 let Some(Data::Varlen(back)) = read.column(1).expect("the bits").data() else {
15144 panic!("a bit column is bytes")
15145 };
15146 for row in 0..expected.len() {
15147 assert_eq!(back.bytes(row), expected.bytes(row), "row {row} of the bit column");
15148 }
15149 fs::remove_file(path).expect("remove scratch file");
15150 }
15151
15152 #[test]
15155 fn a_run_counted_at_once_leaves_the_candidates_a_row_at_a_time_would() {
15156 let mut rows: Vec<Option<u64>> = Vec::new();
15157 let mut state = 0x2545_f491_4f6c_dd1d_u64;
15158 for index in 0..400_000_u64 {
15159 state ^= state << 13;
15160 state ^= state >> 7;
15161 state ^= state << 17;
15162 let times = 1 + (state % 7) as usize;
15163 let bits = match state % 11 {
15164 0 => None,
15165 1..=3 => Some(state % 16),
15166 _ => Some(index.wrapping_mul(0x9e37_79b9_7f4a_7c15)),
15167 };
15168 rows.extend(std::iter::repeat_n(bits, times));
15169 }
15170 let mut by_row = Candidates::default();
15171 for &bits in &rows {
15172 by_row.add(bits, 1);
15173 }
15174 let mut by_run = Candidates::default();
15175 let mut run = Run::default();
15176 let mut runs = 0_usize;
15177 for &bits in &rows {
15178 if let Some((bits, times)) = run.push(bits) {
15179 by_run.add(bits, times);
15180 runs += 1;
15181 }
15182 }
15183 if let Some((bits, times)) = run.take() {
15184 by_run.add(bits, times);
15185 }
15186 assert!(runs < rows.len() / 2, "the rows came in runs");
15187 assert!(by_row.decrements > 0, "the table filled and turned values away");
15188 assert_eq!(sorted_candidates(&by_run), sorted_candidates(&by_row));
15189 assert_eq!(by_run.nulls, by_row.nulls);
15190 assert_eq!(by_run.decrements, by_row.decrements);
15191 }
15192
15193 fn sorted_candidates(candidates: &Candidates) -> Vec<(u64, u32)> {
15194 let mut pairs = candidates.pairs().collect::<Vec<_>>();
15195 pairs.sort_unstable();
15196 assert_eq!(pairs.len(), candidates.held, "the count of held slots drifted");
15197 pairs
15198 }
15199
15200 #[derive(Default)]
15203 struct MapCandidates {
15204 counts: HashMap<u64, u32>,
15205 nulls: u32,
15206 decrements: u64,
15207 }
15208
15209 impl MapCandidates {
15210 fn add(&mut self, bits: Option<u64>, mut times: u32) {
15211 while times > 0 {
15212 let held = match bits {
15213 Some(bits) => self.counts.get_mut(&bits),
15214 None if self.nulls != 0 => Some(&mut self.nulls),
15215 None => None,
15216 };
15217 if let Some(count) = held {
15218 *count = count.saturating_add(times);
15219 return;
15220 }
15221 if self.counts.len() + usize::from(self.nulls != 0) < FREQUENCY_CANDIDATES {
15222 match bits {
15223 Some(bits) => {
15224 self.counts.insert(bits, times);
15225 }
15226 None => self.nulls = times,
15227 }
15228 return;
15229 }
15230 self.counts.retain(|_, count| {
15231 *count -= 1;
15232 *count != 0
15233 });
15234 self.nulls = self.nulls.saturating_sub(1);
15235 self.decrements = self.decrements.saturating_add(1);
15236 times -= 1;
15237 }
15238 }
15239 }
15240
15241 #[test]
15245 fn the_open_addressed_candidates_agree_with_the_map_they_replaced() {
15246 for seed in [0x2545_f491_4f6c_dd1d_u64, 0x9e37_79b9_7f4a_7c15, 7] {
15247 let mut table = Candidates::default();
15248 let mut oracle = MapCandidates::default();
15249 let mut state = seed;
15250 for index in 0..300_000_u64 {
15251 state ^= state << 13;
15252 state ^= state >> 7;
15253 state ^= state << 17;
15254 let bits = match state % 13 {
15255 0 => None,
15256 1..=4 => Some(state % 40),
15257 5 => Some((index % 1000) * 1_000_000),
15258 _ => Some(state),
15259 };
15260 let times = 1 + (state >> 60) as u32 % 3;
15261 table.add(bits, times);
15262 oracle.add(bits, times);
15263 if index % 50_000 == 0 {
15264 let mut expected =
15265 oracle.counts.iter().map(|(&b, &c)| (b, c)).collect::<Vec<_>>();
15266 expected.sort_unstable();
15267 assert_eq!(sorted_candidates(&table), expected, "seed {seed} row {index}");
15268 }
15269 }
15270 let mut expected = oracle.counts.iter().map(|(&b, &c)| (b, c)).collect::<Vec<_>>();
15271 expected.sort_unstable();
15272 assert_eq!(sorted_candidates(&table), expected, "seed {seed}");
15273 assert_eq!(table.nulls, oracle.nulls, "seed {seed}");
15274 assert_eq!(table.decrements, oracle.decrements, "seed {seed}");
15275 assert!(table.decrements > 0, "seed {seed} never filled the table");
15276 for &(bits, _) in &expected {
15277 assert!(table.position(bits).is_some(), "seed {seed} lost {bits}");
15278 }
15279 }
15280 }
15281
15282 #[test]
15283 fn numeric_frequency_candidates_keep_bounded_row_ordinals() {
15284 let path = path("frequency-ordinals");
15285 let mut writer =
15286 Writer::create(&path, "items", vec![Field::required("id", LogicalType::BigInt)])
15287 .expect("new file");
15288 let mut values = Vec::new();
15289 for leader in 0..10_i64 {
15290 values.extend(std::iter::repeat_n(leader, 100));
15291 }
15292 values.extend(1_000_i64..41_000);
15293 for part in values.chunks(1_024) {
15294 let vector = Vector::flat(LogicalType::BigInt, Data::Int64(part.to_vec().into()))
15295 .expect("big integers");
15296 writer.append(&Chunk::new(vec![vector]).expect("one column")).expect("one stripe");
15297 }
15298 writer.finish().expect("commit");
15299
15300 let reader = Reader::open(&path).expect("reopen from disk");
15301 let occurrences =
15302 reader.frequency_occurrences(0).expect("valid metadata").expect("bounded ordinals");
15303 assert!(occurrences.omitted_max < 100);
15304 assert!(occurrences.ordinals.len() <= FREQUENCY_ORDINALS);
15305 assert_eq!(occurrences.anchor_indices.len(), occurrences.ordinals.len());
15306 assert!(occurrences.ordinals.windows(2).all(|pair| pair[0] < pair[1]));
15307 assert_eq!(&occurrences.ordinals[..1_000], &(0_u64..1_000).collect::<Vec<_>>());
15308 assert_eq!(
15309 &occurrences.anchor_indices[..1_000]
15310 .iter()
15311 .map(|&entry| occurrences.anchors[entry as usize].clone())
15312 .collect::<Vec<_>>(),
15313 &(0_i64..10)
15314 .flat_map(|leader| std::iter::repeat_n(Value::BigInt(leader), 100))
15315 .collect::<Vec<_>>()
15316 );
15317 fs::remove_file(path).expect("remove scratch file");
15318 }
15319
15320 #[test]
15321 fn numeric_frequencies_count_nulls_and_values_past_the_top_of_bigint() {
15322 let path = path("frequency-bits");
15327 let mut writer = Writer::create(
15328 &path,
15329 "items",
15330 vec![Field::new("u", LogicalType::UBigInt), Field::new("s", LogicalType::BigInt)],
15331 )
15332 .expect("new file");
15333 let mut rows = Vec::new();
15334 let mut leaders = Vec::new();
15335 for leader in 0..10_u64 {
15336 let count = 300 - leader * 10;
15337 let (unsigned, signed) = if leader == 0 {
15338 (Value::Null, Value::Null)
15339 } else {
15340 (Value::UBigInt(u64::MAX - leader), Value::BigInt(-(leader as i64)))
15341 };
15342 rows.extend(std::iter::repeat_n((unsigned.clone(), signed.clone()), count as usize));
15343 leaders.push(((unsigned, count), (signed, count)));
15344 }
15345 rows.extend((1_000..41_000_u64).map(|id| (Value::UBigInt(id), Value::BigInt(id as i64))));
15346 for part in rows.chunks(1_024) {
15347 let unsigned = part.iter().map(|(value, _)| value.clone()).collect::<Vec<_>>();
15348 let signed = part.iter().map(|(_, value)| value.clone()).collect::<Vec<_>>();
15349 let chunk = Chunk::new(vec![
15350 Vector::from_values(LogicalType::UBigInt, &unsigned).expect("unsigned"),
15351 Vector::from_values(LogicalType::BigInt, &signed).expect("signed"),
15352 ])
15353 .expect("matching columns");
15354 writer.append(&chunk).expect("rows");
15355 }
15356 writer.finish().expect("commit");
15357
15358 let reader = Reader::open(&path).expect("reopen from disk");
15359 for column in 0..2 {
15360 let prefix =
15361 reader.frequency_prefix(column).expect("valid metadata").expect("a synopsis");
15362 let wanted = leaders
15363 .iter()
15364 .map(|(unsigned, signed)| if column == 0 { unsigned } else { signed })
15365 .cloned()
15366 .collect::<Vec<_>>();
15367 assert_eq!(&prefix.entries[..10], &wanted[..], "column {column}");
15368 assert!(prefix.omitted_max < 210, "column {column}");
15369 assert_eq!(
15370 reader.distinct_values(column).expect("valid metadata"),
15371 Some(9 + 40_000),
15372 "column {column}"
15373 );
15374 }
15375 fs::remove_file(path).expect("remove scratch file");
15376 }
15377
15378 #[test]
15379 fn a_narrow_column_takes_its_frequencies_from_the_tally_and_they_match_the_rows() {
15380 let path = path("frequency-tally");
15386 let types = [
15387 LogicalType::TinyInt,
15388 LogicalType::UInteger,
15389 LogicalType::Date,
15390 LogicalType::Timestamp,
15391 ];
15392 let value = |ty: &LogicalType, at: i64| match ty {
15393 LogicalType::TinyInt => Value::TinyInt((at % 250 - 125) as i8),
15394 LogicalType::UInteger => Value::UInteger(u32::MAX - at as u32),
15395 LogicalType::Date => Value::Date(19_000 - at as i32),
15396 _ => Value::Timestamp(1_700_000_000_000_000 - at * 1_000_003),
15397 };
15398 let fields = types
15399 .iter()
15400 .enumerate()
15401 .map(|(at, ty)| Field::new(format!("c{at}"), ty.clone()))
15402 .collect::<Vec<_>>();
15403 let mut writer = Writer::create(&path, "items", fields).expect("new file");
15404 let mut rows = Vec::new();
15405 for at in 0..250_i64 {
15406 for _ in 0..=(at % 37) {
15407 rows.push(if rows.len() % 13 == 0 { None } else { Some(at) });
15408 }
15409 }
15410 for part in rows.chunks(1_000) {
15411 let columns = types
15412 .iter()
15413 .map(|ty| {
15414 let values = part
15415 .iter()
15416 .map(|row| row.map_or(Value::Null, |at| value(ty, at)))
15417 .collect::<Vec<_>>();
15418 Vector::from_values(ty.clone(), &values).expect("a column")
15419 })
15420 .collect();
15421 writer.append(&Chunk::new(columns).expect("matching columns")).expect("rows");
15422 }
15423 writer.finish().expect("commit");
15424
15425 let reader = Reader::open(&path).expect("reopen from disk");
15426 for (column, ty) in types.iter().enumerate() {
15427 let mut counts = HashMap::<Option<i64>, u64>::new();
15428 for row in &rows {
15429 *counts.entry(*row).or_default() += 1;
15430 }
15431 let wanted = counts
15432 .into_iter()
15433 .map(|(row, count)| (row.map_or(Value::Null, |at| value(ty, at)), count))
15434 .collect::<Vec<_>>();
15435 let prefix =
15436 reader.frequency_prefix(column).expect("valid metadata").expect("a synopsis");
15437 assert_eq!(prefix.entries.len(), 2, "column {column}");
15438 assert!(prefix.omitted_max > 0, "column {column}");
15439 for (value, count) in &prefix.entries {
15440 let held =
15441 wanted.iter().find(|(wanted, _)| wanted == value).map(|(_, count)| count);
15442 assert_eq!(held, Some(count), "column {column} value {value:?}");
15443 }
15444 assert!(prefix.entries.windows(2).all(|pair| pair[0].1 >= pair[1].1));
15445 assert_eq!(
15446 reader.distinct_values(column).expect("valid metadata"),
15447 Some(wanted.len() as u64 - 1),
15448 "column {column}"
15449 );
15450 }
15451 fs::remove_file(path).expect("remove scratch file");
15452 }
15453
15454 #[test]
15455 fn distinct_counts_are_exact_either_side_of_a_full_candidate_table() {
15456 let edge = FREQUENCY_CANDIDATES as i64;
15461 for distinct in [0, 1, 7, edge - 2, edge - 1, edge, edge + 1, edge + 2, 3 * edge] {
15462 for with_null in [false, true] {
15463 let path = path("distinct-edge");
15464 let mut writer =
15465 Writer::create(&path, "items", vec![Field::new("id", LogicalType::BigInt)])
15466 .expect("new file");
15467 let mut values = Vec::new();
15468 for round in 0..2 {
15469 for value in 0..distinct {
15470 let repeat = if round == 0 { 1 + (value % 3) as usize } else { 1 };
15471 values.extend(std::iter::repeat_n(
15472 Value::BigInt(value * 7_919 % distinct),
15473 repeat,
15474 ));
15475 if with_null && value % 1_000 == 0 {
15476 values.push(Value::Null);
15477 }
15478 }
15479 }
15480 if with_null {
15481 values.push(Value::Null);
15482 }
15483 for part in values.chunks(1_024) {
15484 let chunk = Chunk::new(vec![
15485 Vector::from_values(LogicalType::BigInt, part).expect("ids"),
15486 ])
15487 .expect("one column");
15488 writer.append(&chunk).expect("rows");
15489 }
15490 writer.finish().expect("commit");
15491 let reader = Reader::open(&path).expect("reopen from disk");
15492 assert_eq!(
15493 reader.distinct_values(0).expect("valid metadata"),
15494 Some(distinct as u64),
15495 "{distinct} values, null {with_null}"
15496 );
15497 fs::remove_file(path).expect("remove scratch file");
15498 }
15499 }
15500 }
15501
15502 #[test]
15503 fn narrow_nonzero_count_matches_the_full_reader_across_stripes() {
15504 let path = path("quick-nonzero");
15505 let mut writer = Writer::create(
15506 &path,
15507 "items",
15508 vec![Field::new("label", LogicalType::Varchar), Field::new("id", LogicalType::Integer)],
15509 )
15510 .expect("create");
15511 for ids in [
15512 &[Value::Integer(0), Value::Null, Value::Integer(3)][..],
15513 &[Value::Integer(0), Value::Integer(7), Value::Null][..],
15514 ] {
15515 let labels = vec![Value::Varchar("same".into()); ids.len()];
15516 writer
15517 .append(
15518 &Chunk::new(vec![
15519 Vector::from_values(LogicalType::Varchar, &labels).expect("labels"),
15520 Vector::from_values(LogicalType::Integer, ids).expect("ids"),
15521 ])
15522 .expect("chunk"),
15523 )
15524 .expect("append");
15525 }
15526 writer.finish().expect("finish");
15527 let catalog = Catalog::open(&path).expect("catalog");
15528 assert_eq!(catalog.entries[0].nonzero, vec![None, None]);
15529 assert_eq!(catalog.entries[0].aggregates, vec![None, Some((10, 4))]);
15530 assert_eq!(catalog.entries[0].distincts, vec![Some(1), Some(3)]);
15531 assert_eq!(catalog.exact_numeric_frequencies("items", 1).expect("frequencies"), None);
15532 let prefix = catalog
15533 .table("items")
15534 .expect("reader")
15535 .frequency_prefix(1)
15536 .expect("valid metadata")
15537 .expect("partial frequencies");
15538 assert_eq!(prefix.entries, vec![(Value::Null, 2), (Value::Integer(0), 2)]);
15539 assert_eq!(prefix.omitted_max, 1);
15540 assert_eq!(catalog.distinct_count("items", 1).expect("distinct count"), Some(3));
15541 assert_eq!(
15542 catalog.integer_extremes("items", 1).expect("extremes"),
15543 Some(IntegerExtremes::Values { low: 0, high: 7 })
15544 );
15545 assert_eq!(
15546 catalog.aggregate_sums("items", &[1]).expect("catalog sums"),
15547 Some(CertifiedSums { columns: vec![(10, 4)], rows: 6 })
15548 );
15549 assert_eq!(catalog.nonzero_count("items", 1).expect("quick count"), Some(2));
15550 let mut legacy = catalog.clone();
15551 Arc::make_mut(&mut legacy.entries)[0].nonzero[1] = Some(999);
15552 assert_eq!(legacy.nonzero_count("items", 1).expect("ignore legacy count"), Some(2));
15553 Arc::make_mut(&mut legacy.entries)[0].frequencies[1] = None;
15554 assert_eq!(legacy.nonzero_count("items", 1).expect("directory fallback"), Some(2));
15555 Writer::certify_counts(&path).expect("recertify");
15556 assert_eq!(
15557 Catalog::open(&path).expect("reopen").nonzero_count("items", 1).expect("count"),
15558 Some(2)
15559 );
15560 assert_eq!(
15561 Catalog::open(&path).expect("reopen").aggregate_sums("items", &[1]).expect("sums"),
15562 Some(CertifiedSums { columns: vec![(10, 4)], rows: 6 })
15563 );
15564 assert_eq!(
15565 Catalog::open(&path).expect("reopen").distinct_count("items", 1).expect("distinct"),
15566 Some(3)
15567 );
15568 assert_eq!(
15569 Catalog::open(&path).expect("reopen").integer_extremes("items", 1).expect("ends"),
15570 Some(IntegerExtremes::Values { low: 0, high: 7 })
15571 );
15572 assert_eq!(
15573 Catalog::open(&path)
15574 .expect("reopen")
15575 .exact_numeric_frequencies("items", 1)
15576 .expect("frequencies"),
15577 None
15578 );
15579 assert_eq!(catalog.table("items").expect("reader").null_count(1).expect("nulls"), 2);
15580 fs::remove_file(path).expect("remove scratch file");
15581 }
15582
15583 #[test]
15584 fn numeric_string_pair_leaders_are_certified_in_the_directory() {
15585 let path = path("pair-frequencies");
15586 let mut pairs = Vec::new();
15587 pairs.extend(std::iter::repeat_n((1_i64, "alpha".to_string()), 100));
15588 pairs.extend(std::iter::repeat_n((1_i64, "beta".to_string()), 50));
15589 pairs.extend(std::iter::repeat_n((2_i64, "gamma".to_string()), 40));
15590 pairs.extend((1_000_i64..1_600).map(|id| (id, format!("tail {id}"))));
15591 let mut writer = Writer::create(
15592 &path,
15593 "items",
15594 vec![
15595 Field::required("id", LogicalType::BigInt),
15596 Field::required("phrase", LogicalType::Varchar),
15597 ],
15598 )
15599 .expect("new file");
15600 for part in pairs.chunks(1_024) {
15601 let ids = part.iter().map(|(id, _)| Value::BigInt(*id)).collect::<Vec<_>>();
15602 let phrases =
15603 part.iter().map(|(_, phrase)| Value::Varchar(phrase.clone())).collect::<Vec<_>>();
15604 writer
15605 .append(
15606 &Chunk::new(vec![
15607 Vector::from_values(LogicalType::BigInt, &ids).expect("ids"),
15608 Vector::from_values(LogicalType::Varchar, &phrases).expect("phrases"),
15609 ])
15610 .expect("matching columns"),
15611 )
15612 .expect("rows");
15613 }
15614 writer.finish().expect("commit");
15615
15616 let reader = Reader::open(&path).expect("reopen from disk");
15617 assert!(
15618 reader.table.pair_frequencies.is_empty(),
15619 "no query-specific pair result is stored"
15620 );
15621 fs::remove_file(path).expect("remove scratch file");
15622 }
15623
15624 #[test]
15625 fn legacy_group_answers_are_ignored() {
15626 let path = path("legacy-group-answers");
15627 let mut writer = Writer::create(
15628 &path,
15629 "items",
15630 vec![
15631 Field::required("id", LogicalType::BigInt),
15632 Field::required("text", LogicalType::Varchar),
15633 ],
15634 )
15635 .expect("new file");
15636 writer
15637 .append(
15638 &Chunk::new(vec![
15639 Vector::from_values(LogicalType::BigInt, &[Value::BigInt(1)]).expect("id"),
15640 Vector::from_values(LogicalType::Varchar, &[Value::Varchar("x".into())])
15641 .expect("text"),
15642 ])
15643 .expect("row"),
15644 )
15645 .expect("append");
15646 writer.finish().expect("commit");
15647 let mut reader = Reader::open(&path).expect("reopen");
15648 let table = Arc::make_mut(&mut reader.table);
15649 table.pair_frequencies.push(PairFrequencySummary {
15650 first: 0,
15651 second: 1,
15652 entries: vec![PairFrequencyEntry { first_entry: 0, second: Some(0), count: 999 }],
15653 omitted_max: 0,
15654 });
15655 table.host_groups = Some(host::HostSummary {
15656 column: 1,
15657 omitted_max: 0,
15658 entries: vec![host::HostEntry {
15659 host: "fake.test".into(),
15660 count: 999,
15661 bytes_sum: 999,
15662 minimum: "x".into(),
15663 }],
15664 });
15665 assert_eq!(reader.top_pair_frequencies(0, 1, 1).expect("legacy pair"), None);
15666 assert_eq!(reader.host_groups(1, 1).expect("legacy host"), None);
15667 fs::remove_file(path).expect("remove scratch file");
15668 }
15669
15670 #[test]
15676 fn a_file_from_another_format_says_which_format_it_is() {
15677 let older = path("older-format");
15678 let mut writer =
15679 Writer::create(&older, "items", vec![Field::new("id", LogicalType::Integer)])
15680 .expect("new file");
15681 let chunk = Chunk::new(vec![
15682 Vector::flat(LogicalType::Integer, Data::Int32(vec![1, 2, 3].into()))
15683 .expect("integers"),
15684 ])
15685 .expect("chunk");
15686 writer.append(&chunk).expect("page written");
15687 writer.finish().expect("commit");
15688
15689 let unreadable =
15693 READABLE.iter().copied().min().expect("at least one format is readable") - 1;
15694 let mut file = OpenOptions::new().write(true).open(&older).expect("open for the header");
15695 file.seek(SeekFrom::Start(8)).expect("the version follows the magic");
15696 file.write_all(&unreadable.to_le_bytes()).expect("write an older version");
15697 drop(file);
15698 let complaint = Reader::open(&older).expect_err("an older format is refused").to_string();
15699 assert!(complaint.contains(&format!("format {unreadable}")), "{complaint}");
15700 assert!(complaint.contains(&format!("format {FORMAT}")), "{complaint}");
15701
15702 let mut file = OpenOptions::new().write(true).open(&older).expect("open for the header");
15703 file.seek(SeekFrom::Start(0)).expect("the magic is first");
15704 file.write_all(b"NOTRUDB!").expect("write another engine's magic");
15705 drop(file);
15706 let complaint = Reader::open(&older).expect_err("a foreign file is refused").to_string();
15707 assert!(complaint.contains("magic"), "{complaint}");
15708 assert!(!complaint.contains("format"), "a version has nothing to do with it: {complaint}");
15709 fs::remove_file(older).expect("remove scratch file");
15710 }
15711
15712 #[test]
15713 fn an_unfinished_or_damaged_file_does_not_answer_with_partial_rows() {
15714 let unfinished = path("unfinished");
15715 let mut writer =
15716 Writer::create(&unfinished, "items", vec![Field::new("id", LogicalType::Integer)])
15717 .expect("new file");
15718 let chunk = Chunk::new(vec![
15719 Vector::flat(LogicalType::Integer, Data::Int32(vec![1, 2, 3].into()))
15720 .expect("integers"),
15721 ])
15722 .expect("chunk");
15723 writer.append(&chunk).expect("page written");
15724 drop(writer);
15725 assert!(Reader::open(&unfinished).is_err(), "no directory was committed");
15726 fs::remove_file(unfinished).expect("remove scratch file");
15727
15728 let damaged = path("damaged");
15729 let mut writer =
15730 Writer::create(&damaged, "items", vec![Field::new("id", LogicalType::Integer)])
15731 .expect("new file");
15732 writer.append(&chunk).expect("page written");
15733 writer.finish().expect("commit");
15734 let reader = Reader::open(&damaged).expect("valid directory");
15735 let mut file =
15736 OpenOptions::new().write(true).open(&damaged).expect("open for a damaged page");
15737 file.seek(SeekFrom::Start(HEADER + 1)).expect("inside first page");
15738 file.write_all(&[255]).expect("damage one byte");
15739 assert!(reader.read(0, &[0]).is_err(), "page checksum rejects corruption");
15740 fs::remove_file(damaged).expect("remove scratch file");
15741 }
15742
15743 #[test]
15744 fn damaged_lazy_dictionary_payload_is_an_error() {
15745 let path = path("damaged-dictionary");
15746 let mut writer = Writer::create(
15747 &path,
15748 "items",
15749 vec![
15750 Field::required("id", LogicalType::Integer),
15751 Field::new("text", LogicalType::Varchar),
15752 ],
15753 )
15754 .expect("new file");
15755 writer.append(&sample()).expect("stripe written");
15756 writer.finish().expect("commit");
15757
15758 let reader = Reader::open(&path).expect("valid directory");
15759 let dictionary = reader.table.dictionaries[1].expect("string dictionary page");
15760 let mut header = [0; DICTIONARY_HEADER];
15763 read_at(&reader.file, dictionary.offset, &mut header).expect("dictionary header");
15764 let count = u32::from_le_bytes(header[0..4].try_into().expect("four bytes")) as usize;
15767 let width = u32::from_le_bytes(header[12..16].try_into().expect("four bytes"));
15768 assert_ne!(width & DICTIONARY_SCATTERED, 0, "the blocks say where they are");
15769 let bits = (width & !DICTIONARY_FLAGS) as usize;
15770 let mut start = [0; 8];
15771 let at = dictionary.offset + (DICTIONARY_HEADER + offset_bytes(count, bits)) as u64;
15772 read_at(&reader.file, at, &mut start).expect("the first block's start");
15773 let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
15774 file.seek(SeekFrom::Start(u64::from_le_bytes(start))).expect("inside dictionary payload");
15775 file.write_all(&[255]).expect("damage dictionary payload");
15776
15777 let chunk = reader.read(0, &[1]).expect("code page and dictionary index remain valid");
15778 let error =
15779 chunk.validate_external().expect_err("payload corruption must reach the caller");
15780 assert!(error.message().contains("payload checksum differs"), "{error}");
15781 fs::remove_file(path).expect("remove scratch file");
15782 }
15783
15784 #[test]
15794 fn a_column_of_all_different_values_is_written_without_a_dictionary() {
15795 let path = path("dictionary-decide");
15796 let rows = 20_000;
15797 let unique =
15799 |row: usize| format!("{row:09} a value that appears exactly once in the table");
15800 let repeated = |row: usize| unique(row / 40);
15802 let mut writer = Writer::create(
15803 &path,
15804 "items",
15805 vec![
15806 Field::required("unique", LogicalType::Varchar),
15807 Field::required("repeated", LogicalType::Varchar),
15808 ],
15809 )
15810 .expect("new file");
15811 for part in (0..rows).step_by(1_000) {
15812 let span = part..(part + 1_000).min(rows);
15813 let left = span.clone().map(|row| Value::Varchar(unique(row))).collect::<Vec<_>>();
15814 let right = span.map(|row| Value::Varchar(repeated(row))).collect::<Vec<_>>();
15815 writer
15816 .append(
15817 &Chunk::new(vec![
15818 Vector::from_values(LogicalType::Varchar, &left).expect("strings"),
15819 Vector::from_values(LogicalType::Varchar, &right).expect("strings"),
15820 ])
15821 .expect("two columns"),
15822 )
15823 .expect("a part");
15824 }
15825 writer.finish().expect("commit");
15826
15827 let reader = Reader::open(&path).expect("reopen from disk");
15828 assert!(
15829 reader.table.dictionaries[0].is_none(),
15830 "a column with no repeats has nothing to say twice"
15831 );
15832 assert!(
15833 reader.table.dictionaries[1].is_some(),
15834 "a column whose values come round again keeps its dictionary"
15835 );
15836 let mut first = 0;
15837 for part in 0..reader.parts() {
15838 let chunk = reader.read(part, &[0, 1]).expect("a part");
15839 for row in 0..chunk.len() {
15840 assert_eq!(chunk.value_at(row, 0), Value::Varchar(unique(first + row)));
15841 assert_eq!(chunk.value_at(row, 1), Value::Varchar(repeated(first + row)));
15842 }
15843 first += chunk.len();
15844 }
15845 assert_eq!(first, rows, "every row was read back");
15846 let raw = (0..rows).map(|row| unique(row).len()).sum::<usize>();
15847 let size = fs::metadata(&path).expect("the file is there").len() as usize;
15848 assert!(size < raw, "a column without a dictionary is still encoded: {size} against {raw}");
15849 fs::remove_file(path).expect("remove scratch file");
15850 }
15851
15852 #[test]
15865 fn a_dictionary_over_many_blocks_checks_every_block_of_it() {
15866 let path = path("dictionary-blocks");
15867 let value = |row: usize| {
15868 let row = row.saturating_sub(8_000);
15869 format!("{row:07} a value long enough to be worth a payload block")
15870 };
15871 let parts = 40;
15872 let per_part = 1000;
15873 let mut writer =
15874 Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
15875 .expect("new file");
15876 for part in 0..parts {
15877 let values = (0..per_part)
15878 .map(|row| Value::Varchar(value(part * per_part + row)))
15879 .collect::<Vec<_>>();
15880 let chunk = Chunk::new(vec![
15881 Vector::from_values(LogicalType::Varchar, &values).expect("strings"),
15882 ])
15883 .expect("matching rows");
15884 writer.append(&chunk).expect("a part");
15885 }
15886 writer.finish().expect("commit");
15887
15888 let reader = Reader::open(&path).expect("reopen from disk");
15889 let dictionary = reader.table.dictionaries[0].expect("string dictionary page");
15890 assert!(
15891 parts * per_part > TEXT_PAYLOAD_VALUES * 4,
15892 "the dictionary has to be several blocks for this to be testing anything"
15893 );
15894 for part in [0, parts - 1] {
15895 let chunk = reader.read(part, &[0]).expect("a part");
15896 chunk.validate_external().expect("every payload block checks out");
15897 assert_eq!(chunk.value_at(0, 0), Value::Varchar(value(part * per_part)));
15898 }
15899
15900 let mut header = [0; DICTIONARY_HEADER];
15902 read_at(&reader.file, dictionary.offset, &mut header).expect("dictionary header");
15903 let count = u32::from_le_bytes(header[0..4].try_into().expect("four bytes")) as usize;
15904 let blocks = u32::from_le_bytes(header[8..12].try_into().expect("four bytes")) as usize;
15905 let width = u32::from_le_bytes(header[12..16].try_into().expect("four bytes"));
15906 let bits = (width & !DICTIONARY_FLAGS) as usize;
15907 let mut place = [0; 16];
15908 let at = DICTIONARY_HEADER + offset_bytes(count, bits) + (blocks - 1) * 16;
15909 read_at(&reader.file, dictionary.offset + at as u64, &mut place).expect("its place");
15910 let start = u64::from_le_bytes(place[..8].try_into().expect("eight bytes"));
15911 let length = u64::from_le_bytes(place[8..].try_into().expect("eight bytes"));
15912 let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
15913 file.seek(SeekFrom::Start(start + length - 4)).expect("the last bytes of the last block");
15914 file.write_all(&[255]).expect("damage the last payload block");
15915 let reader = Reader::open(&path).expect("the directory and the index are untouched");
15916 let chunk = reader.read(parts - 1, &[0]).expect("the code page remains valid");
15917 let error = chunk.validate_external().expect_err("the damage must reach the caller");
15918 assert!(error.message().contains("payload checksum differs"), "{error}");
15919 fs::remove_file(path).expect("remove scratch file");
15920 }
15921
15922 #[test]
15936 fn values_of_different_lengths_read_back_out_of_packed_offsets() {
15937 let path = path("dictionary-offsets");
15938 let value = |row: usize| {
15939 let row = row % 5_000;
15940 if row % 511 == 3 { String::new() } else { "x".repeat(row % 97) + &format!("{row:05}") }
15941 };
15942 let rows = 6_000;
15943 let mut writer =
15944 Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
15945 .expect("new file");
15946 let values = (0..rows).map(|row| Value::Varchar(value(row))).collect::<Vec<_>>();
15947 for part in values.chunks(1_000) {
15948 let chunk =
15949 Chunk::new(vec![Vector::from_values(LogicalType::Varchar, part).expect("strings")])
15950 .expect("matching rows");
15951 writer.append(&chunk).expect("a part");
15952 }
15953 writer.finish().expect("commit");
15954
15955 let reader = Reader::open(&path).expect("reopen from disk");
15956 assert!(
15957 rows > TEXT_PAYLOAD_VALUES * 4,
15958 "the dictionary has to be several blocks for this to be testing anything"
15959 );
15960 for part in 0..rows / 1_000 {
15961 let chunk = reader.read(part, &[0]).expect("a part");
15962 for row in 0..1_000 {
15963 let row = part * 1_000 + row;
15964 assert_eq!(
15965 chunk.value_at(row % 1_000, 0),
15966 Value::Varchar(value(row)),
15967 "value {row}"
15968 );
15969 }
15970 }
15971 for _ in 0..2 {
15974 for part in 0..rows / 1_000 {
15975 let chunk = reader.read(part, &[0]).expect("a part");
15976 let mut lens = vec![0_i64; 1_000];
15977 let column = chunk.column(0).expect("one column");
15978 assert!(column.try_bytes_lens(&mut lens).expect("lengths"), "a stored column");
15979 for (row, &len) in lens.iter().enumerate() {
15980 let row = part * 1_000 + row;
15981 assert_eq!(len as usize, value(row).len(), "the length of value {row}");
15982 }
15983 }
15984 }
15985 fs::remove_file(path).expect("remove scratch file");
15986 }
15987
15988 #[test]
15990 fn lengths_restart_at_each_block_and_refuse_ends_that_go_backwards() {
15991 let mut ends: Vec<u32> = (1..=TEXT_PAYLOAD_VALUES as u32).map(|at| at * 2).collect();
15992 ends.extend([3, 3, 10]);
15993 let Some(Lengths::Narrow(lens)) = lengths_of(&ends) else { panic!("short ordered ends") };
15994 assert!(lens[..TEXT_PAYLOAD_VALUES].iter().all(|&len| len == 2));
15995 assert_eq!(&lens[TEXT_PAYLOAD_VALUES..], &[3, 0, 7]);
15996 let long = [5, 70_005, 70_006];
15998 let Some(Lengths::Wide(lens)) = lengths_of(&long) else { panic!("long ordered ends") };
15999 assert_eq!(lens, [5, 70_000, 1]);
16000 let mut read = Vec::new();
16001 Lengths::Wide(lens).extend_at(&[1, 9, 0], &mut read);
16002 assert_eq!(read, [70_000, 0, 5], "a position past the end is no length");
16003 ends.push(9);
16004 assert!(lengths_of(&ends).is_none());
16005 }
16006
16007 #[test]
16019 fn a_global_dictionary_is_opened_once_however_many_workers_ask_at_once() {
16020 let path = path("dictionary-once");
16021 let parts = 8;
16022 let per_part = 500;
16023 let value =
16024 |row: usize| format!("{row:07} a value long enough to be worth a payload block");
16025 let mut writer =
16026 Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
16027 .expect("new file");
16028 for part in 0..parts {
16029 let values = (0..per_part)
16030 .map(|row| Value::Varchar(value(part * per_part + row)))
16031 .collect::<Vec<_>>();
16032 let chunk = Chunk::new(vec![
16033 Vector::from_values(LogicalType::Varchar, &values).expect("strings"),
16034 ])
16035 .expect("matching rows");
16036 writer.append(&chunk).expect("a part");
16037 }
16038 writer.finish().expect("commit");
16039
16040 let reader = Reader::open(&path).expect("reopen from disk");
16041 assert!(reader.table.dictionaries[0].is_some(), "the column has to have one to share");
16042 assert_eq!(reader.reads().dictionaries, 0, "opening the file does not open a dictionary");
16043
16044 let workers = 16;
16045 let gate = std::sync::Barrier::new(workers);
16046 std::thread::scope(|scope| {
16047 for worker in 0..workers {
16048 let reader = reader.clone();
16049 let gate = &gate;
16050 scope.spawn(move || {
16051 gate.wait();
16052 let chunk = reader.read(worker % parts, &[0]).expect("a part");
16053 assert_eq!(
16054 chunk.value_at(0, 0),
16055 Value::Varchar(value((worker % parts) * per_part))
16056 );
16057 });
16058 }
16059 });
16060
16061 assert_eq!(reader.reads().dictionaries, 1, "sixteen workers, one dictionary, one open");
16062 fs::remove_file(path).expect("remove scratch file");
16063 }
16064
16065 #[test]
16070 fn a_damaged_sorted_order_is_an_error() {
16071 let path = path("damaged-order");
16072 let mut writer = Writer::create(
16073 &path,
16074 "items",
16075 vec![
16076 Field::required("id", LogicalType::Integer),
16077 Field::new("text", LogicalType::Varchar),
16078 ],
16079 )
16080 .expect("new file");
16081 writer.append(&sample()).expect("stripe written");
16082 writer.finish().expect("commit");
16083
16084 let reader = Reader::open(&path).expect("valid directory");
16085 let page = reader.table.dictionaries[1].expect("string dictionary page");
16086 let mut header = [0; DICTIONARY_HEADER];
16087 read_at(&reader.file, page.offset, &mut header).expect("dictionary header");
16088 let index_len = dictionary_index_len(&header);
16089 let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
16090 file.seek(SeekFrom::Start(page.offset + index_len)).expect("the first head");
16091 file.write_all(&[255]).expect("damage the order");
16092
16093 let dictionary = reader.dictionary(1).expect("read").expect("a string column has one");
16094 let error = dictionary.compare_rank(0, b"anything").expect_err("a damaged order is caught");
16095 assert!(error.message().contains("rank checksum differs"), "{error}");
16096 fs::remove_file(path).expect("remove scratch file");
16097 }
16098
16099 #[test]
16103 fn a_global_dictionary_carries_the_sorted_order_of_its_values() {
16104 let spellings = ["overlong1z", "b", "", "overlong1a", "overlong", "ab", "a", "overlong1"];
16107 let path = path("dictionary-order");
16108 let mut writer =
16109 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
16110 .expect("new file");
16111 writer
16112 .append(
16113 &Chunk::new(vec![
16114 Vector::from_values(
16115 LogicalType::Varchar,
16116 &spellings.map(|text| Value::Varchar(text.into())),
16117 )
16118 .expect("strings"),
16119 ])
16120 .expect("one column"),
16121 )
16122 .expect("stripe written");
16123 writer.finish().expect("commit");
16124
16125 let reader = Reader::open(&path).expect("valid directory");
16126 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
16127 let count = dictionary.ranks().expect("a v10 file stores one");
16128 assert_eq!(count, spellings.len(), "every distinct value has a rank");
16129 let order = (0..count)
16130 .map(|rank| dictionary.code_at_rank(rank).expect("a code"))
16131 .collect::<Vec<_>>();
16132 let mut seen = order.clone();
16133 seen.sort_unstable();
16134 assert_eq!(seen, (0..spellings.len() as u32).collect::<Vec<_>>(), "a permutation of codes");
16135
16136 let ranked = order
16137 .iter()
16138 .map(|&code| {
16139 dictionary.try_bytes_at(code as usize).expect("read").expect("a value").to_vec()
16140 })
16141 .collect::<Vec<_>>();
16142 let mut expected = spellings.map(|text| text.as_bytes().to_vec()).to_vec();
16143 expected.sort();
16144 assert_eq!(ranked, expected, "rank order is value order");
16145
16146 for (rank, value) in expected.iter().enumerate() {
16149 assert_eq!(
16150 dictionary.compare_rank(rank, value).expect("compare"),
16151 Ordering::Equal,
16152 "rank {rank} is its own value"
16153 );
16154 if rank > 0 {
16155 assert_eq!(
16156 dictionary.compare_rank(rank - 1, value).expect("compare"),
16157 Ordering::Less,
16158 "rank {rank} follows the one before it"
16159 );
16160 }
16161 }
16162 fs::remove_file(path).expect("remove scratch file");
16163 }
16164
16165 #[test]
16172 fn text_columns_closed_at_once_each_keep_their_own_dictionary() {
16173 let sizes = [300_usize, 5_000, 40, 2_000, 1_200];
16174 let path = path("dictionaries-at-once");
16175 let fields = (0..sizes.len())
16176 .map(|column| Field::new(format!("text{column}"), LogicalType::Varchar))
16177 .collect::<Vec<_>>();
16178 let mut writer = Writer::create(&path, "items", fields).expect("new file");
16179 let rows = 10_000_usize;
16180 for start in (0..rows).step_by(1_024) {
16181 let columns = sizes
16182 .iter()
16183 .enumerate()
16184 .map(|(column, &size)| {
16185 let values = (start..(start + 1_024).min(rows))
16186 .map(|row| Value::Varchar(format!("c{column}-{:05}", (row * 7919) % size)))
16187 .collect::<Vec<_>>();
16188 Vector::from_values(LogicalType::Varchar, &values).expect("strings")
16189 })
16190 .collect::<Vec<_>>();
16191 writer.append(&Chunk::new(columns).expect("five columns")).expect("stripe written");
16192 }
16193 writer.finish().expect("commit");
16194
16195 let reader = Reader::open(&path).expect("valid directory");
16196 for (column, &size) in sizes.iter().enumerate() {
16197 let dictionary =
16198 reader.dictionary(column).expect("read").expect("a string column has one");
16199 let count = dictionary.ranks().expect("a v10 file stores one");
16200 assert_eq!(count, size, "column {column} has its own distinct count");
16201 let ranked = (0..count)
16202 .map(|rank| {
16203 let code = dictionary.code_at_rank(rank).expect("a code");
16204 dictionary.try_bytes_at(code as usize).expect("read").expect("a value").to_vec()
16205 })
16206 .collect::<Vec<_>>();
16207 let expected = (0..size)
16208 .map(|value| format!("c{column}-{value:05}").into_bytes())
16209 .collect::<Vec<_>>();
16210 assert_eq!(ranked, expected, "column {column} ranks its own values in order");
16211 }
16212 fs::remove_file(path).expect("remove scratch file");
16213 }
16214
16215 #[test]
16223 fn a_large_dictionary_ranks_in_value_order() {
16224 let path = path("dictionary-large-rank");
16225 let value = |row: u64| {
16226 let mixed = row.wrapping_mul(0x9e37_79b9_7f4a_7c15) >> 40;
16227 match row % 3 {
16228 0 => format!("https://example.com/a/long/shared/path/{mixed:08}"),
16229 1 => format!("{mixed}"),
16230 _ => format!("x{}", row % 1000).repeat(1 + (row % 4) as usize) + &row.to_string(),
16231 }
16232 };
16233 let distinct = 70_000;
16234 let parts = 4 * distinct / 1000;
16235 let per_part = 1000;
16236 let mut writer =
16237 Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
16238 .expect("new file");
16239 for part in 0..parts {
16240 let values = (0..per_part)
16241 .map(|row| Value::Varchar(value((part * per_part + row) / 4)))
16242 .collect::<Vec<_>>();
16243 let chunk = Chunk::new(vec![
16244 Vector::from_values(LogicalType::Varchar, &values).expect("strings"),
16245 ])
16246 .expect("matching rows");
16247 writer.append(&chunk).expect("a part");
16248 }
16249 writer.finish().expect("commit");
16250
16251 let reader = Reader::open(&path).expect("reopen from disk");
16252 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
16253 let count = dictionary.ranks().expect("a ranked dictionary");
16254 assert_eq!(count, distinct as usize, "every distinct value has a rank");
16255 assert!(count >= PARALLEL_SORT_MIN, "too few values to be sorted on more than one thread");
16256 let ranked = (0..count)
16257 .map(|rank| {
16258 let code = dictionary.code_at_rank(rank).expect("a code");
16259 dictionary.try_bytes_at(code as usize).expect("read").expect("a value").to_vec()
16260 })
16261 .collect::<Vec<_>>();
16262 let mut expected = (0..distinct).map(|row| value(row).into_bytes()).collect::<Vec<_>>();
16263 expected.sort();
16264 assert_eq!(ranked, expected, "rank order is value order");
16265 fs::remove_file(path).expect("remove scratch file");
16266 }
16267
16268 #[test]
16281 fn a_directory_read_a_window_at_a_time_is_the_directory_read_whole() {
16282 let path = path("windowed-directory");
16283 let fields = vec![
16284 Field::required("id", LogicalType::BigInt),
16285 Field::required("word", LogicalType::Varchar),
16286 Field::new("score", LogicalType::Double),
16287 ];
16288 let mut writer = Writer::create(&path, "items", fields).expect("new file");
16289 for part in 0..70_i64 {
16290 let ids = (0..100).map(|row| Value::BigInt(part * 100 + row % 7)).collect::<Vec<_>>();
16291 let words = (0..100)
16292 .map(|row| Value::Varchar(format!("word {}", row % 13)))
16293 .collect::<Vec<_>>();
16294 let scores = (0..100)
16295 .map(|row| if row % 4 == 0 { Value::Null } else { Value::Double(row as f64) })
16296 .collect::<Vec<_>>();
16297 let chunk = Chunk::new(vec![
16298 Vector::from_values(LogicalType::BigInt, &ids).expect("integers"),
16299 Vector::from_values(LogicalType::Varchar, &words).expect("strings"),
16300 Vector::from_values(LogicalType::Double, &scores).expect("doubles"),
16301 ])
16302 .expect("three columns");
16303 writer.append(&chunk).expect("a part");
16304 }
16305 writer.finish().expect("commit");
16306
16307 let catalog = Catalog::open(&path).expect("reopen");
16308 let entry = catalog.entries.first().expect("one table").directory;
16309 let (offset, length) = (entry.offset, entry.length as usize);
16310 let mut bytes = vec![0; length];
16311 read_at(&catalog.file, offset, &mut bytes).expect("the directory");
16312 assert_eq!(file_checksum(&catalog.file, offset, length).expect("checksum"), entry.hash);
16313 let whole = decode_directory(&bytes, catalog.size).expect("whole");
16314 assert!(whole.stripes.len() > 1, "the table should span stripes");
16315 for size in [1, 7, 33, 4_096] {
16316 let mut cursor = Cursor::over(&catalog.file, offset, length);
16317 cursor.window.as_mut().expect("a window").size = size;
16318 let windowed = read_directory(cursor, catalog.size, Some(offset)).expect("windowed");
16319 assert_eq!(format!("{:?}", windowed.stripes), format!("{:?}", whole.stripes));
16320 assert_eq!(format!("{:?}", windowed.fields), format!("{:?}", whole.fields));
16321 let mut stored = 0;
16322 for (column, (left, held)) in
16323 windowed.frequencies.iter().zip(&whole.frequencies).enumerate()
16324 {
16325 match (left, held) {
16326 (None, None) => {}
16327 (
16328 Some(super::Frequencies::Stored { span, values, entries }),
16329 Some(super::Frequencies::Held(summary)),
16330 ) => {
16331 let mut one = vec![0; span.length as usize];
16332 read_at(&catalog.file, span.offset, &mut one).expect("a synopsis");
16333 let read = decode_summary(
16334 &mut Cursor::new(&one),
16335 &whole.fields[column],
16336 whole.rows,
16337 *values,
16338 )
16339 .expect("a valid synopsis")
16340 .expect("one is there");
16341 assert_eq!(*entries, read.entries.len());
16342 assert_eq!(format!("{read:?}"), format!("{summary:?}"));
16343 stored += 1;
16344 }
16345 other => panic!("column {column} came back as {other:?}"),
16346 }
16347 }
16348 assert!(stored >= 2, "only {stored} synopses were left in the file");
16349 }
16350 let reader = catalog.table("items").expect("the table");
16351 assert!(reader.frequency_summaries[1].get().is_none());
16352 assert!(reader.top_frequencies(1, 1).expect("a readable synopsis").is_some());
16353 let first = reader.frequency_summaries[1].get().expect("decoded synopsis");
16354 let clone = reader.clone();
16355 assert!(clone.top_frequencies(1, 1).expect("cached synopsis").is_some());
16356 assert!(Arc::ptr_eq(first, clone.frequency_summaries[1].get().expect("same synopsis")));
16357 fs::remove_file(path).expect("remove scratch file");
16358 }
16359
16360 #[test]
16361 fn a_checksum_carried_across_reads_is_the_checksum_of_the_whole() {
16362 let path = path("file-checksum");
16363 let bytes = (0..200_000_u32)
16364 .map(|at| (at.wrapping_mul(2_654_435_761) >> 13) as u8)
16365 .collect::<Vec<_>>();
16366 fs::write(&path, &bytes).expect("scratch file");
16367 let file = File::open(&path).expect("open");
16368 for (offset, length) in [
16369 (0, 0),
16370 (3, 1),
16371 (5, 31),
16372 (0, 32),
16373 (9, 33),
16374 (1, 65_536),
16375 (7, 65_567),
16376 (0, 200_000),
16377 (11, 131_101),
16378 ] {
16379 let whole = checksum(&bytes[offset..offset + length]);
16380 assert_eq!(
16381 file_checksum(&file, offset as u64, length).expect("read"),
16382 whole,
16383 "{offset} {length}"
16384 );
16385 }
16386 fs::remove_file(path).expect("remove scratch file");
16387 }
16388
16389 #[test]
16390 fn a_string_synopsis_is_read_without_keeping_the_dictionary_blocks() {
16391 let path = path("synopsis-keeps-no-block");
16392 let spelled = |index: usize| Value::Varchar(format!("phrase {index:05}"));
16393 let mut values = (0..3_000).map(spelled).collect::<Vec<_>>();
16394 for _ in 0..3 {
16395 values.extend((0..3_000).step_by(5).map(spelled));
16396 }
16397 let mut writer =
16398 Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
16399 .expect("new file");
16400 for part in values.chunks(1_024) {
16401 writer
16402 .append(
16403 &Chunk::new(vec![
16404 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
16405 ])
16406 .expect("one column"),
16407 )
16408 .expect("a part");
16409 }
16410 writer.finish().expect("commit");
16411
16412 let reader = Reader::open(&path).expect("reopen from disk");
16413 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
16414 let resting = dictionary.footprint();
16415 let prefix = reader.frequency_prefix(0).expect("a readable synopsis").expect("one");
16416 assert_eq!(prefix.entries.len(), 512);
16417 for (value, count) in &prefix.entries {
16418 let Value::Varchar(text) = value else { panic!("a string column gave {value:?}") };
16419 let index = text["phrase ".len()..].parse::<usize>().expect("a spelled number");
16420 assert_eq!((index % 5, *count), (0, 4), "{text} came back with {count}");
16421 }
16422 assert_eq!(dictionary.footprint(), resting, "reading the synopsis kept a decoded block");
16423 let again = reader.frequency_prefix(0).expect("a readable synopsis").expect("one");
16424 assert_eq!(again.entries, prefix.entries);
16425 fs::remove_file(path).expect("remove scratch file");
16426 }
16427
16428 #[test]
16435 fn character_lengths_are_counted_without_keeping_the_dictionary_blocks() {
16436 let path = path("character-lengths");
16437 let spellings = (0..2_500)
16438 .map(|index| Value::Varchar(format!("héllo {index:05} {}", "ü".repeat(index % 30))))
16439 .collect::<Vec<_>>();
16440 let mut writer =
16441 Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
16442 .expect("new file");
16443 for part in spellings.chunks(1_024) {
16444 writer
16445 .append(
16446 &Chunk::new(vec![
16447 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
16448 ])
16449 .expect("one column"),
16450 )
16451 .expect("a part");
16452 }
16453 writer.finish().expect("commit");
16454
16455 let reader = Reader::open(&path).expect("reopen from disk");
16456 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
16457 let resting = dictionary.footprint();
16458 let mut lens = Vec::new();
16459 assert!(dictionary.try_chars_lens(&mut lens).expect("counted"), "a stored source counts");
16460 let counted = dictionary.footprint() - resting;
16461 let blocks = dictionary.len().div_ceil(TEXT_PAYLOAD_VALUES);
16462 assert!(
16463 counted <= blocks * TEXT_PAYLOAD_VALUES * size_of::<u32>(),
16464 "counting kept {counted} bytes, more than a count a value"
16465 );
16466 let expected = (0..dictionary.len())
16467 .map(|code| {
16468 let bytes = dictionary.try_bytes_at(code).expect("read").expect("a value");
16469 i64::try_from(std::str::from_utf8(bytes).expect("utf-8").chars().count())
16470 .expect("small")
16471 })
16472 .collect::<Vec<_>>();
16473 assert_eq!(lens, expected, "a count is the number of characters, not of bytes");
16474 let mut again = Vec::new();
16475 assert!(dictionary.try_chars_lens(&mut again).expect("counted"));
16476 assert_eq!(again, lens, "the kept counts answer the second time");
16477 fs::remove_file(path).expect("remove scratch file");
16478 }
16479
16480 fn stored_spellings(label: &str, spellings: &[String]) -> (PathBuf, Reader) {
16482 let path = path(label);
16483 let values = spellings.iter().map(|text| Value::Varchar(text.clone())).collect::<Vec<_>>();
16484 let mut writer =
16485 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
16486 .expect("new file");
16487 for part in values.chunks(1_024) {
16488 writer
16489 .append(
16490 &Chunk::new(vec![
16491 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
16492 ])
16493 .expect("one column"),
16494 )
16495 .expect("a part");
16496 }
16497 writer.finish().expect("commit");
16498 let reader = Reader::open(&path).expect("reopen from disk");
16499 (path, reader)
16500 }
16501
16502 fn scattered_rows(len: usize) -> (Vec<u32>, Vec<bool>) {
16508 let codes = (0..len)
16509 .map(|row| u32::try_from(row * 7_919 % len).expect("a small dictionary"))
16510 .collect::<Vec<_>>();
16511 let valid = (0..len).map(|row| row % 7 != 3).collect::<Vec<_>>();
16512 (codes, valid)
16513 }
16514
16515 #[test]
16522 fn character_lengths_with_nulls_are_counted_without_keeping_the_dictionary_blocks() {
16523 let spellings = (0..2_500)
16524 .map(|index| format!("héllo {index:05} {}", "ü".repeat(index % 30)))
16525 .collect::<Vec<_>>();
16526 let (path, reader) = stored_spellings("character-lengths-nulls", &spellings);
16527 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
16528 let (codes, valid) = scattered_rows(spellings.len());
16529 let rows = Vector::dictionary_over(codes.clone(), Arc::clone(&dictionary))
16530 .expect("every code is inside")
16531 .with_validity(Validity::from_run(&valid));
16532
16533 let resting = dictionary.footprint();
16534 let lens = rudb_kernels::call("length", &[&rows], &LogicalType::BigInt, None)
16535 .expect("length reads");
16536 let counted = dictionary.footprint() - resting;
16537 let blocks = dictionary.len().div_ceil(TEXT_PAYLOAD_VALUES);
16538 assert!(
16539 counted <= blocks * TEXT_PAYLOAD_VALUES * size_of::<u32>(),
16540 "length over a vector with nulls kept {counted} bytes, more than a count a value"
16541 );
16542 let expected = (0..rows.len())
16543 .map(|row| match valid[row] {
16544 true => Value::BigInt(
16545 i64::try_from(spellings[codes[row] as usize].chars().count()).expect("small"),
16546 ),
16547 false => Value::Null,
16548 })
16549 .collect::<Vec<_>>();
16550 let answers = (0..lens.len()).map(|row| lens.value_at(row)).collect::<Vec<_>>();
16551 assert_eq!(answers, expected, "a count of characters where a row has one, null elsewhere");
16552 fs::remove_file(path).expect("remove scratch file");
16553 }
16554
16555 #[test]
16565 fn string_kernels_read_a_stored_dictionary_without_keeping_its_blocks() {
16566 let spellings = (0..2_500)
16567 .map(|index| format!("HéLLo {index:05} {}", "Üß".repeat(index % 30)))
16568 .collect::<Vec<_>>();
16569 let (path, reader) = stored_spellings("string-kernels", &spellings);
16570 let page = reader.table.dictionaries[0].expect("a string column has one");
16571 let starved =
16572 open_global_dictionary(Arc::clone(&reader.file), page, &LogicalType::Varchar, 0)
16573 .expect("a dictionary opens whatever it may keep");
16574 let starved = Arc::new(starved);
16575 let (codes, valid) = scattered_rows(spellings.len());
16576 let rows = Vector::dictionary_over(codes.clone(), Arc::clone(&starved))
16577 .expect("every code is inside")
16578 .with_validity(Validity::from_run(&valid));
16579 let expected = |each: &dyn Fn(&str) -> String| {
16580 (0..rows.len())
16581 .map(|row| match valid[row] {
16582 true => Value::Varchar(each(&spellings[codes[row] as usize])),
16583 false => Value::Null,
16584 })
16585 .collect::<Vec<_>>()
16586 };
16587 let answers =
16588 |vector: &Vector| (0..vector.len()).map(|row| vector.value_at(row)).collect::<Vec<_>>();
16589
16590 let resting = starved.footprint();
16593 let ends = spellings.len() * size_of::<u32>();
16594 let lowered = rudb_kernels::call("lower", &[&rows], &LogicalType::Varchar, None)
16595 .expect("lower reads");
16596 assert_eq!(answers(&lowered), expected(&|text| text.to_lowercase()), "lower");
16597 assert!(starved.footprint() <= resting + ends, "lower kept a block it read");
16598
16599 let start = Vector::constant(LogicalType::BigInt, Value::BigInt(3), rows.len());
16600 let length = Vector::constant(LogicalType::BigInt, Value::BigInt(9), rows.len());
16601 let cut =
16602 rudb_kernels::call("substring", &[&rows, &start, &length], &LogicalType::Varchar, None)
16603 .expect("substring reads");
16604 let cut_of = |text: &str| text.chars().skip(2).take(9).collect::<String>();
16605 assert_eq!(answers(&cut), expected(&cut_of), "substring");
16606 assert!(starved.footprint() <= resting + ends, "substring kept a block it read");
16607
16608 let raised = rudb_kernels::call("upper", &[&rows], &LogicalType::Varchar, None)
16611 .expect("upper reads");
16612 assert_eq!(answers(&raised), expected(&|text| text.to_uppercase()), "upper");
16613 let payload = spellings.iter().map(String::len).sum::<usize>();
16614 assert!(
16615 starved.footprint() >= resting + payload,
16616 "a visit that has dropped a column's worth of blocks keeps what it reads"
16617 );
16618 let again = rudb_kernels::call("upper", &[&rows], &LogicalType::Varchar, None)
16619 .expect("upper reads kept blocks");
16620 assert_eq!(answers(&again), answers(&raised), "the kept blocks answer the same");
16621 fs::remove_file(path).expect("remove scratch file");
16622 }
16623
16624 #[test]
16634 fn a_dictionary_sweep_reads_every_value_and_keeps_it_under_the_budget() {
16635 let path = path("dictionary-sweep");
16636 let spellings = (0..2_500)
16639 .map(|index| Value::Varchar(format!("value {index:08} {}", "x".repeat(index % 40))))
16640 .collect::<Vec<_>>();
16641 let mut writer =
16642 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
16643 .expect("new file");
16644 for part in spellings.chunks(1_024) {
16647 writer
16648 .append(
16649 &Chunk::new(vec![
16650 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
16651 ])
16652 .expect("one column"),
16653 )
16654 .expect("stripe written");
16655 }
16656 writer.finish().expect("commit");
16657
16658 let reader = Reader::open(&path).expect("valid directory");
16659 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
16660 assert_eq!(dictionary.len(), spellings.len(), "every value is distinct");
16661 for first in [0, TEXT_PAYLOAD_VALUES, TEXT_PAYLOAD_VALUES * 2] {
16662 assert!(dictionary.text_block_might_contain(first, b"value").expect("signature"));
16663 assert!(!dictionary.text_block_might_contain(first, b"google").expect("signature"));
16664 }
16665
16666 let resting = dictionary.footprint();
16667 let sweep = || {
16668 let mut swept: Vec<Vec<u8>> = Vec::new();
16669 let mut at = 0;
16670 let mut calls = 0;
16671 while at < dictionary.len() {
16672 let stopped = dictionary
16673 .sweep_text(at, dictionary.len(), &mut |index: usize, text: &[u8]| {
16674 assert_eq!(index, swept.len(), "a sweep hands its values over in order");
16675 swept.push(text.to_vec());
16676 Ok(())
16677 })
16678 .expect("a sweep reads");
16679 assert!(stopped > at, "a sweep moves");
16680 at = stopped;
16681 calls += 1;
16682 }
16683 assert_eq!(calls, 3, "a sweep hands over one block at a time");
16684 swept
16685 };
16686 let swept = sweep();
16687 assert_eq!(dictionary.footprint(), resting, "a first sweep keeps nothing it decoded");
16688 assert_eq!(sweep(), swept, "a second sweep reads what the first did");
16689 let after = dictionary.footprint();
16690 assert!(after > resting, "a second sweep under the budget keeps what it decoded");
16691
16692 let read = (0..dictionary.len())
16693 .map(|code| dictionary.try_bytes_at(code).expect("read").expect("a value").to_vec())
16694 .collect::<Vec<_>>();
16695 assert_eq!(swept, read, "a sweep answers what a point read answers");
16696 let grown = dictionary.footprint() - after;
16700 assert!(
16701 grown == 0 || grown == dictionary.len() * size_of::<u32>(),
16702 "a point read of a kept block decodes nothing, and {grown} bytes grew"
16703 );
16704 fs::remove_file(path).expect("remove scratch file");
16705 }
16706
16707 #[test]
16708 fn a_narrow_signature_of_an_older_file_answers_by_its_own_width() {
16709 let path = path("narrow-substring-signature");
16710 let blocks = [&b"https://google.com/"[..], b"https://example.org/", b"mail.google.com"];
16711 let mut grams = Vec::new();
16712 for text in blocks {
16713 let mut bits = vec![0_u8; NARROW_GRAM_BYTES];
16714 for gram in text.windows(4) {
16715 for bit in gram_bits(gram, NARROW_GRAM_BYTES) {
16716 bits[bit / 8] |= 1 << (bit % 8);
16717 }
16718 }
16719 grams.extend(bits);
16720 }
16721 fs::write(&path, &grams).expect("scratch file");
16722 let file = File::open(&path).expect("open scratch file");
16723 let signatures = NativeGrams {
16724 start: 0,
16725 length: grams.len(),
16726 width: NARROW_GRAM_BYTES,
16727 hash: checksum(&grams),
16728 verdicts: Mutex::new(Vec::new()),
16729 };
16730 let verdict = signatures.verdicts(&file, b"google").expect("signatures read");
16731 assert_eq!(&verdict[..], &[true, false, true], "one verdict a block, at the narrow width");
16732 assert!(signatures.footprint() > 0, "a verdict is remembered");
16733 let again = signatures.verdicts(&file, b"google").expect("remembered");
16734 assert!(Arc::ptr_eq(&verdict, &again), "a second question about a literal reads nothing");
16735
16736 let damaged = NativeGrams {
16737 hash: signatures.hash ^ 1,
16738 verdicts: Mutex::new(Vec::new()),
16739 ..signatures
16740 };
16741 let error = damaged.verdicts(&file, b"google").expect_err("a damaged region is refused");
16742 assert!(error.to_string().contains("substring signatures checksum differs"), "{error}");
16743 fs::remove_file(path).expect("remove scratch file");
16744 }
16745
16746 #[test]
16747 fn a_damaged_substring_signature_is_checked_only_when_used() {
16748 let path = path("damaged-substring-signature");
16749 let mut writer =
16750 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
16751 .expect("new file");
16752 let rows = [Value::Varchar("google".into()), Value::Varchar("example".into())];
16753 writer
16754 .append(
16755 &Chunk::new(vec![
16756 Vector::from_values(LogicalType::Varchar, &rows).expect("strings"),
16757 ])
16758 .expect("one column"),
16759 )
16760 .expect("stripe written");
16761 writer.finish().expect("commit");
16762
16763 let reader = Reader::open(&path).expect("valid directory");
16764 let page = reader.table.dictionaries[0].expect("string dictionary page");
16765 let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
16766 file.seek(SeekFrom::Start(page.offset + u64::from(page.length) - 1))
16767 .expect("last signature byte");
16768 file.write_all(&[255]).expect("damage signature");
16769 let reader = Reader::open(&path).expect("the directory is still valid");
16770 let dictionary = reader.dictionary(0).expect("index is still valid").expect("dictionary");
16771 let error = dictionary
16772 .text_block_might_contain(0, b"goog")
16773 .expect_err("a used signature checks its own checksum");
16774 assert!(error.message().contains("substring signatures checksum differs"), "{error}");
16775 fs::remove_file(path).expect("remove scratch file");
16776 }
16777
16778 #[test]
16789 fn a_sweep_over_a_block_with_a_short_second_run_reads_what_a_point_read_reads() {
16790 let path = path("dictionary-sweep-short-run");
16791 let spellings = (0..2_800)
16792 .map(|index| Value::Varchar(format!("value {index:08} {}", "x".repeat(index % 40))))
16793 .collect::<Vec<_>>();
16794 let mut writer =
16795 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
16796 .expect("new file");
16797 for part in spellings.chunks(1_024) {
16798 writer
16799 .append(
16800 &Chunk::new(vec![
16801 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
16802 ])
16803 .expect("one column"),
16804 )
16805 .expect("stripe written");
16806 }
16807 writer.finish().expect("commit");
16808
16809 let reader = Reader::open(&path).expect("valid directory");
16810 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
16811 assert_eq!(dictionary.len(), spellings.len(), "every value is distinct");
16812 let last = dictionary.len() % TEXT_PAYLOAD_VALUES;
16813 assert!(last > TEXT_OFFSET_RUN, "the last block has to reach into a second run of offsets");
16814 assert!(last < TEXT_PAYLOAD_VALUES, "and that second run has to be short of a whole one");
16815
16816 let mut swept: Vec<Vec<u8>> = Vec::new();
16817 let mut at = 0;
16818 while at < dictionary.len() {
16819 let stopped = dictionary
16820 .sweep_text(at, dictionary.len(), &mut |index: usize, text: &[u8]| {
16821 assert_eq!(index, swept.len(), "a sweep hands its values over in order");
16822 swept.push(text.to_vec());
16823 Ok(())
16824 })
16825 .expect("a sweep reads");
16826 assert!(stopped > at, "a sweep moves");
16827 at = stopped;
16828 }
16829 let read = (0..dictionary.len())
16830 .map(|code| dictionary.try_bytes_at(code).expect("read").expect("a value").to_vec())
16831 .collect::<Vec<_>>();
16832 assert_eq!(swept, read, "a sweep answers what a point read answers");
16833 fs::remove_file(path).expect("remove scratch file");
16834 }
16835
16836 #[test]
16845 fn the_unpacked_ends_answer_what_the_packed_ends_answer() {
16846 let path = path("dictionary-unpacked-ends");
16847 let spellings = (0..2_800)
16848 .map(|index| Value::Varchar(format!("value {index:08} {}", "x".repeat(index % 40))))
16849 .collect::<Vec<_>>();
16850 let mut writer =
16851 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
16852 .expect("new file");
16853 for part in spellings.chunks(1_024) {
16854 writer
16855 .append(
16856 &Chunk::new(vec![
16857 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
16858 ])
16859 .expect("one column"),
16860 )
16861 .expect("stripe written");
16862 }
16863 writer.finish().expect("commit");
16864
16865 let reader = Reader::open(&path).expect("valid directory");
16866 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
16867 assert_eq!(dictionary.len(), spellings.len(), "every value is distinct");
16868 let wanted = (0..spellings.len())
16869 .map(|index| format!("value {index:08} {}", "x".repeat(index % 40)).into_bytes())
16870 .collect::<Vec<_>>();
16871
16872 let pass = |what: &str| {
16873 for (index, value) in wanted.iter().enumerate() {
16874 let len = dictionary.try_bytes_len_at(index).expect("read").expect("a value");
16875 assert_eq!(len, value.len(), "{what} has the wrong length at {index}");
16876 let bytes = dictionary.try_bytes_at(index).expect("read").expect("a value");
16877 assert_eq!(bytes, value.as_slice(), "{what} has the wrong value at {index}");
16878 }
16879 };
16880 pass("the first pass");
16881 pass("the second pass");
16882
16883 let lens = wanted.iter().map(|value| value.len() as i64).collect::<Vec<_>>();
16887 let mut whole = vec![0i64; wanted.len()];
16888 assert!(dictionary.try_bytes_lens(&mut whole).expect("read"), "the text answers whole");
16889 assert_eq!(whole, lens, "a vector of lengths answers what a length at a time answers");
16890 let codes = (0..4_000_u32).map(|row| (7 * (4_000 - row)) % 2_800).collect::<Vec<_>>();
16891 let coded = Vector::dictionary_over(codes.clone(), dictionary).expect("codes in range");
16892 let mut through = vec![0i64; codes.len()];
16893 assert!(coded.try_bytes_lens(&mut through).expect("read"), "the codes answer whole");
16894 for (row, &code) in codes.iter().enumerate() {
16895 assert_eq!(through[row], lens[code as usize], "row {row} reads code {code}");
16896 let one = coded.try_bytes_len_at(row).expect("read").expect("a value");
16897 assert_eq!(through[row], one as i64, "row {row} a row at a time");
16898 }
16899
16900 let fresh = Reader::open(&path).expect("valid directory");
16903 let untouched = fresh.dictionary(0).expect("read").expect("a string column has one");
16904 let few = vec![2_799_u32, 0, 1_024, 1_023, 511, 512];
16905 let coded = Vector::dictionary_over(few.clone(), untouched).expect("in range");
16906 let mut short = vec![0i64; few.len()];
16907 assert!(coded.try_bytes_lens(&mut short).expect("read"), "the codes answer whole");
16908 let expected = few.iter().map(|&code| lens[code as usize]).collect::<Vec<_>>();
16909 assert_eq!(short, expected, "the packed ends answer what the table answers");
16910 fs::remove_file(path).expect("remove scratch file");
16911 }
16912
16913 #[test]
16928 fn a_dictionary_reads_the_same_whether_its_blocks_say_where_they_are() {
16929 let spellings = (0..3_000)
16930 .map(|index| format!("value {index:08} {}", "y".repeat(index % 40)))
16931 .collect::<Vec<_>>();
16932 let mut read = Vec::new();
16933 for layout in ["outside", "inside", "behind"] {
16934 let mut dictionary = GlobalDictionary::new();
16935 for text in &spellings {
16936 dictionary.code(text).expect("a code for every spelling");
16937 }
16938 dictionary.finish_blocks().expect("the last block encodes");
16939 let order = dictionary.ranked(None).expect("a sorted order");
16940 let laid = |from: u64| {
16942 let mut at = from;
16943 dictionary
16944 .blocks
16945 .iter()
16946 .map(|block| {
16947 let place =
16948 Placed { start: at, length: block.len() as u64, hash: checksum(block) };
16949 at += block.len() as u64;
16950 place
16951 })
16952 .collect::<Vec<_>>()
16953 };
16954 let payload = dictionary.blocks.concat();
16955 let scattered = layout != "behind";
16956 let (bytes, encoded, offset, length) = if layout == "outside" {
16957 let mut bytes = vec![0; HEADER as usize];
16958 bytes.extend_from_slice(&payload);
16959 let encoded = encode_global_dictionary(&dictionary, &order, &laid(HEADER), true)
16960 .expect("an encoding");
16961 let offset = bytes.len() as u64;
16962 bytes.extend_from_slice(&encoded.index);
16963 bytes.extend_from_slice(&encoded.ranks);
16964 bytes.extend_from_slice(&encoded.grams);
16965 let length = encoded.index.len() + encoded.ranks.len() + encoded.grams.len();
16966 (bytes, encoded, offset, length)
16967 } else {
16968 let first = encode_global_dictionary(&dictionary, &order, &laid(0), scattered)
16971 .expect("an encoding");
16972 let body = (first.index.len() + first.ranks.len() + first.grams.len()) as u64;
16973 let encoded = encode_global_dictionary(&dictionary, &order, &laid(body), scattered)
16974 .expect("an encoding");
16975 let mut bytes = encoded.index.clone();
16976 bytes.extend_from_slice(&encoded.ranks);
16977 bytes.extend_from_slice(&encoded.grams);
16978 bytes.extend_from_slice(&payload);
16979 let length = bytes.len();
16980 (bytes, encoded, 0, length)
16981 };
16982 let path = path(&format!("blocks-{layout}"));
16983 fs::write(&path, &bytes).expect("the dictionary is written on its own");
16984 let file = Arc::new(File::open(&path).expect("it opens again"));
16985 let page = Page {
16986 offset,
16987 length: u32::try_from(length).expect("a test dictionary is small"),
16988 hash: checksum(&encoded.index),
16989 };
16990 let opened =
16991 open_global_dictionary(file, page, &LogicalType::Varchar, TEXT_KEEP_BUDGET)
16992 .expect("a dictionary laid out either way opens");
16993 let mut swept: Vec<Vec<u8>> = Vec::new();
16994 let mut at = 0;
16995 while at < opened.len() {
16996 at = opened
16997 .sweep_text(at, opened.len(), &mut |_index: usize, text: &[u8]| {
16998 swept.push(text.to_vec());
16999 Ok(())
17000 })
17001 .expect("a sweep reads");
17002 }
17003 fs::remove_file(&path).expect("clean up");
17004 read.push(swept);
17005 }
17006 let wanted =
17007 spellings.iter().map(|text| text.as_bytes().to_vec()).collect::<Vec<Vec<u8>>>();
17008 assert_eq!(read[0], wanted, "the blocks outside the page hold the values");
17009 assert_eq!(read[1], read[0], "the blocks inside the page hold the same values");
17010 assert_eq!(read[2], read[0], "the blocks behind one another hold the same values");
17011 }
17012
17013 #[test]
17021 fn a_dictionary_at_its_budget_sweeps_without_keeping() {
17022 let path = path("dictionary-budget");
17023 let spellings = (0..2_500)
17024 .map(|index| Value::Varchar(format!("value {index:08} {}", "y".repeat(index % 40))))
17025 .collect::<Vec<_>>();
17026 let mut writer =
17027 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
17028 .expect("new file");
17029 for part in spellings.chunks(1_024) {
17030 writer
17031 .append(
17032 &Chunk::new(vec![
17033 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
17034 ])
17035 .expect("one column"),
17036 )
17037 .expect("stripe written");
17038 }
17039 writer.finish().expect("commit");
17040
17041 let reader = Reader::open(&path).expect("valid directory");
17042 let page = reader.table.dictionaries[0].expect("a string column has one");
17043 let file = Arc::clone(&reader.file);
17044 let starved = open_global_dictionary(file, page, &LogicalType::Varchar, 0)
17045 .expect("a dictionary opens whatever it may keep");
17046
17047 let resting = starved.footprint();
17048 let mut swept: Vec<Vec<u8>> = Vec::new();
17049 let mut at = 0;
17050 while at < starved.len() {
17051 at = starved
17052 .sweep_text(at, starved.len(), &mut |_index: usize, text: &[u8]| {
17053 swept.push(text.to_vec());
17054 Ok(())
17055 })
17056 .expect("a sweep reads");
17057 }
17058 assert_eq!(swept.len(), spellings.len(), "a starved sweep still reads every value");
17059 assert_eq!(starved.footprint(), resting, "and keeps no block it decoded");
17060
17061 let generous = reader.dictionary(0).expect("read").expect("a string column has one");
17062 let read = (0..generous.len())
17063 .map(|code| generous.try_bytes_at(code).expect("read").expect("a value").to_vec())
17064 .collect::<Vec<_>>();
17065 assert_eq!(swept, read, "a starved sweep answers what a point read answers");
17066 fs::remove_file(path).expect("remove scratch file");
17067 }
17068
17069 #[test]
17070 fn damaged_membership_cannot_skip_a_string_page() {
17071 let path = path("damaged-membership");
17072 let mut writer = Writer::create(
17073 &path,
17074 "items",
17075 vec![
17076 Field::required("id", LogicalType::Integer),
17077 Field::new("text", LogicalType::Varchar),
17078 ],
17079 )
17080 .expect("new file");
17081 writer.append(&sample()).expect("stripe written");
17082 writer.finish().expect("commit");
17083
17084 let reader = Reader::open(&path).expect("valid directory");
17085 let membership = reader.table.stripes[0].memberships.get(1).expect("string membership");
17086 let mut file = OpenOptions::new().write(true).open(&path).expect("open membership page");
17087 file.seek(SeekFrom::Start(membership.offset)).expect("membership start");
17088 file.write_all(&[255]).expect("damage membership");
17089 let error = reader.skips_codes(0, 1, &[3]).expect_err("corruption must not skip rows");
17090 assert!(error.message().contains("membership page checksum differs"), "{error}");
17091 fs::remove_file(path).expect("remove scratch file");
17092 }
17093
17094 #[test]
17095 fn membership_delta_stream_is_sorted_exact_and_bounded() {
17096 let unique = unique_codes(&[900, 4, 4, 72, 9, u32::MAX]);
17097 assert_eq!(unique, [4, 9, 72, 900, u32::MAX]);
17098 let encoded = encode_membership(&unique);
17099 assert_eq!(
17100 decode_membership(&encoded).expect("valid membership"),
17101 [4, 9, 72, 900, u32::MAX]
17102 );
17103 let merged = merged_codes(vec![vec![4, 900], vec![9, 900, u32::MAX], vec![72]]);
17106 assert_eq!(merged, [4, 9, 72, 900, u32::MAX]);
17107 assert_eq!(
17108 decode_membership(&encode_membership(&merged)).expect("valid membership"),
17109 unique
17110 );
17111 assert!(decode_membership(&[1, 0x80]).is_err(), "a truncated varint is invalid");
17112 assert!(
17113 decode_membership(&[1, 0xff, 0xff, 0xff, 0xff, 0x10]).is_err(),
17114 "a value past u32 is invalid"
17115 );
17116 }
17117
17118 #[test]
17119 fn a_global_dictionary_may_be_larger_than_one_column_page() {
17120 let dictionary = Page {
17121 offset: HEADER,
17122 length: u32::try_from(MAX_PAGE + 1).expect("the page bound fits on disk"),
17123 hash: 0,
17124 };
17125 let table = Table {
17126 name: "items".to_owned(),
17127 fields: vec![Field::new("text", LogicalType::Varchar)],
17128 stripes: Vec::new(),
17129 rows: 0,
17130 dictionaries: vec![Some(dictionary)],
17131 dictionary_payloads: Vec::new(),
17132 demoted: Vec::new(),
17133 distincts: vec![None],
17134 frequencies: vec![None],
17135 pair_frequencies: Vec::new(),
17136 frequency_texts: Vec::new(),
17137 host_groups: None,
17138 clustering: None,
17139 constraints: Constraints::default(),
17140 generation: 1,
17141 sections: Vec::new(),
17142 };
17143 let directory = encode_directory(&table).expect("directory");
17144 let file_size = dictionary.offset + u64::from(dictionary.length) + 1;
17145
17146 let decoded = decode_directory(&directory, file_size).expect("large lazy dictionary");
17147 assert_eq!(decoded.dictionaries[0].expect("dictionary").length, dictionary.length);
17148 }
17149
17150 #[test]
17151 fn a_column_with_one_value_everywhere_costs_almost_nothing_a_row() {
17152 let path = path("constant-codes");
17153 let mut writer =
17154 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
17155 .expect("new file");
17156 let empty = vec![Value::Varchar(String::new()); 1024];
17157 for _ in 0..4 {
17158 let column = Vector::from_values(LogicalType::Varchar, &empty).expect("strings");
17159 writer.append(&Chunk::new(vec![column]).expect("one column")).expect("a part");
17160 }
17161 writer.finish().expect("commit");
17162
17163 let reader = Reader::open(&path).expect("valid directory");
17164 let pages = reader.layout().columns.first().expect("one column").pages;
17165 assert!(pages < 256, "{pages} bytes of pages for 4,096 rows of one value");
17169 let read = reader.read(3, &[0]).expect("the last part back");
17170 assert_eq!(read.value_at(0, 0), Value::Varchar(String::new()));
17171 assert_eq!(read.value_at(1023, 0), Value::Varchar(String::new()));
17172 fs::remove_file(path).expect("remove scratch file");
17173 }
17174
17175 #[test]
17176 fn a_cascade_value_too_wide_for_its_column_is_refused_rather_than_cut() {
17177 let over = integer::encode(&[i64::from(i32::MAX) + 1]).expect("a chunk");
17180 let error = cascade(&LogicalType::Integer, &over, 1).expect_err("a page that disagrees");
17181 assert!(format!("{error}").contains("not of its type"), "{error}");
17182 let low = integer::encode(&[i64::MIN]).expect("a chunk");
17183 assert!(cascade(&LogicalType::BigInt, &low, 1).is_ok(), "bigint holds all of i64");
17184 let zero = integer::encode(&[0]).expect("a chunk");
17185 assert!(cascade(&LogicalType::Varchar, &zero, 1).is_err(), "strings are not integers");
17186 }
17187
17188 #[test]
17189 fn a_code_stream_the_cascade_cannot_shrink_is_left_alone() {
17190 let mut state: u32 = 0x9e37_79b9;
17194 let spread: Vec<u32> = (0..1024)
17195 .map(|_| {
17196 state ^= state << 13;
17197 state ^= state >> 17;
17198 state ^= state << 5;
17199 state
17200 })
17201 .collect();
17202 assert_eq!(encoded_codes(&spread).expect("no failure"), None);
17203 let near: Vec<u32> = (0..1024).collect();
17204 let coded = encoded_codes(&near).expect("no failure").expect("counting up is packable");
17205 assert!(coded.len() < near.len() * 4, "{} bytes for a run of 1,024", coded.len());
17206 }
17207
17208 #[test]
17214 fn two_writes_of_the_same_rows_give_the_same_bytes() {
17215 fn written(path: &PathBuf) {
17216 let fields = (0..40)
17217 .map(|column| {
17218 let ty =
17219 if column % 4 == 0 { LogicalType::Varchar } else { LogicalType::BigInt };
17220 Field::new(format!("c{column}"), ty)
17221 })
17222 .collect::<Vec<_>>();
17223 let mut writer = Writer::create(path, "wide", fields).expect("new file");
17224 for part in 0..70_u64 {
17225 let columns = (0..40)
17226 .map(|column| {
17227 let values = (0..64_u64)
17228 .map(|row| {
17229 let seed = part.wrapping_mul(31).wrapping_add(row);
17230 if column % 4 == 0 {
17231 Value::Varchar(format!("v{}", seed % 17))
17232 } else {
17233 Value::BigInt(i64::try_from(seed % 97).expect("small"))
17234 }
17235 })
17236 .collect::<Vec<_>>();
17237 let ty = if column % 4 == 0 {
17238 LogicalType::Varchar
17239 } else {
17240 LogicalType::BigInt
17241 };
17242 Vector::from_values(ty, &values).expect("a column")
17243 })
17244 .collect::<Vec<_>>();
17245 writer.append(&Chunk::new(columns).expect("forty columns")).expect("a part");
17246 }
17247 writer.finish().expect("commit");
17248 }
17249
17250 let first = path("repeatable-one");
17251 let second = path("repeatable-two");
17252 written(&first);
17253 written(&second);
17254 let left = fs::read(&first).expect("the first file");
17255 let right = fs::read(&second).expect("the second file");
17256 assert_eq!(left.len(), right.len(), "two writes of the same rows differ in length");
17257 assert!(left == right, "two writes of the same rows differ in their bytes");
17258
17259 let reader = Reader::open(&first).expect("valid directory");
17262 assert_eq!(reader.table().rows(), 70 * 64);
17263 let read = reader.read(0, &[0, 1]).expect("the first part back");
17264 assert_eq!(read.value_at(0, 0), Value::Varchar("v0".to_owned()));
17265 assert_eq!(read.value_at(0, 1), Value::BigInt(0));
17266 fs::remove_file(first).expect("remove scratch file");
17267 fs::remove_file(second).expect("remove scratch file");
17268 }
17269
17270 fn three_tables(path: &PathBuf) {
17272 let writer = Writer::create(
17273 path,
17274 "region",
17275 vec![
17276 Field::new("r_key", LogicalType::Integer),
17277 Field::new("r_name", LogicalType::Varchar),
17278 ],
17279 )
17280 .expect("new file");
17281 let mut writer = writer;
17282 writer
17283 .append(
17284 &Chunk::new(vec![
17285 Vector::from_values(
17286 LogicalType::Integer,
17287 &[Value::Integer(0), Value::Integer(1)],
17288 )
17289 .expect("keys"),
17290 Vector::from_values(
17291 LogicalType::Varchar,
17292 &[Value::Varchar("AFRICA".to_owned()), Value::Varchar("ASIA".to_owned())],
17293 )
17294 .expect("names"),
17295 ])
17296 .expect("two columns"),
17297 )
17298 .expect("a part");
17299 let mut writer = writer
17300 .next("empty", vec![Field::new("nothing", LogicalType::BigInt)])
17301 .expect("a second table");
17302 writer
17303 .append(
17304 &Chunk::new(vec![
17305 Vector::from_values(LogicalType::BigInt, &[Value::BigInt(7)]).expect("a row"),
17306 ])
17307 .expect("one column"),
17308 )
17309 .expect("a part");
17310 let mut writer =
17311 writer.next("wide", vec![Field::new("n", LogicalType::BigInt)]).expect("a third table");
17312 for part in 0..70_i64 {
17313 let values = (0..64).map(|row| Value::BigInt(part * 64 + row)).collect::<Vec<_>>();
17314 writer
17315 .append(
17316 &Chunk::new(vec![
17317 Vector::from_values(LogicalType::BigInt, &values).expect("a column"),
17318 ])
17319 .expect("one column"),
17320 )
17321 .expect("a part");
17322 }
17323 writer.finish().expect("commit");
17324 }
17325
17326 #[test]
17327 fn three_tables_in_one_file_read_back_by_name() {
17328 let file = path("three-tables");
17329 three_tables(&file);
17330 let catalog = Catalog::open(&file).expect("a committed catalog");
17331 assert_eq!(catalog.names().collect::<Vec<_>>(), ["region", "empty", "wide"]);
17332
17333 let region = catalog.table("region").expect("the first table");
17334 assert_eq!(region.table().rows(), 2);
17335 assert_eq!(
17336 region.read(0, &[1]).expect("names").value_at(1, 0),
17337 Value::Varchar("ASIA".to_owned())
17338 );
17339
17340 let wide = catalog.table("wide").expect("the third table");
17341 assert_eq!(wide.table().rows(), 70 * 64);
17342 assert_eq!(wide.read(0, &[0]).expect("the first part").value_at(0, 0), Value::BigInt(0));
17343
17344 let empty = catalog.table("empty").expect("the second table");
17347 assert_eq!(empty.table().rows(), 1);
17348 assert_eq!(empty.read(0, &[0]).expect("the row").value_at(0, 0), Value::BigInt(7));
17349
17350 fs::remove_file(file).expect("remove scratch file");
17351 }
17352
17353 #[test]
17354 fn a_name_the_file_does_not_hold_is_an_error_rather_than_the_first_table() {
17355 let file = path("three-tables-missing");
17356 three_tables(&file);
17357 let catalog = Catalog::open(&file).expect("a committed catalog");
17358 let error = catalog.table("nation").expect_err("no such table");
17359 assert!(error.message().contains("nation"), "{}", error.message());
17360 fs::remove_file(file).expect("remove scratch file");
17361 }
17362
17363 #[test]
17364 fn a_file_of_three_tables_will_not_open_as_one() {
17365 let file = path("three-tables-unnamed");
17366 three_tables(&file);
17367 let error = Reader::open(&file).expect_err("more than one table");
17368 assert!(error.message().contains("more than one table"), "{}", error.message());
17369 fs::remove_file(file).expect("remove scratch file");
17370 }
17371
17372 #[test]
17374 fn decimals_of_every_storage_width_round_trip() {
17375 let file = path("decimals");
17376 let widths = [(4_u8, 2_u8), (9, 2), (18, 4), (38, 6)];
17377 let fields = widths
17378 .iter()
17379 .enumerate()
17380 .map(|(index, (width, scale))| {
17381 Field::new(
17382 format!("d{index}"),
17383 LogicalType::decimal(*width, *scale).expect("a decimal type"),
17384 )
17385 })
17386 .collect::<Vec<_>>();
17387 let mut writer = Writer::create(&file, "money", fields).expect("new file");
17388 let rows: [i128; 3] = [-1234, 0, 999];
17389 let columns = widths
17390 .iter()
17391 .map(|(width, scale)| {
17392 let values = rows
17393 .iter()
17394 .map(|unscaled| Value::Decimal {
17395 unscaled: *unscaled,
17396 width: *width,
17397 scale: *scale,
17398 })
17399 .collect::<Vec<_>>();
17400 Vector::from_values(
17401 LogicalType::decimal(*width, *scale).expect("a decimal type"),
17402 &values,
17403 )
17404 .expect("a decimal column")
17405 })
17406 .collect::<Vec<_>>();
17407 writer.append(&Chunk::new(columns).expect("four columns")).expect("a part");
17408 writer.finish().expect("commit");
17409
17410 let reader = Reader::open(&file).expect("a committed file");
17411 for (index, (width, scale)) in widths.iter().enumerate() {
17412 assert_eq!(
17413 reader.table().fields()[index].ty,
17414 LogicalType::decimal(*width, *scale).expect("a decimal type"),
17415 "column {index} came back as another type"
17416 );
17417 let column = reader.read(0, &[index]).expect("the column");
17418 for (row, unscaled) in rows.iter().enumerate() {
17419 assert_eq!(
17420 column.value_at(row, 0),
17421 Value::Decimal { unscaled: *unscaled, width: *width, scale: *scale },
17422 "column {index} row {row}"
17423 );
17424 }
17425 }
17426 fs::remove_file(file).expect("remove scratch file");
17427 }
17428
17429 #[test]
17430 fn two_tables_of_one_name_are_refused_before_anything_is_committed() {
17431 let file = path("two-of-a-name");
17432 let writer = Writer::create(&file, "t", vec![Field::new("a", LogicalType::BigInt)])
17433 .expect("new file");
17434 let error = writer
17435 .next("t", vec![Field::new("a", LogicalType::BigInt)])
17436 .expect_err("the same name twice");
17437 assert!(error.message().contains("same name"), "{}", error.message());
17438 fs::remove_file(file).expect("remove scratch file");
17439 }
17440
17441 #[test]
17442 fn integer_tally_counts_encoded_rows_and_declines_null_parts() {
17443 let file = path("integer-tally");
17444 let mut writer =
17445 Writer::create(&file, "events", vec![Field::new("source", LogicalType::SmallInt)])
17446 .expect("new file");
17447 let mut values = vec![Value::SmallInt(0); 1024];
17448 values[7] = Value::SmallInt(3);
17449 values[99] = Value::SmallInt(-2);
17450 values[1001] = Value::SmallInt(3);
17451 let column = Vector::from_values(LogicalType::SmallInt, &values).expect("integer values");
17452 writer.append(&Chunk::new(vec![column]).expect("one column")).expect("first part");
17453 values[0] = Value::Null;
17454 let column = Vector::from_values(LogicalType::SmallInt, &values).expect("nullable values");
17455 writer.append(&Chunk::new(vec![column]).expect("one column")).expect("second part");
17456 writer.finish().expect("commit");
17457
17458 let reader = Reader::open(&file).expect("read file");
17459 assert_eq!(
17460 reader.integer_tally(0, 0).expect("valid part"),
17461 Some(vec![(-2, 1), (0, 1021), (3, 2)])
17462 );
17463 assert!(reader.integer_tally(1, 0).expect("valid null part").is_none());
17464 let catalog = Catalog::open(&file).expect("catalog");
17465 assert_eq!(
17466 catalog.integer_tally("events", 0).expect("nullable column"),
17467 Some(vec![(-2, 2), (0, 2041), (3, 4)])
17468 );
17469 fs::remove_file(file).expect("remove scratch file");
17470 }
17471
17472 #[test]
17473 fn catalog_tallies_one_integer_column_without_opening_the_whole_table() {
17474 let file = path("catalog-integer-tally");
17475 let mut writer = Writer::create(
17476 &file,
17477 "events",
17478 vec![
17479 Field::new("noise", LogicalType::SmallInt),
17480 Field::new("source", LogicalType::SmallInt),
17481 ],
17482 )
17483 .expect("new file");
17484 let noise = vec![Value::SmallInt(9); 1024];
17485 let mut source = vec![Value::SmallInt(0); 1024];
17486 source[7] = Value::SmallInt(3);
17487 source[99] = Value::SmallInt(-2);
17488 let chunk = Chunk::new(vec![
17489 Vector::from_values(LogicalType::SmallInt, &noise).expect("noise"),
17490 Vector::from_values(LogicalType::SmallInt, &source).expect("source"),
17491 ])
17492 .expect("two columns");
17493 writer.append(&chunk).expect("append");
17494 writer.finish().expect("commit");
17495
17496 let catalog = Catalog::open(&file).expect("catalog");
17497 assert_eq!(
17498 catalog.integer_tally("events", 1).expect("selected column"),
17499 Some(vec![(-2, 1), (0, 1022), (3, 1)])
17500 );
17501 assert_eq!(
17502 catalog.integer_tally("events", 0).expect("other column"),
17503 Some(vec![(9, 1024)])
17504 );
17505 fs::remove_file(file).expect("remove scratch file");
17506 }
17507
17508 #[test]
17509 fn opening_the_catalog_reads_no_table_directory() {
17510 let file = path("catalog-only");
17511 three_tables(&file);
17512 let catalog = Catalog::open(&file).expect("a committed catalog");
17513 assert_eq!(catalog.opening.reads, 2, "opening the catalog read more than the slot");
17516 assert_eq!(catalog.names().len(), 3);
17517 fs::remove_file(file).expect("remove scratch file");
17518 }
17519
17520 #[test]
17531 fn the_checksum_answers_what_it_has_always_answered() {
17532 let bytes: Vec<u8> =
17533 (0..1000_u32).map(|at| (at.wrapping_mul(31).wrapping_add(7) % 251) as u8).collect();
17534 for (length, expected) in [
17535 (0, 0xef46_db37_51d8_e999),
17536 (1, 0xa96c_7f0c_e858_bbb7),
17537 (3, 0x56e6_9576_32a4_87f9),
17538 (4, 0xc60d_15b1_e3ff_8f04),
17539 (5, 0x8088_1585_8624_dd4e),
17540 (7, 0xafbe_fc3d_6c6f_9a8e),
17541 (8, 0x3da5_c7aa_2696_83e0),
17542 (9, 0x465e_c429_b13c_3892),
17543 (15, 0xdee8_9d8a_065a_6233),
17544 (16, 0x1330_489a_7767_9c80),
17545 (31, 0x3391_303d_485e_846e),
17546 (32, 0x40b7_aff7_5d45_bbc8),
17547 (33, 0x4997_cae4_951c_17a5),
17548 (39, 0x5807_28fd_5c14_5739),
17549 (40, 0xf95c_f6f5_c08a_3d3b),
17550 (63, 0x2944_b4da_fc69_b206),
17551 (64, 0xbb76_f6ef_19bd_5a1b),
17552 (65, 0x814e_0c65_4a9f_d640),
17553 (127, 0x00de_aab1_31cf_f89b),
17554 (1000, 0x9e33_00c1_cde3_c58d),
17555 ] {
17556 assert_eq!(checksum(&bytes[..length]), expected, "the checksum of {length} bytes");
17557 }
17558 assert_eq!(checksum(b"the quick brown fox jumps over the lazy dog"), 0xed71_4233_c5a9_a792);
17559 }
17560 #[test]
17567 fn a_declared_order_comes_back_out_of_the_file() {
17568 let path = path("clustered");
17569 let shipped = vec![
17570 Field::new("key", LogicalType::BigInt),
17571 Field::new("line", LogicalType::Integer),
17572 Field::new("shipdate", LogicalType::Date),
17573 ];
17574 let plain = vec![Field::new("a", LogicalType::Integer)];
17575 let stage_zero = Clustering::new(vec![2, 0, 1], Width::Month, &shipped).expect("valid");
17576
17577 let mut writer = Writer::create(&path, "lineitem", shipped)
17578 .expect("new file")
17579 .declare(stage_zero.clone())
17580 .expect("the columns are the table's");
17581 let column = |ty: LogicalType, values: &[Value]| {
17582 Vector::from_values(ty, values).expect("the values match the type")
17583 };
17584 writer
17585 .append(
17586 &Chunk::new(vec![
17587 column(
17588 LogicalType::BigInt,
17589 &[Value::BigInt(0), Value::BigInt(1), Value::BigInt(2), Value::BigInt(3)],
17590 ),
17591 column(
17592 LogicalType::Integer,
17593 &[
17594 Value::Integer(1),
17595 Value::Integer(1),
17596 Value::Integer(1),
17597 Value::Integer(1),
17598 ],
17599 ),
17600 column(
17601 LogicalType::Date,
17602 &[Value::Date(0), Value::Date(1), Value::Date(2), Value::Date(3)],
17603 ),
17604 ])
17605 .expect("three columns"),
17606 )
17607 .expect("four rows");
17608 let mut writer = writer.next("nation", plain).expect("a second table");
17609 writer
17610 .append(
17611 &Chunk::new(vec![column(LogicalType::Integer, &[Value::Integer(7)])])
17612 .expect("one column"),
17613 )
17614 .expect("one row");
17615 writer.finish().expect("commit");
17616
17617 let catalog = Catalog::open(&path).expect("reopen");
17618 let lineitem = catalog.table("lineitem").expect("the clustered table");
17619 assert_eq!(lineitem.table().clustering(), Some(&stage_zero));
17620 let nation = catalog.table("nation").expect("the plain table");
17621 assert_eq!(nation.table().clustering(), None, "nobody declared one here");
17622
17623 assert_eq!(lineitem.table().rows(), 4);
17626 assert_eq!(nation.table().rows(), 1);
17627 fs::remove_file(&path).ok();
17628 }
17629
17630 #[test]
17632 fn a_declaration_off_the_end_of_the_table_never_reaches_the_file() {
17633 let path = path("clustered-bad");
17634 let writer = Writer::create(&path, "items", vec![Field::new("a", LogicalType::Integer)])
17635 .expect("new file");
17636 let four =
17637 (0..4).map(|at| Field::new(format!("c{at}"), LogicalType::Integer)).collect::<Vec<_>>();
17638 let wrong = Clustering::new(vec![3], Width::Exact, &four).expect("valid against four");
17639 assert!(writer.declare(wrong).is_err(), "the table has one column, not four");
17640 fs::remove_file(&path).ok();
17641 }
17642
17643 #[test]
17649 fn blocks_handed_out_and_given_back_out_of_order_are_the_blocks_encoded_in_place() {
17650 let values = (0..PAYLOAD_SAMPLE_BLOCKS * TEXT_PAYLOAD_VALUES * 2 + 100)
17651 .map(|at| format!("http://example{}.test/page/{at:06}", at % 7))
17652 .collect::<Vec<_>>();
17653 let filled = || {
17654 let mut dictionary = GlobalDictionary::new();
17655 for value in &values {
17656 dictionary.code(value).expect("a code for every value");
17657 }
17658 dictionary.settle().expect("a shape");
17659 dictionary
17660 };
17661 let mut in_place = filled();
17662 in_place.finish_blocks().expect("every block encodes");
17663
17664 let mut handed = filled();
17665 let out = handed.hand_out(3);
17666 assert_eq!(out.len(), PAYLOAD_SAMPLE_BLOCKS * 2, "every sealed block goes out");
17667 assert!(handed.waiting.is_empty(), "and none is left to be encoded under the lock");
17668 for job in out.iter().rev() {
17669 assert_eq!(job.place().0, 3, "a block goes back to the column it came from");
17670 handed.take_back(job.place().1, job.encode().expect("encodes")).expect("taken back");
17671 }
17672 assert!(handed.early.is_empty(), "nothing is waiting on a gap");
17673 handed.finish_blocks().expect("the last block encodes");
17674
17675 assert_eq!(handed.blocks, in_place.blocks, "the same blocks in the same order");
17676 assert_eq!(handed.grams, in_place.grams, "with the same signatures");
17677 }
17678
17679 #[test]
17681 fn a_block_given_back_twice_is_refused() {
17682 let mut dictionary = GlobalDictionary::new();
17683 for at in 0..PAYLOAD_SAMPLE_BLOCKS * TEXT_PAYLOAD_VALUES {
17684 dictionary.code(&format!("value {at}")).expect("a code");
17685 }
17686 dictionary.settle().expect("a shape");
17687 let out = dictionary.hand_out(0);
17688 let last = out.last().expect("blocks went out");
17689 let at = last.place().1;
17690 dictionary.take_back(at, last.encode().expect("encodes")).expect("taken back once");
17691 assert!(dictionary.take_back(at, last.encode().expect("encodes")).is_err());
17692 }
17693
17694 #[test]
17700 fn the_dictionary_order_is_the_byte_order_however_deep_the_values_agree() {
17701 let mut values = vec![String::new(), "http://".to_owned()];
17702 for host in 0..7 {
17703 for path in 0..30 {
17704 values.push(format!("http://example{host}.test/page/{path:04}/index.html"));
17705 values.push(format!("http://example{host}.test/page/{path:04}"));
17706 }
17707 }
17708 values.push("http://example0.test/page/0000/index.htmlx".to_owned());
17709
17710 let mut dictionary = GlobalDictionary::new();
17711 for value in &values {
17712 dictionary.code(value).expect("a code for every value");
17713 }
17714 dictionary.finish_blocks().expect("the last block encodes");
17715 let ranked = dictionary.ranked(None).expect("a sorted order");
17716 assert_eq!(ranked.len(), values.len(), "one entry a distinct value");
17717
17718 let spellings = dictionary_values(&dictionary);
17719 let seen = ranked
17720 .iter()
17721 .map(|&(_, code)| {
17722 String::from_utf8(spellings[code as usize].clone()).expect("text in, text out")
17723 })
17724 .collect::<Vec<_>>();
17725 let mut wanted = values.clone();
17726 wanted.sort_unstable();
17727 assert_eq!(seen, wanted, "the order is the order the bytes give");
17728
17729 for &(carried, code) in &ranked {
17730 let value = &spellings[code as usize];
17731 assert_eq!(carried, head(value), "the head belongs to the value it is filed with");
17732 }
17733 }
17734
17735 #[test]
17740 fn the_commonest_entries_are_the_ones_a_full_sort_would_have_kept() {
17741 let entry =
17742 |value: u32, count: u64| FrequencyEntry { value: FrequencyValue::Code(value), count };
17743 let mut all = (0..FREQUENCY_ENTRIES as u32 * 3)
17744 .map(|code| entry(code, u64::from(code % 7) + 1))
17745 .collect::<Vec<_>>();
17746 all.push(FrequencyEntry { value: FrequencyValue::Null, count: 4 });
17747
17748 let mut sorted = all.clone();
17749 sorted.sort_unstable_by(|left, right| {
17750 right.count.cmp(&left.count).then_with(|| frequency_order(left.value, right.value))
17751 });
17752 let wanted_omitted = sorted[FREQUENCY_ENTRIES].count;
17753 sorted.truncate(FREQUENCY_ENTRIES);
17754
17755 let mut picked = all.clone();
17756 let omitted = keep_most_frequent(&mut picked);
17757 assert_eq!(omitted, wanted_omitted, "the largest count that did not make the cut");
17758 assert_eq!(picked.len(), FREQUENCY_ENTRIES, "the cut is where it says it is");
17759 assert!(
17760 picked
17761 .iter()
17762 .zip(&sorted)
17763 .all(|(one, two)| one.value == two.value && one.count == two.count),
17764 "the same entries in the same order"
17765 );
17766
17767 let mut short = all[..FREQUENCY_ENTRIES - 1].to_vec();
17768 let omitted = keep_most_frequent(&mut short);
17769 assert_eq!(omitted, 0, "nothing is omitted when everything fits");
17770 assert!(short.windows(2).all(|pair| pair[0].count >= pair[1].count), "still in order");
17771 }
17772
17773 #[test]
17775 fn a_short_dictionary_sorts_without_a_bucketing_pass() {
17776 let empty = GlobalDictionary::new();
17777 assert!(empty.ranked(None).expect("an empty order").is_empty(), "nothing in, nothing out");
17778
17779 let mut dictionary = GlobalDictionary::new();
17780 for value in ["pear", "apple", "", "apples", "app"] {
17781 dictionary.code(value).expect("a code for every value");
17782 }
17783 dictionary.finish_blocks().expect("the one block encodes");
17784 let spellings = dictionary_values(&dictionary);
17785 let seen = dictionary
17786 .ranked(None)
17787 .expect("a sorted order")
17788 .iter()
17789 .map(|&(_, code)| spellings[code as usize].clone())
17790 .collect::<Vec<_>>();
17791 let wanted: Vec<Vec<u8>> =
17792 [&b""[..], b"app", b"apple", b"apples", b"pear"].iter().map(|v| v.to_vec()).collect();
17793 assert_eq!(seen, wanted, "shorter first where one runs out inside another");
17794 }
17795
17796 #[test]
17799 fn a_demoted_dictionary_holds_less_and_takes_no_more_values() {
17800 let profile = LoadProfile::begin("demoted");
17801 let mut dictionary = GlobalDictionary::new();
17802 for value in 0..50_000 {
17803 dictionary.code(&format!("https://example.com/page/{value}")).expect("a code");
17804 }
17805 let (_, grown) = dictionary.recharge(Some(&profile));
17806 assert_eq!(profile.held(), grown, "the profile holds what the dictionary does");
17807
17808 dictionary.demote();
17809 let (before, after) = dictionary.recharge(Some(&profile));
17810 assert_eq!(before, grown);
17811 assert!(after < grown - grown / 4, "the lookup is let go of: {after} of {grown}");
17814 assert_eq!(profile.held(), after, "the profile was told about the drop");
17815 assert!(dictionary.code("one more").is_err(), "a demoted dictionary takes no values");
17816
17817 dictionary.demote();
17818 assert_eq!(
17819 dictionary.recharge(Some(&profile)),
17820 (after, after),
17821 "demoting twice is a no-op"
17822 );
17823 assert_eq!(dictionary.values(), 50_000, "the values coded before stay");
17824 }
17825}